Files
cx-ui/triagelib/alerts.py
Parham Monfared a039e0b5fd CX Triage: alert diagnosis over the CX-Tools collectors
Read-only triage for the Infrahub error alerts. Pulls the Prometheus alert
queue, re-checks each alert's condition against live state to separate real
work from noise, diagnoses it using the CX runbooks, and drafts the customer
comms with contacts resolved from Infrahub.

Findings from validating against production:
- "Suspected Rogue VM" fires on spare GPU capacity, not rogue VMs: In_Use_Gpus
  equals the physical count on 71 of 75 firing hosts, so the rule reduces to
  "this host has a free GPU". Verified against OpenStack on 10 hosts.
- "Exists in Infrahub but does not exist in OpenStack" matches every VM because
  openstack_nova_server_status returns no series; excluded as a rule defect.
- Prometheus activeAt is reset several times a day by dips in the Resources
  metric, so alert ages are recovered from ALERTS history instead.

Takes ~2,650 firing alerts down to ~20 that need a decision.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-06 06:48:34 +01:00

422 lines
16 KiB
Python

"""Normalizes Prometheus alerts into the alert kinds the CX runbooks cover."""
from __future__ import annotations
import datetime as dt
import hashlib
import re
from dataclasses import dataclass, field
from typing import Any, Optional
EMOJI_RE = re.compile(r":[a-z0-9_+\-]+:")
THRESHOLD_RE = re.compile(r"greater than\s*(\d+)\s*min", re.I)
ORG_RE = re.compile(r"^\s*(\d+)\s*-\s*(.*)$")
NONE_VALUES = {"", "none", "null", "unknown", "n/a"}
# Alerts excluded outright. "Exists in Infrahub but does not exist in OpenStack"
# is built as `Resources unless on(openstack_id) openstack_nova_server_status`,
# and that right-hand metric is currently empty - so every Infrahub VM matches
# and the alert fires thousands of times. It is a monitoring fault, not a queue
# of work, so it never reaches the UI.
EXCLUDED_ALERTNAMES = frozenset({
"Exists in Infrahub but does not exist in OpenStack",
})
# Rule files whose alerts belong to CX. Everything else is infrastructure.
CX_RULE_FILES = ("infrahub-rules",)
NODE_RULE_FILE = "node-exporter-rules.yml"
# The order CX wants to work the queue in.
FOCUS_ORDER = (
"rogue_vm", "duplicate_ip", "total_gpus", "hibernating",
"creating", "shutoff", "deleting", "error",
"restoring", "rebooting", "build", "orphan_vm", "status_mismatch",
)
# Priority and estimated time to resolve, from the "Infrahub Errors
# Remediation" alert-conditions and runbook tables.
KIND_META: dict[str, dict[str, str]] = {
"error": {"title": "Instance in ERROR state", "priority": "LOW-HIGH", "ettr": "5-30 min", "delay": "none"},
"deleting": {"title": "Instance in DELETING state", "priority": "LOW", "ettr": "5-15 min", "delay": "30 min"},
"shutoff": {"title": "Instance in SHUTOFF state", "priority": "LOW", "ettr": "5-10 min", "delay": "30 min"},
"hibernating": {"title": "Instance in HIBERNATING state", "priority": "HIGH", "ettr": "5-30 min", "delay": "30 min"},
"creating": {"title": "Instance in CREATING state", "priority": "MEDIUM", "ettr": "5-15 min", "delay": "30 min"},
"restoring": {"title": "Instance in RESTORING state", "priority": "HIGH", "ettr": "5-15 min", "delay": "30 min"},
"rebooting": {"title": "Instance in REBOOTING state", "priority": "HIGH", "ettr": "5-15 min", "delay": "30 min"},
"build": {"title": "Instance in BUILD state", "priority": "MEDIUM", "ettr": "5-15 min", "delay": "30 min"},
"rogue_vm": {"title": "Suspected Rogue VM", "priority": "HIGH", "ettr": "5-30 min", "delay": "4 hours"},
"duplicate_ip": {"title": "Duplicated IPs", "priority": "HIGH", "ettr": "5-15 min", "delay": "10 min"},
"total_gpus": {"title": "Problem with Total GPUs in a System", "priority": "HIGH", "ettr": "5-15 min", "delay": "5 min"},
"orphan_vm": {"title": "Suspected Orphan VM", "priority": "HIGH", "ettr": "5-30 min", "delay": "4 hours"},
"status_mismatch": {"title": "Infrahub/OpenStack status mismatch", "priority": "HIGH", "ettr": "5-30 min", "delay": "30 min"},
}
STATE_KINDS = ("error", "deleting", "shutoff", "hibernating", "creating", "restoring", "rebooting", "build")
def clean_alertname(name: str) -> str:
"""Strip the Slack emoji shortcodes Prometheus embeds in alert names."""
return EMOJI_RE.sub("", str(name or "")).strip()
def classify(alertname: str) -> str:
name = clean_alertname(alertname).lower()
if alertname in EXCLUDED_ALERTNAMES or clean_alertname(alertname) in EXCLUDED_ALERTNAMES:
return "excluded"
if "rogue vm" in name:
return "rogue_vm"
if "orphan vm" in name:
return "orphan_vm"
if "duplicated ip" in name or "duplicate ip" in name:
return "duplicate_ip"
if "total gpus" in name:
return "total_gpus"
match = re.search(r"instance in (\w+) state", name)
if match:
state = match.group(1).lower()
if state in STATE_KINDS:
return state
# The per-region cross-check rules, e.g.
# "Openstack status=SHUTOFF and Infrahub status!=SHUTOFF in CANADA-1".
if "failure of hibernation" in name:
return "hibernating"
if "openstack status=" in name and "infrahub status" in name:
return "status_mismatch"
return "other"
def category(alertname: str, rule_file: str = "") -> str:
"""Which tab an alert belongs in: 'cx', 'node', or 'infra'."""
if any(token in rule_file for token in CX_RULE_FILES):
return "cx"
if rule_file == NODE_RULE_FILE:
return "node"
if rule_file:
return "infra"
# No rule metadata (e.g. a pasted alert): fall back to the classifier.
return "cx" if classify(alertname) not in ("other", "excluded") else "infra"
def _clean(value: Any) -> str:
text = str(value or "").strip()
return "" if text.lower() in NONE_VALUES else text
def split_organization(value: str) -> tuple[str, str]:
"""Split the `organization` label ("8463 - Some Org") into id and name."""
match = ORG_RE.match(str(value or ""))
if match:
return match.group(1), match.group(2).strip()
return "", _clean(value)
def _parse_active_at(value: Any) -> Optional[dt.datetime]:
text = str(value or "").strip()
if not text:
return None
text = re.sub(r"(\.\d{1,6})\d*Z?$", r"\1", text.replace("Z", "+00:00"))
if text.endswith("+00:00") is False and "+" not in text[10:]:
text += "+00:00"
try:
return dt.datetime.fromisoformat(text)
except ValueError:
return None
@dataclass
class Alert:
"""One normalized Prometheus alert."""
kind: str
alertname: str
labels: dict[str, str] = field(default_factory=dict)
annotations: dict[str, str] = field(default_factory=dict)
state: str = "firing"
active_at: Optional[dt.datetime] = None
# Fields the runbooks key off.
openstack_id: str = ""
instance_name: str = ""
host: str = ""
region_label: str = ""
region: str = ""
status: str = ""
floating_ip: str = ""
flavor_name: str = ""
flavor_gpu: str = ""
org_id: str = ""
org_name: str = ""
contract_id: str = ""
gpu_name: str = ""
threshold_min: Optional[int] = None
# Where the rule came from (from RuleIndex), and the screening result.
rule_file: str = ""
rule_group: str = ""
for_seconds: int = 0
category: str = "cx"
screen: dict[str, Any] = field(default_factory=dict)
# How long the condition has actually held, recovered from ALERTS history.
# activeAt alone is unreliable: a metric-pipeline dip resets it on every
# live alert at once, which is why raw ages cluster on one timestamp.
true_age_minutes: Optional[int] = None
true_age_capped: bool = False
@property
def title(self) -> str:
return KIND_META.get(self.kind, {}).get("title", clean_alertname(self.alertname))
@property
def priority(self) -> str:
return KIND_META.get(self.kind, {}).get("priority", "UNKNOWN")
@property
def ettr(self) -> str:
return KIND_META.get(self.kind, {}).get("ettr", "unknown")
@property
def is_kubernetes(self) -> bool:
"""Per the general process: kube* instance names are likely K8s nodes."""
return self.instance_name.lower().startswith("kube") or "-minion-" in self.instance_name.lower()
@property
def age_minutes(self) -> Optional[int]:
if not self.active_at:
return None
now = dt.datetime.now(dt.timezone.utc)
return max(0, int((now - self.active_at).total_seconds() // 60))
@staticmethod
def _duration_text(minutes: Optional[int]) -> str:
if minutes is None:
return "unknown"
if minutes < 60:
return f"{minutes}m"
hours, mins = divmod(minutes, 60)
if hours < 24:
return f"{hours}h {mins}m" if mins else f"{hours}h"
days, hours = divmod(hours, 24)
return f"{days}d {hours}h" if hours else f"{days}d"
@property
def age_text(self) -> str:
"""Raw Prometheus activeAt duration."""
return self._duration_text(self.age_minutes)
@property
def effective_age_minutes(self) -> Optional[int]:
"""True condition duration where known, else the raw activeAt age."""
return self.true_age_minutes if self.true_age_minutes is not None else self.age_minutes
@property
def effective_age_text(self) -> str:
text = self._duration_text(self.effective_age_minutes)
if self.true_age_minutes is not None and self.true_age_capped:
return f"{text}+"
return text
@property
def age_is_reset(self) -> bool:
"""True when activeAt materially understates how long this has held."""
if self.true_age_minutes is None or self.age_minutes is None:
return False
return self.true_age_minutes - self.age_minutes > 60
@property
def is_internal_org(self) -> bool:
"""Internal/test organizations are not customer-impacting."""
return "nexgencloud.com" in self.org_name.lower()
@property
def is_infra_owned(self) -> bool:
"""Platform-owned nodes (storage etc.) name themselves after their host."""
return bool(self.instance_name) and self.instance_name.lower() == self.host.lower()
def fingerprint(self) -> str:
basis = "|".join([
self.kind,
self.openstack_id or self.instance_name or "",
self.host,
self.floating_ip,
self.region,
])
return hashlib.sha1(basis.encode()).hexdigest()[:16]
def to_json(self) -> dict[str, Any]:
return {
"id": self.fingerprint(),
"kind": self.kind,
"title": self.title,
"alertname": clean_alertname(self.alertname),
"raw_alertname": self.alertname,
"state": self.state,
"priority": self.priority,
"ettr": self.ettr,
"active_at": self.active_at.isoformat() if self.active_at else "",
"age_minutes": self.age_minutes,
"age_text": self.age_text,
"true_age_minutes": self.true_age_minutes,
"true_age_capped": self.true_age_capped,
"effective_age_minutes": self.effective_age_minutes,
"effective_age_text": self.effective_age_text,
"age_is_reset": self.age_is_reset,
"threshold_min": self.threshold_min,
"rule_file": self.rule_file,
"rule_group": self.rule_group,
"for_seconds": self.for_seconds,
"category": self.category,
"screen": self.screen,
"is_internal_org": self.is_internal_org,
"is_infra_owned": self.is_infra_owned,
"openstack_id": self.openstack_id,
"instance_name": self.instance_name,
"host": self.host,
"region": self.region,
"region_label": self.region_label,
"status": self.status,
"floating_ip": self.floating_ip,
"flavor_name": self.flavor_name,
"flavor_gpu": self.flavor_gpu,
"gpu_name": self.gpu_name,
"org_id": self.org_id,
"org_name": self.org_name,
"contract_id": self.contract_id,
"is_kubernetes": self.is_kubernetes,
"labels": self.labels,
"annotations": self.annotations,
}
def map_region(region_label: str) -> str:
"""CANADA-1 -> ca1.
Deliberately a local table rather than a call into CX-Tools: building the
alert queue must not touch cxlib, because constructing a CX-Tools Config
loads credentials and would pop a 1Password prompt just to list alerts.
Kept in sync with SUPPORTED_REGIONS in cxlib/constants.py.
"""
fallback = {
"canada-1": "ca1", "canada-2": "ca2", "us-1": "us1", "norway-1": "no1",
"ca-1": "ca1", "ca-2": "ca2", "no-1": "no1",
"ca1": "ca1", "ca2": "ca2", "us1": "us1", "no1": "no1",
}
return fallback.get(str(region_label or "").strip().lower(), "")
def from_labels(labels: dict[str, str], annotations: Optional[dict[str, str]] = None,
state: str = "firing", active_at: Any = None) -> Alert:
labels = {str(k): str(v) for k, v in (labels or {}).items()}
alertname = labels.get("alertname", "")
kind = classify(alertname)
region_label = _clean(labels.get("region"))
org_id, org_name = split_organization(labels.get("organization", ""))
threshold = THRESHOLD_RE.search(alertname)
# For host-scoped alerts the `instance` label is the hypervisor; for
# VM-scoped alerts it is the hypervisor too, or "Unknown" when the VM never
# landed on a host.
host = _clean(labels.get("instance"))
alert = Alert(
kind=kind,
alertname=alertname,
labels=labels,
annotations={str(k): str(v) for k, v in (annotations or {}).items()},
state=str(state or "firing"),
active_at=_parse_active_at(active_at),
openstack_id=_clean(labels.get("openstack_id")),
instance_name=_clean(labels.get("instance_name")),
host=host,
region_label=region_label,
region=map_region(region_label) or _infer_region_from_host(host),
status=_clean(labels.get("status")),
floating_ip=_clean(labels.get("floating_ip")),
flavor_name=_clean(labels.get("flavor_name")),
flavor_gpu=_clean(labels.get("flavor_gpu")),
gpu_name=_clean(labels.get("gpu_name")),
org_id=org_id,
org_name=org_name,
contract_id=_clean(labels.get("contract_id")),
threshold_min=int(threshold.group(1)) if threshold else None,
)
return alert
def _infer_region_from_host(host: str) -> str:
match = re.match(r"^(ca1|ca2|no1|us1)-", str(host or "").strip(), re.I)
return match.group(1).lower() if match else ""
def from_prometheus(raw: dict[str, Any], rule_index: Any = None, true_age: Any = None) -> Alert:
alert = from_labels(
raw.get("labels") or {},
raw.get("annotations") or {},
state=str(raw.get("state") or "firing"),
active_at=raw.get("activeAt"),
)
meta = rule_index.get(alert.alertname) if rule_index is not None else {}
if meta:
alert.rule_file = str(meta.get("file") or "")
alert.rule_group = str(meta.get("group") or "")
alert.for_seconds = int(meta.get("for_seconds") or 0)
alert.category = category(alert.alertname, alert.rule_file)
if true_age is not None:
alert.true_age_minutes, alert.true_age_capped = true_age.lookup(alert.labels)
return alert
def cx_relevant(alert: Alert) -> bool:
"""True for alert kinds the CX runbooks cover."""
return alert.kind in KIND_META and alert.kind != "excluded"
def is_excluded(alert: Alert) -> bool:
return alert.kind == "excluded" or alert.alertname in EXCLUDED_ALERTNAMES
def sort_key(alert: Alert) -> tuple:
"""Newest first: a fresh alert is the one that still needs a decision.
Long-running alerts sink to the bottom - they are either chronic and
already ticketed, or noise nobody has silenced. Alerts with no known start
time sort last rather than jumping to the top.
Sorted on the true condition duration, not activeAt: a pipeline dip resets
activeAt on every live alert at once, which would otherwise flatten the
ordering into a single tie.
"""
age = alert.effective_age_minutes
return (age if age is not None else 10**9, alert.title)
def focus_rank(kind: str) -> int:
try:
return FOCUS_ORDER.index(kind)
except ValueError:
return len(FOCUS_ORDER)
def group_alerts(items: list[Alert]) -> list[dict[str, Any]]:
"""Group alerts into collapsible sections, in CX's working order."""
buckets: dict[str, list[Alert]] = {}
for alert in items:
buckets.setdefault(alert.kind, []).append(alert)
groups: list[dict[str, Any]] = []
for kind, members in buckets.items():
members.sort(key=sort_key)
actionable = [a for a in members if a.screen.get("actionable", True)]
groups.append({
"kind": kind,
"title": KIND_META.get(kind, {}).get("title", kind),
"priority": KIND_META.get(kind, {}).get("priority", "UNKNOWN"),
"ettr": KIND_META.get(kind, {}).get("ettr", "unknown"),
"total": len(members),
"actionable": len(actionable),
"noise": len(members) - len(actionable),
"alerts": [a.to_json() for a in members],
})
groups.sort(key=lambda g: (focus_rank(g["kind"]), -g["actionable"]))
return groups