"""Normalizes Prometheus alerts into the alert kinds the CX runbooks cover.""" from __future__ import annotations import datetime as dt import hashlib import re from dataclasses import dataclass, field from typing import Any, Optional EMOJI_RE = re.compile(r":[a-z0-9_+\-]+:") THRESHOLD_RE = re.compile(r"greater than\s*(\d+)\s*min", re.I) ORG_RE = re.compile(r"^\s*(\d+)\s*-\s*(.*)$") NONE_VALUES = {"", "none", "null", "unknown", "n/a"} # Alerts excluded outright. "Exists in Infrahub but does not exist in OpenStack" # is built as `Resources unless on(openstack_id) openstack_nova_server_status`, # and that right-hand metric is currently empty - so every Infrahub VM matches # and the alert fires thousands of times. It is a monitoring fault, not a queue # of work, so it never reaches the UI. EXCLUDED_ALERTNAMES = frozenset({ "Exists in Infrahub but does not exist in OpenStack", }) # Rule files whose alerts belong to CX. Everything else is infrastructure. CX_RULE_FILES = ("infrahub-rules",) NODE_RULE_FILE = "node-exporter-rules.yml" # The order CX wants to work the queue in. FOCUS_ORDER = ( "rogue_vm", "duplicate_ip", "total_gpus", "hibernating", "creating", "shutoff", "deleting", "error", "restoring", "rebooting", "build", "orphan_vm", "status_mismatch", ) # Priority and estimated time to resolve, from the "Infrahub Errors # Remediation" alert-conditions and runbook tables. KIND_META: dict[str, dict[str, str]] = { "error": {"title": "Instance in ERROR state", "priority": "LOW-HIGH", "ettr": "5-30 min", "delay": "none"}, "deleting": {"title": "Instance in DELETING state", "priority": "LOW", "ettr": "5-15 min", "delay": "30 min"}, "shutoff": {"title": "Instance in SHUTOFF state", "priority": "LOW", "ettr": "5-10 min", "delay": "30 min"}, "hibernating": {"title": "Instance in HIBERNATING state", "priority": "HIGH", "ettr": "5-30 min", "delay": "30 min"}, "creating": {"title": "Instance in CREATING state", "priority": "MEDIUM", "ettr": "5-15 min", "delay": "30 min"}, "restoring": {"title": "Instance in RESTORING state", "priority": "HIGH", "ettr": "5-15 min", "delay": "30 min"}, "rebooting": {"title": "Instance in REBOOTING state", "priority": "HIGH", "ettr": "5-15 min", "delay": "30 min"}, "build": {"title": "Instance in BUILD state", "priority": "MEDIUM", "ettr": "5-15 min", "delay": "30 min"}, "rogue_vm": {"title": "Suspected Rogue VM", "priority": "HIGH", "ettr": "5-30 min", "delay": "4 hours"}, "duplicate_ip": {"title": "Duplicated IPs", "priority": "HIGH", "ettr": "5-15 min", "delay": "10 min"}, "total_gpus": {"title": "Problem with Total GPUs in a System", "priority": "HIGH", "ettr": "5-15 min", "delay": "5 min"}, "orphan_vm": {"title": "Suspected Orphan VM", "priority": "HIGH", "ettr": "5-30 min", "delay": "4 hours"}, "status_mismatch": {"title": "Infrahub/OpenStack status mismatch", "priority": "HIGH", "ettr": "5-30 min", "delay": "30 min"}, } STATE_KINDS = ("error", "deleting", "shutoff", "hibernating", "creating", "restoring", "rebooting", "build") def clean_alertname(name: str) -> str: """Strip the Slack emoji shortcodes Prometheus embeds in alert names.""" return EMOJI_RE.sub("", str(name or "")).strip() def classify(alertname: str) -> str: name = clean_alertname(alertname).lower() if alertname in EXCLUDED_ALERTNAMES or clean_alertname(alertname) in EXCLUDED_ALERTNAMES: return "excluded" if "rogue vm" in name: return "rogue_vm" if "orphan vm" in name: return "orphan_vm" if "duplicated ip" in name or "duplicate ip" in name: return "duplicate_ip" if "total gpus" in name: return "total_gpus" match = re.search(r"instance in (\w+) state", name) if match: state = match.group(1).lower() if state in STATE_KINDS: return state # The per-region cross-check rules, e.g. # "Openstack status=SHUTOFF and Infrahub status!=SHUTOFF in CANADA-1". if "failure of hibernation" in name: return "hibernating" if "openstack status=" in name and "infrahub status" in name: return "status_mismatch" return "other" def category(alertname: str, rule_file: str = "") -> str: """Which tab an alert belongs in: 'cx', 'node', or 'infra'.""" if any(token in rule_file for token in CX_RULE_FILES): return "cx" if rule_file == NODE_RULE_FILE: return "node" if rule_file: return "infra" # No rule metadata (e.g. a pasted alert): fall back to the classifier. return "cx" if classify(alertname) not in ("other", "excluded") else "infra" def _clean(value: Any) -> str: text = str(value or "").strip() return "" if text.lower() in NONE_VALUES else text def split_organization(value: str) -> tuple[str, str]: """Split the `organization` label ("8463 - Some Org") into id and name.""" match = ORG_RE.match(str(value or "")) if match: return match.group(1), match.group(2).strip() return "", _clean(value) def _parse_active_at(value: Any) -> Optional[dt.datetime]: text = str(value or "").strip() if not text: return None text = re.sub(r"(\.\d{1,6})\d*Z?$", r"\1", text.replace("Z", "+00:00")) if text.endswith("+00:00") is False and "+" not in text[10:]: text += "+00:00" try: return dt.datetime.fromisoformat(text) except ValueError: return None @dataclass class Alert: """One normalized Prometheus alert.""" kind: str alertname: str labels: dict[str, str] = field(default_factory=dict) annotations: dict[str, str] = field(default_factory=dict) state: str = "firing" active_at: Optional[dt.datetime] = None # Fields the runbooks key off. openstack_id: str = "" instance_name: str = "" host: str = "" region_label: str = "" region: str = "" status: str = "" floating_ip: str = "" flavor_name: str = "" flavor_gpu: str = "" org_id: str = "" org_name: str = "" contract_id: str = "" gpu_name: str = "" threshold_min: Optional[int] = None # Where the rule came from (from RuleIndex), and the screening result. rule_file: str = "" rule_group: str = "" for_seconds: int = 0 category: str = "cx" screen: dict[str, Any] = field(default_factory=dict) # How long the condition has actually held, recovered from ALERTS history. # activeAt alone is unreliable: a metric-pipeline dip resets it on every # live alert at once, which is why raw ages cluster on one timestamp. true_age_minutes: Optional[int] = None true_age_capped: bool = False @property def title(self) -> str: return KIND_META.get(self.kind, {}).get("title", clean_alertname(self.alertname)) @property def priority(self) -> str: return KIND_META.get(self.kind, {}).get("priority", "UNKNOWN") @property def ettr(self) -> str: return KIND_META.get(self.kind, {}).get("ettr", "unknown") @property def is_kubernetes(self) -> bool: """Per the general process: kube* instance names are likely K8s nodes.""" return self.instance_name.lower().startswith("kube") or "-minion-" in self.instance_name.lower() @property def age_minutes(self) -> Optional[int]: if not self.active_at: return None now = dt.datetime.now(dt.timezone.utc) return max(0, int((now - self.active_at).total_seconds() // 60)) @staticmethod def _duration_text(minutes: Optional[int]) -> str: if minutes is None: return "unknown" if minutes < 60: return f"{minutes}m" hours, mins = divmod(minutes, 60) if hours < 24: return f"{hours}h {mins}m" if mins else f"{hours}h" days, hours = divmod(hours, 24) return f"{days}d {hours}h" if hours else f"{days}d" @property def age_text(self) -> str: """Raw Prometheus activeAt duration.""" return self._duration_text(self.age_minutes) @property def effective_age_minutes(self) -> Optional[int]: """True condition duration where known, else the raw activeAt age.""" return self.true_age_minutes if self.true_age_minutes is not None else self.age_minutes @property def effective_age_text(self) -> str: text = self._duration_text(self.effective_age_minutes) if self.true_age_minutes is not None and self.true_age_capped: return f"{text}+" return text @property def age_is_reset(self) -> bool: """True when activeAt materially understates how long this has held.""" if self.true_age_minutes is None or self.age_minutes is None: return False return self.true_age_minutes - self.age_minutes > 60 @property def is_internal_org(self) -> bool: """Internal/test organizations are not customer-impacting.""" return "nexgencloud.com" in self.org_name.lower() @property def is_infra_owned(self) -> bool: """Platform-owned nodes (storage etc.) name themselves after their host.""" return bool(self.instance_name) and self.instance_name.lower() == self.host.lower() def fingerprint(self) -> str: basis = "|".join([ self.kind, self.openstack_id or self.instance_name or "", self.host, self.floating_ip, self.region, ]) return hashlib.sha1(basis.encode()).hexdigest()[:16] def to_json(self) -> dict[str, Any]: return { "id": self.fingerprint(), "kind": self.kind, "title": self.title, "alertname": clean_alertname(self.alertname), "raw_alertname": self.alertname, "state": self.state, "priority": self.priority, "ettr": self.ettr, "active_at": self.active_at.isoformat() if self.active_at else "", "age_minutes": self.age_minutes, "age_text": self.age_text, "true_age_minutes": self.true_age_minutes, "true_age_capped": self.true_age_capped, "effective_age_minutes": self.effective_age_minutes, "effective_age_text": self.effective_age_text, "age_is_reset": self.age_is_reset, "threshold_min": self.threshold_min, "rule_file": self.rule_file, "rule_group": self.rule_group, "for_seconds": self.for_seconds, "category": self.category, "screen": self.screen, "is_internal_org": self.is_internal_org, "is_infra_owned": self.is_infra_owned, "openstack_id": self.openstack_id, "instance_name": self.instance_name, "host": self.host, "region": self.region, "region_label": self.region_label, "status": self.status, "floating_ip": self.floating_ip, "flavor_name": self.flavor_name, "flavor_gpu": self.flavor_gpu, "gpu_name": self.gpu_name, "org_id": self.org_id, "org_name": self.org_name, "contract_id": self.contract_id, "is_kubernetes": self.is_kubernetes, "labels": self.labels, "annotations": self.annotations, } def map_region(region_label: str) -> str: """CANADA-1 -> ca1. Deliberately a local table rather than a call into CX-Tools: building the alert queue must not touch cxlib, because constructing a CX-Tools Config loads credentials and would pop a 1Password prompt just to list alerts. Kept in sync with SUPPORTED_REGIONS in cxlib/constants.py. """ fallback = { "canada-1": "ca1", "canada-2": "ca2", "us-1": "us1", "norway-1": "no1", "ca-1": "ca1", "ca-2": "ca2", "no-1": "no1", "ca1": "ca1", "ca2": "ca2", "us1": "us1", "no1": "no1", } return fallback.get(str(region_label or "").strip().lower(), "") def from_labels(labels: dict[str, str], annotations: Optional[dict[str, str]] = None, state: str = "firing", active_at: Any = None) -> Alert: labels = {str(k): str(v) for k, v in (labels or {}).items()} alertname = labels.get("alertname", "") kind = classify(alertname) region_label = _clean(labels.get("region")) org_id, org_name = split_organization(labels.get("organization", "")) threshold = THRESHOLD_RE.search(alertname) # For host-scoped alerts the `instance` label is the hypervisor; for # VM-scoped alerts it is the hypervisor too, or "Unknown" when the VM never # landed on a host. host = _clean(labels.get("instance")) alert = Alert( kind=kind, alertname=alertname, labels=labels, annotations={str(k): str(v) for k, v in (annotations or {}).items()}, state=str(state or "firing"), active_at=_parse_active_at(active_at), openstack_id=_clean(labels.get("openstack_id")), instance_name=_clean(labels.get("instance_name")), host=host, region_label=region_label, region=map_region(region_label) or _infer_region_from_host(host), status=_clean(labels.get("status")), floating_ip=_clean(labels.get("floating_ip")), flavor_name=_clean(labels.get("flavor_name")), flavor_gpu=_clean(labels.get("flavor_gpu")), gpu_name=_clean(labels.get("gpu_name")), org_id=org_id, org_name=org_name, contract_id=_clean(labels.get("contract_id")), threshold_min=int(threshold.group(1)) if threshold else None, ) return alert def _infer_region_from_host(host: str) -> str: match = re.match(r"^(ca1|ca2|no1|us1)-", str(host or "").strip(), re.I) return match.group(1).lower() if match else "" def from_prometheus(raw: dict[str, Any], rule_index: Any = None, true_age: Any = None) -> Alert: alert = from_labels( raw.get("labels") or {}, raw.get("annotations") or {}, state=str(raw.get("state") or "firing"), active_at=raw.get("activeAt"), ) meta = rule_index.get(alert.alertname) if rule_index is not None else {} if meta: alert.rule_file = str(meta.get("file") or "") alert.rule_group = str(meta.get("group") or "") alert.for_seconds = int(meta.get("for_seconds") or 0) alert.category = category(alert.alertname, alert.rule_file) if true_age is not None: alert.true_age_minutes, alert.true_age_capped = true_age.lookup(alert.labels) return alert def cx_relevant(alert: Alert) -> bool: """True for alert kinds the CX runbooks cover.""" return alert.kind in KIND_META and alert.kind != "excluded" def is_excluded(alert: Alert) -> bool: return alert.kind == "excluded" or alert.alertname in EXCLUDED_ALERTNAMES def sort_key(alert: Alert) -> tuple: """Newest first: a fresh alert is the one that still needs a decision. Long-running alerts sink to the bottom - they are either chronic and already ticketed, or noise nobody has silenced. Alerts with no known start time sort last rather than jumping to the top. Sorted on the true condition duration, not activeAt: a pipeline dip resets activeAt on every live alert at once, which would otherwise flatten the ordering into a single tie. """ age = alert.effective_age_minutes return (age if age is not None else 10**9, alert.title) def focus_rank(kind: str) -> int: try: return FOCUS_ORDER.index(kind) except ValueError: return len(FOCUS_ORDER) def group_alerts(items: list[Alert]) -> list[dict[str, Any]]: """Group alerts into collapsible sections, in CX's working order.""" buckets: dict[str, list[Alert]] = {} for alert in items: buckets.setdefault(alert.kind, []).append(alert) groups: list[dict[str, Any]] = [] for kind, members in buckets.items(): members.sort(key=sort_key) actionable = [a for a in members if a.screen.get("actionable", True)] groups.append({ "kind": kind, "title": KIND_META.get(kind, {}).get("title", kind), "priority": KIND_META.get(kind, {}).get("priority", "UNKNOWN"), "ettr": KIND_META.get(kind, {}).get("ettr", "unknown"), "total": len(members), "actionable": len(actionable), "noise": len(members) - len(actionable), "alerts": [a.to_json() for a in members], }) groups.sort(key=lambda g: (focus_rank(g["kind"]), -g["actionable"])) return groups