Read-only triage for the Infrahub error alerts. Pulls the Prometheus alert queue, re-checks each alert's condition against live state to separate real work from noise, diagnoses it using the CX runbooks, and drafts the customer comms with contacts resolved from Infrahub. Findings from validating against production: - "Suspected Rogue VM" fires on spare GPU capacity, not rogue VMs: In_Use_Gpus equals the physical count on 71 of 75 firing hosts, so the rule reduces to "this host has a free GPU". Verified against OpenStack on 10 hosts. - "Exists in Infrahub but does not exist in OpenStack" matches every VM because openstack_nova_server_status returns no series; excluded as a rule defect. - Prometheus activeAt is reset several times a day by dips in the Resources metric, so alert ages are recovered from ALERTS history instead. Takes ~2,650 firing alerts down to ~20 that need a decision. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
422 lines
16 KiB
Python
422 lines
16 KiB
Python
"""Normalizes Prometheus alerts into the alert kinds the CX runbooks cover."""
|
|
from __future__ import annotations
|
|
|
|
import datetime as dt
|
|
import hashlib
|
|
import re
|
|
from dataclasses import dataclass, field
|
|
from typing import Any, Optional
|
|
|
|
EMOJI_RE = re.compile(r":[a-z0-9_+\-]+:")
|
|
THRESHOLD_RE = re.compile(r"greater than\s*(\d+)\s*min", re.I)
|
|
ORG_RE = re.compile(r"^\s*(\d+)\s*-\s*(.*)$")
|
|
NONE_VALUES = {"", "none", "null", "unknown", "n/a"}
|
|
|
|
# Alerts excluded outright. "Exists in Infrahub but does not exist in OpenStack"
|
|
# is built as `Resources unless on(openstack_id) openstack_nova_server_status`,
|
|
# and that right-hand metric is currently empty - so every Infrahub VM matches
|
|
# and the alert fires thousands of times. It is a monitoring fault, not a queue
|
|
# of work, so it never reaches the UI.
|
|
EXCLUDED_ALERTNAMES = frozenset({
|
|
"Exists in Infrahub but does not exist in OpenStack",
|
|
})
|
|
|
|
# Rule files whose alerts belong to CX. Everything else is infrastructure.
|
|
CX_RULE_FILES = ("infrahub-rules",)
|
|
NODE_RULE_FILE = "node-exporter-rules.yml"
|
|
|
|
# The order CX wants to work the queue in.
|
|
FOCUS_ORDER = (
|
|
"rogue_vm", "duplicate_ip", "total_gpus", "hibernating",
|
|
"creating", "shutoff", "deleting", "error",
|
|
"restoring", "rebooting", "build", "orphan_vm", "status_mismatch",
|
|
)
|
|
|
|
# Priority and estimated time to resolve, from the "Infrahub Errors
|
|
# Remediation" alert-conditions and runbook tables.
|
|
KIND_META: dict[str, dict[str, str]] = {
|
|
"error": {"title": "Instance in ERROR state", "priority": "LOW-HIGH", "ettr": "5-30 min", "delay": "none"},
|
|
"deleting": {"title": "Instance in DELETING state", "priority": "LOW", "ettr": "5-15 min", "delay": "30 min"},
|
|
"shutoff": {"title": "Instance in SHUTOFF state", "priority": "LOW", "ettr": "5-10 min", "delay": "30 min"},
|
|
"hibernating": {"title": "Instance in HIBERNATING state", "priority": "HIGH", "ettr": "5-30 min", "delay": "30 min"},
|
|
"creating": {"title": "Instance in CREATING state", "priority": "MEDIUM", "ettr": "5-15 min", "delay": "30 min"},
|
|
"restoring": {"title": "Instance in RESTORING state", "priority": "HIGH", "ettr": "5-15 min", "delay": "30 min"},
|
|
"rebooting": {"title": "Instance in REBOOTING state", "priority": "HIGH", "ettr": "5-15 min", "delay": "30 min"},
|
|
"build": {"title": "Instance in BUILD state", "priority": "MEDIUM", "ettr": "5-15 min", "delay": "30 min"},
|
|
"rogue_vm": {"title": "Suspected Rogue VM", "priority": "HIGH", "ettr": "5-30 min", "delay": "4 hours"},
|
|
"duplicate_ip": {"title": "Duplicated IPs", "priority": "HIGH", "ettr": "5-15 min", "delay": "10 min"},
|
|
"total_gpus": {"title": "Problem with Total GPUs in a System", "priority": "HIGH", "ettr": "5-15 min", "delay": "5 min"},
|
|
"orphan_vm": {"title": "Suspected Orphan VM", "priority": "HIGH", "ettr": "5-30 min", "delay": "4 hours"},
|
|
"status_mismatch": {"title": "Infrahub/OpenStack status mismatch", "priority": "HIGH", "ettr": "5-30 min", "delay": "30 min"},
|
|
}
|
|
|
|
STATE_KINDS = ("error", "deleting", "shutoff", "hibernating", "creating", "restoring", "rebooting", "build")
|
|
|
|
|
|
def clean_alertname(name: str) -> str:
|
|
"""Strip the Slack emoji shortcodes Prometheus embeds in alert names."""
|
|
return EMOJI_RE.sub("", str(name or "")).strip()
|
|
|
|
|
|
def classify(alertname: str) -> str:
|
|
name = clean_alertname(alertname).lower()
|
|
if alertname in EXCLUDED_ALERTNAMES or clean_alertname(alertname) in EXCLUDED_ALERTNAMES:
|
|
return "excluded"
|
|
if "rogue vm" in name:
|
|
return "rogue_vm"
|
|
if "orphan vm" in name:
|
|
return "orphan_vm"
|
|
if "duplicated ip" in name or "duplicate ip" in name:
|
|
return "duplicate_ip"
|
|
if "total gpus" in name:
|
|
return "total_gpus"
|
|
match = re.search(r"instance in (\w+) state", name)
|
|
if match:
|
|
state = match.group(1).lower()
|
|
if state in STATE_KINDS:
|
|
return state
|
|
# The per-region cross-check rules, e.g.
|
|
# "Openstack status=SHUTOFF and Infrahub status!=SHUTOFF in CANADA-1".
|
|
if "failure of hibernation" in name:
|
|
return "hibernating"
|
|
if "openstack status=" in name and "infrahub status" in name:
|
|
return "status_mismatch"
|
|
return "other"
|
|
|
|
|
|
def category(alertname: str, rule_file: str = "") -> str:
|
|
"""Which tab an alert belongs in: 'cx', 'node', or 'infra'."""
|
|
if any(token in rule_file for token in CX_RULE_FILES):
|
|
return "cx"
|
|
if rule_file == NODE_RULE_FILE:
|
|
return "node"
|
|
if rule_file:
|
|
return "infra"
|
|
# No rule metadata (e.g. a pasted alert): fall back to the classifier.
|
|
return "cx" if classify(alertname) not in ("other", "excluded") else "infra"
|
|
|
|
|
|
def _clean(value: Any) -> str:
|
|
text = str(value or "").strip()
|
|
return "" if text.lower() in NONE_VALUES else text
|
|
|
|
|
|
def split_organization(value: str) -> tuple[str, str]:
|
|
"""Split the `organization` label ("8463 - Some Org") into id and name."""
|
|
match = ORG_RE.match(str(value or ""))
|
|
if match:
|
|
return match.group(1), match.group(2).strip()
|
|
return "", _clean(value)
|
|
|
|
|
|
def _parse_active_at(value: Any) -> Optional[dt.datetime]:
|
|
text = str(value or "").strip()
|
|
if not text:
|
|
return None
|
|
text = re.sub(r"(\.\d{1,6})\d*Z?$", r"\1", text.replace("Z", "+00:00"))
|
|
if text.endswith("+00:00") is False and "+" not in text[10:]:
|
|
text += "+00:00"
|
|
try:
|
|
return dt.datetime.fromisoformat(text)
|
|
except ValueError:
|
|
return None
|
|
|
|
|
|
@dataclass
|
|
class Alert:
|
|
"""One normalized Prometheus alert."""
|
|
|
|
kind: str
|
|
alertname: str
|
|
labels: dict[str, str] = field(default_factory=dict)
|
|
annotations: dict[str, str] = field(default_factory=dict)
|
|
state: str = "firing"
|
|
active_at: Optional[dt.datetime] = None
|
|
|
|
# Fields the runbooks key off.
|
|
openstack_id: str = ""
|
|
instance_name: str = ""
|
|
host: str = ""
|
|
region_label: str = ""
|
|
region: str = ""
|
|
status: str = ""
|
|
floating_ip: str = ""
|
|
flavor_name: str = ""
|
|
flavor_gpu: str = ""
|
|
org_id: str = ""
|
|
org_name: str = ""
|
|
contract_id: str = ""
|
|
gpu_name: str = ""
|
|
threshold_min: Optional[int] = None
|
|
|
|
# Where the rule came from (from RuleIndex), and the screening result.
|
|
rule_file: str = ""
|
|
rule_group: str = ""
|
|
for_seconds: int = 0
|
|
category: str = "cx"
|
|
screen: dict[str, Any] = field(default_factory=dict)
|
|
|
|
# How long the condition has actually held, recovered from ALERTS history.
|
|
# activeAt alone is unreliable: a metric-pipeline dip resets it on every
|
|
# live alert at once, which is why raw ages cluster on one timestamp.
|
|
true_age_minutes: Optional[int] = None
|
|
true_age_capped: bool = False
|
|
|
|
@property
|
|
def title(self) -> str:
|
|
return KIND_META.get(self.kind, {}).get("title", clean_alertname(self.alertname))
|
|
|
|
@property
|
|
def priority(self) -> str:
|
|
return KIND_META.get(self.kind, {}).get("priority", "UNKNOWN")
|
|
|
|
@property
|
|
def ettr(self) -> str:
|
|
return KIND_META.get(self.kind, {}).get("ettr", "unknown")
|
|
|
|
@property
|
|
def is_kubernetes(self) -> bool:
|
|
"""Per the general process: kube* instance names are likely K8s nodes."""
|
|
return self.instance_name.lower().startswith("kube") or "-minion-" in self.instance_name.lower()
|
|
|
|
@property
|
|
def age_minutes(self) -> Optional[int]:
|
|
if not self.active_at:
|
|
return None
|
|
now = dt.datetime.now(dt.timezone.utc)
|
|
return max(0, int((now - self.active_at).total_seconds() // 60))
|
|
|
|
@staticmethod
|
|
def _duration_text(minutes: Optional[int]) -> str:
|
|
if minutes is None:
|
|
return "unknown"
|
|
if minutes < 60:
|
|
return f"{minutes}m"
|
|
hours, mins = divmod(minutes, 60)
|
|
if hours < 24:
|
|
return f"{hours}h {mins}m" if mins else f"{hours}h"
|
|
days, hours = divmod(hours, 24)
|
|
return f"{days}d {hours}h" if hours else f"{days}d"
|
|
|
|
@property
|
|
def age_text(self) -> str:
|
|
"""Raw Prometheus activeAt duration."""
|
|
return self._duration_text(self.age_minutes)
|
|
|
|
@property
|
|
def effective_age_minutes(self) -> Optional[int]:
|
|
"""True condition duration where known, else the raw activeAt age."""
|
|
return self.true_age_minutes if self.true_age_minutes is not None else self.age_minutes
|
|
|
|
@property
|
|
def effective_age_text(self) -> str:
|
|
text = self._duration_text(self.effective_age_minutes)
|
|
if self.true_age_minutes is not None and self.true_age_capped:
|
|
return f"{text}+"
|
|
return text
|
|
|
|
@property
|
|
def age_is_reset(self) -> bool:
|
|
"""True when activeAt materially understates how long this has held."""
|
|
if self.true_age_minutes is None or self.age_minutes is None:
|
|
return False
|
|
return self.true_age_minutes - self.age_minutes > 60
|
|
|
|
@property
|
|
def is_internal_org(self) -> bool:
|
|
"""Internal/test organizations are not customer-impacting."""
|
|
return "nexgencloud.com" in self.org_name.lower()
|
|
|
|
@property
|
|
def is_infra_owned(self) -> bool:
|
|
"""Platform-owned nodes (storage etc.) name themselves after their host."""
|
|
return bool(self.instance_name) and self.instance_name.lower() == self.host.lower()
|
|
|
|
def fingerprint(self) -> str:
|
|
basis = "|".join([
|
|
self.kind,
|
|
self.openstack_id or self.instance_name or "",
|
|
self.host,
|
|
self.floating_ip,
|
|
self.region,
|
|
])
|
|
return hashlib.sha1(basis.encode()).hexdigest()[:16]
|
|
|
|
def to_json(self) -> dict[str, Any]:
|
|
return {
|
|
"id": self.fingerprint(),
|
|
"kind": self.kind,
|
|
"title": self.title,
|
|
"alertname": clean_alertname(self.alertname),
|
|
"raw_alertname": self.alertname,
|
|
"state": self.state,
|
|
"priority": self.priority,
|
|
"ettr": self.ettr,
|
|
"active_at": self.active_at.isoformat() if self.active_at else "",
|
|
"age_minutes": self.age_minutes,
|
|
"age_text": self.age_text,
|
|
"true_age_minutes": self.true_age_minutes,
|
|
"true_age_capped": self.true_age_capped,
|
|
"effective_age_minutes": self.effective_age_minutes,
|
|
"effective_age_text": self.effective_age_text,
|
|
"age_is_reset": self.age_is_reset,
|
|
"threshold_min": self.threshold_min,
|
|
"rule_file": self.rule_file,
|
|
"rule_group": self.rule_group,
|
|
"for_seconds": self.for_seconds,
|
|
"category": self.category,
|
|
"screen": self.screen,
|
|
"is_internal_org": self.is_internal_org,
|
|
"is_infra_owned": self.is_infra_owned,
|
|
"openstack_id": self.openstack_id,
|
|
"instance_name": self.instance_name,
|
|
"host": self.host,
|
|
"region": self.region,
|
|
"region_label": self.region_label,
|
|
"status": self.status,
|
|
"floating_ip": self.floating_ip,
|
|
"flavor_name": self.flavor_name,
|
|
"flavor_gpu": self.flavor_gpu,
|
|
"gpu_name": self.gpu_name,
|
|
"org_id": self.org_id,
|
|
"org_name": self.org_name,
|
|
"contract_id": self.contract_id,
|
|
"is_kubernetes": self.is_kubernetes,
|
|
"labels": self.labels,
|
|
"annotations": self.annotations,
|
|
}
|
|
|
|
|
|
def map_region(region_label: str) -> str:
|
|
"""CANADA-1 -> ca1.
|
|
|
|
Deliberately a local table rather than a call into CX-Tools: building the
|
|
alert queue must not touch cxlib, because constructing a CX-Tools Config
|
|
loads credentials and would pop a 1Password prompt just to list alerts.
|
|
Kept in sync with SUPPORTED_REGIONS in cxlib/constants.py.
|
|
"""
|
|
fallback = {
|
|
"canada-1": "ca1", "canada-2": "ca2", "us-1": "us1", "norway-1": "no1",
|
|
"ca-1": "ca1", "ca-2": "ca2", "no-1": "no1",
|
|
"ca1": "ca1", "ca2": "ca2", "us1": "us1", "no1": "no1",
|
|
}
|
|
return fallback.get(str(region_label or "").strip().lower(), "")
|
|
|
|
|
|
def from_labels(labels: dict[str, str], annotations: Optional[dict[str, str]] = None,
|
|
state: str = "firing", active_at: Any = None) -> Alert:
|
|
labels = {str(k): str(v) for k, v in (labels or {}).items()}
|
|
alertname = labels.get("alertname", "")
|
|
kind = classify(alertname)
|
|
|
|
region_label = _clean(labels.get("region"))
|
|
org_id, org_name = split_organization(labels.get("organization", ""))
|
|
threshold = THRESHOLD_RE.search(alertname)
|
|
|
|
# For host-scoped alerts the `instance` label is the hypervisor; for
|
|
# VM-scoped alerts it is the hypervisor too, or "Unknown" when the VM never
|
|
# landed on a host.
|
|
host = _clean(labels.get("instance"))
|
|
|
|
alert = Alert(
|
|
kind=kind,
|
|
alertname=alertname,
|
|
labels=labels,
|
|
annotations={str(k): str(v) for k, v in (annotations or {}).items()},
|
|
state=str(state or "firing"),
|
|
active_at=_parse_active_at(active_at),
|
|
openstack_id=_clean(labels.get("openstack_id")),
|
|
instance_name=_clean(labels.get("instance_name")),
|
|
host=host,
|
|
region_label=region_label,
|
|
region=map_region(region_label) or _infer_region_from_host(host),
|
|
status=_clean(labels.get("status")),
|
|
floating_ip=_clean(labels.get("floating_ip")),
|
|
flavor_name=_clean(labels.get("flavor_name")),
|
|
flavor_gpu=_clean(labels.get("flavor_gpu")),
|
|
gpu_name=_clean(labels.get("gpu_name")),
|
|
org_id=org_id,
|
|
org_name=org_name,
|
|
contract_id=_clean(labels.get("contract_id")),
|
|
threshold_min=int(threshold.group(1)) if threshold else None,
|
|
)
|
|
return alert
|
|
|
|
|
|
def _infer_region_from_host(host: str) -> str:
|
|
match = re.match(r"^(ca1|ca2|no1|us1)-", str(host or "").strip(), re.I)
|
|
return match.group(1).lower() if match else ""
|
|
|
|
|
|
def from_prometheus(raw: dict[str, Any], rule_index: Any = None, true_age: Any = None) -> Alert:
|
|
alert = from_labels(
|
|
raw.get("labels") or {},
|
|
raw.get("annotations") or {},
|
|
state=str(raw.get("state") or "firing"),
|
|
active_at=raw.get("activeAt"),
|
|
)
|
|
meta = rule_index.get(alert.alertname) if rule_index is not None else {}
|
|
if meta:
|
|
alert.rule_file = str(meta.get("file") or "")
|
|
alert.rule_group = str(meta.get("group") or "")
|
|
alert.for_seconds = int(meta.get("for_seconds") or 0)
|
|
alert.category = category(alert.alertname, alert.rule_file)
|
|
if true_age is not None:
|
|
alert.true_age_minutes, alert.true_age_capped = true_age.lookup(alert.labels)
|
|
return alert
|
|
|
|
|
|
def cx_relevant(alert: Alert) -> bool:
|
|
"""True for alert kinds the CX runbooks cover."""
|
|
return alert.kind in KIND_META and alert.kind != "excluded"
|
|
|
|
|
|
def is_excluded(alert: Alert) -> bool:
|
|
return alert.kind == "excluded" or alert.alertname in EXCLUDED_ALERTNAMES
|
|
|
|
|
|
def sort_key(alert: Alert) -> tuple:
|
|
"""Newest first: a fresh alert is the one that still needs a decision.
|
|
|
|
Long-running alerts sink to the bottom - they are either chronic and
|
|
already ticketed, or noise nobody has silenced. Alerts with no known start
|
|
time sort last rather than jumping to the top.
|
|
|
|
Sorted on the true condition duration, not activeAt: a pipeline dip resets
|
|
activeAt on every live alert at once, which would otherwise flatten the
|
|
ordering into a single tie.
|
|
"""
|
|
age = alert.effective_age_minutes
|
|
return (age if age is not None else 10**9, alert.title)
|
|
|
|
|
|
def focus_rank(kind: str) -> int:
|
|
try:
|
|
return FOCUS_ORDER.index(kind)
|
|
except ValueError:
|
|
return len(FOCUS_ORDER)
|
|
|
|
|
|
def group_alerts(items: list[Alert]) -> list[dict[str, Any]]:
|
|
"""Group alerts into collapsible sections, in CX's working order."""
|
|
buckets: dict[str, list[Alert]] = {}
|
|
for alert in items:
|
|
buckets.setdefault(alert.kind, []).append(alert)
|
|
|
|
groups: list[dict[str, Any]] = []
|
|
for kind, members in buckets.items():
|
|
members.sort(key=sort_key)
|
|
actionable = [a for a in members if a.screen.get("actionable", True)]
|
|
groups.append({
|
|
"kind": kind,
|
|
"title": KIND_META.get(kind, {}).get("title", kind),
|
|
"priority": KIND_META.get(kind, {}).get("priority", "UNKNOWN"),
|
|
"ettr": KIND_META.get(kind, {}).get("ettr", "unknown"),
|
|
"total": len(members),
|
|
"actionable": len(actionable),
|
|
"noise": len(members) - len(actionable),
|
|
"alerts": [a.to_json() for a in members],
|
|
})
|
|
groups.sort(key=lambda g: (focus_rank(g["kind"]), -g["actionable"]))
|
|
return groups
|