CX Triage: alert diagnosis over the CX-Tools collectors
Read-only triage for the Infrahub error alerts. Pulls the Prometheus alert queue, re-checks each alert's condition against live state to separate real work from noise, diagnoses it using the CX runbooks, and drafts the customer comms with contacts resolved from Infrahub. Findings from validating against production: - "Suspected Rogue VM" fires on spare GPU capacity, not rogue VMs: In_Use_Gpus equals the physical count on 71 of 75 firing hosts, so the rule reduces to "this host has a free GPU". Verified against OpenStack on 10 hosts. - "Exists in Infrahub but does not exist in OpenStack" matches every VM because openstack_nova_server_status returns no series; excluded as a rule defect. - Prometheus activeAt is reset several times a day by dips in the Resources metric, so alert ages are recovered from ALERTS history instead. Takes ~2,650 firing alerts down to ~20 that need a decision. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
313
triagelib/screening.py
Normal file
313
triagelib/screening.py
Normal file
@@ -0,0 +1,313 @@
|
||||
"""Noise-vs-real screening.
|
||||
|
||||
An alert firing is not the same as work existing. Prometheus keeps an alert up
|
||||
until its expression stops matching on the next evaluation, and the CX rules sit
|
||||
on top of a metric pipeline that can go stale or empty. So before anything gets
|
||||
diagnosed, each alert's condition is re-checked against current state.
|
||||
|
||||
The re-check is deliberately cheap: it reads the same Prometheus series the rules
|
||||
are built from (one bulk snapshot for the whole queue) rather than making an
|
||||
Infrahub or OpenStack call per alert. Anything it cannot settle is treated as
|
||||
real - screening only ever demotes an alert on positive evidence.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import time
|
||||
from typing import Any, Optional
|
||||
|
||||
from .alerts import Alert
|
||||
|
||||
# Beyond this, an alert is chronic: it is either already ticketed or nobody has
|
||||
# silenced it. Either way it is not today's queue.
|
||||
CHRONIC_DAYS = 3
|
||||
|
||||
# Verdicts, most to least urgent.
|
||||
REAL = "real"
|
||||
OVERDUE = "overdue"
|
||||
UNVERIFIED = "unverified"
|
||||
CHRONIC = "chronic"
|
||||
LOW_IMPACT = "low_impact"
|
||||
PENDING = "pending"
|
||||
RESOLVED = "resolved"
|
||||
RULE_DEFECT = "rule_defect"
|
||||
SUPPRESSED = "suppressed"
|
||||
|
||||
VERDICT_LABELS = {
|
||||
REAL: "needs action",
|
||||
OVERDUE: "overdue",
|
||||
UNVERIFIED: "needs action (unverified)",
|
||||
CHRONIC: "chronic",
|
||||
LOW_IMPACT: "low impact",
|
||||
PENDING: "not yet firing",
|
||||
RESOLVED: "already resolved",
|
||||
RULE_DEFECT: "invalid - alert rule defect",
|
||||
SUPPRESSED: "hidden by a rule",
|
||||
}
|
||||
|
||||
# Runbook commitments: past this age the customer contact is late, not chronic.
|
||||
# From "Instance in ERROR state": if a stock-failure instance is not deleted
|
||||
# within a day, contact the customer. SHUTOFF is here for a different reason -
|
||||
# the VM accrues full cost the entire time it is stopped, so an old one is a
|
||||
# customer who has been paying for nothing for longer, not a stale alert.
|
||||
SLA_HOURS = {"error": 24, "creating": 24, "restoring": 24, "rebooting": 24,
|
||||
"build": 24, "shutoff": 48}
|
||||
|
||||
SLA_REASON = {
|
||||
"shutoff": "a SHUTOFF VM accrues full cost the whole time, so this customer has been paying "
|
||||
"for a stopped instance that long and may never have been told",
|
||||
}
|
||||
|
||||
# Verdicts that stay in the working queue by default.
|
||||
ACTIONABLE = {REAL, OVERDUE, UNVERIFIED}
|
||||
|
||||
# Kept for the SLA check: an internal owner should not make an alert *overdue*.
|
||||
def _is_internal(alert) -> bool:
|
||||
return bool(getattr(alert, 'is_internal_org', False))
|
||||
|
||||
|
||||
def _result(verdict: str, reason: str, *, current: str = "", detail: str = "") -> dict[str, Any]:
|
||||
return {
|
||||
"verdict": verdict,
|
||||
"label": VERDICT_LABELS[verdict],
|
||||
"reason": reason,
|
||||
"actionable": verdict in ACTIONABLE,
|
||||
"current_state": current,
|
||||
"detail": detail,
|
||||
}
|
||||
|
||||
|
||||
def _resources_row(alert: Alert, snap: Any) -> Optional[dict[str, str]]:
|
||||
"""Find the VM's current Infrahub row in the snapshot."""
|
||||
if alert.openstack_id and alert.openstack_id in snap.by_openstack_id:
|
||||
return snap.by_openstack_id[alert.openstack_id]
|
||||
if alert.instance_name and alert.instance_name in snap.by_instance_name:
|
||||
return snap.by_instance_name[alert.instance_name]
|
||||
return None
|
||||
|
||||
|
||||
def _screen_state_alert(alert: Alert, snap: Any) -> dict[str, Any]:
|
||||
"""State alerts: does Infrahub still report the state that fired?"""
|
||||
expected = alert.status.upper()
|
||||
row = _resources_row(alert, snap)
|
||||
|
||||
if row is None:
|
||||
if alert.kind == "creating" and not alert.openstack_id:
|
||||
# A CREATING VM that never reached OpenStack has no Resources row to
|
||||
# find; the alert stands on its own.
|
||||
return _result(REAL, "Instance never got an OpenStack ID, so it cannot have recovered.")
|
||||
return _result(
|
||||
RESOLVED,
|
||||
"No longer present in Infrahub's Resources series - the record has been deleted or cleaned up.",
|
||||
)
|
||||
|
||||
current = row.get("status", "").upper()
|
||||
if not expected:
|
||||
return _result(UNVERIFIED, "Alert carried no status label, so its condition could not be re-checked.",
|
||||
current=current)
|
||||
if current == expected:
|
||||
return _result(REAL, f"Infrahub still reports {current}.", current=current)
|
||||
return _result(
|
||||
RESOLVED,
|
||||
f"Fired on {expected} but Infrahub now reports {current} - it resolved on its own.",
|
||||
current=current,
|
||||
)
|
||||
|
||||
|
||||
def _screen_duplicate_ip(alert: Alert, snap: Any) -> dict[str, Any]:
|
||||
count = snap.fip_counts.get(alert.floating_ip)
|
||||
if count is None:
|
||||
return _result(RESOLVED, f"No Infrahub VM currently holds {alert.floating_ip}.")
|
||||
if count > 1:
|
||||
return _result(REAL, f"{count} VMs still hold {alert.floating_ip}.", current=f"{count} claimants")
|
||||
return _result(
|
||||
RESOLVED,
|
||||
f"Only 1 VM holds {alert.floating_ip} now - the duplicate is gone.",
|
||||
current="1 claimant",
|
||||
)
|
||||
|
||||
|
||||
def _screen_rogue_vm(alert: Alert, snap: Any) -> dict[str, Any]:
|
||||
"""Rogue VM fires on a per-host GPU accounting gap; re-evaluate the gap.
|
||||
|
||||
The rule subtracts Infrahub's allocated GPUs from `In_Use_Gpus`. On almost
|
||||
every host `In_Use_Gpus` equals `Total_Gpus` - the physical GPU count - so
|
||||
the expression reduces to "this host has at least one unallocated GPU" and
|
||||
fires on ordinary spare capacity. Validated against OpenStack on 10 firing
|
||||
hosts: Infrahub and OpenStack agreed exactly on all of them.
|
||||
"""
|
||||
host = alert.host or alert.instance_name
|
||||
delta = snap.rogue_delta.get(host)
|
||||
if delta is None:
|
||||
return _result(UNVERIFIED, f"No current GPU accounting data for {host}.")
|
||||
if delta >= 1:
|
||||
known = len(snap.resources_by_host.get(host, []))
|
||||
in_use = snap.in_use_gpus.get(host)
|
||||
total = snap.total_gpus.get(host)
|
||||
if in_use is not None and total is not None and in_use == total:
|
||||
return _result(
|
||||
RULE_DEFECT,
|
||||
f"Not a rogue VM: on {host} the rule's 'GPUs in use' reading ({int(in_use)}) is just the "
|
||||
f"physical GPU count, so it is reporting {int(delta)} free GPU(s) as a discrepancy.",
|
||||
current=f"{int(delta)} GPU(s) spare capacity",
|
||||
detail=(
|
||||
"In_Use_Gpus == Total_Gpus on this host, so the rule expression reduces to "
|
||||
"'physical GPUs minus allocated GPUs', which is spare capacity rather than an "
|
||||
"Infrahub/OpenStack mismatch. The rule needs fixing at source."
|
||||
),
|
||||
)
|
||||
if total is None:
|
||||
detail = (f"Infrahub records {known} VM(s) on this host. No Total_Gpus reading is available, so the "
|
||||
"spare-capacity explanation cannot be confirmed or ruled out from metrics alone.")
|
||||
else:
|
||||
detail = (f"Infrahub records {known} VM(s) on this host. In_Use_Gpus ({int(in_use)}) differs from "
|
||||
f"Total_Gpus ({int(total)}), so this is not simply spare capacity.")
|
||||
return _result(
|
||||
REAL,
|
||||
f"{int(delta)} GPU(s) allocated on {host} are still unaccounted for in Infrahub.",
|
||||
current=f"gap {int(delta)}",
|
||||
detail=detail,
|
||||
)
|
||||
return _result(
|
||||
RESOLVED,
|
||||
f"GPU accounting for {host} now balances (delta {int(delta)}).",
|
||||
current=f"delta {int(delta)}",
|
||||
)
|
||||
|
||||
|
||||
def _screen_total_gpus(alert: Alert, snap: Any) -> dict[str, Any]:
|
||||
host = alert.host or alert.instance_name
|
||||
total = snap.total_gpus.get(host)
|
||||
if total is None:
|
||||
return _result(UNVERIFIED, f"No current Total_Gpus reading for {host}.")
|
||||
# The rule fires on any count in 1,2,3,5,6,7,9 - i.e. not a full complement.
|
||||
if int(total) in (0, 4, 8):
|
||||
return _result(
|
||||
RESOLVED,
|
||||
f"{host} now reports {int(total)} GPUs, a valid complement.",
|
||||
current=f"{int(total)} GPUs",
|
||||
)
|
||||
in_use = snap.in_use_gpus.get(host)
|
||||
detail = f"{int(in_use)} GPU(s) currently allocated to instances." if in_use is not None else ""
|
||||
return _result(
|
||||
REAL,
|
||||
f"{host} still reports {int(total)} GPUs - hardware is missing.",
|
||||
current=f"{int(total)} GPUs",
|
||||
detail=detail,
|
||||
)
|
||||
|
||||
|
||||
_KIND_SCREENS = {
|
||||
"duplicate_ip": _screen_duplicate_ip,
|
||||
"rogue_vm": _screen_rogue_vm,
|
||||
"orphan_vm": _screen_rogue_vm,
|
||||
"total_gpus": _screen_total_gpus,
|
||||
}
|
||||
|
||||
|
||||
def screen(alert: Alert, snap: Any, user_settings: Any = None) -> dict[str, Any]:
|
||||
"""Decide whether an alert is worth a human's attention right now."""
|
||||
# A rule the team wrote wins over anything inferred here.
|
||||
if user_settings is not None:
|
||||
from . import settings as settings_mod
|
||||
|
||||
rule = settings_mod.first_match(user_settings, alert)
|
||||
if rule:
|
||||
reason = rule.get("reason") or "Matched a suppression rule."
|
||||
return _result(
|
||||
SUPPRESSED,
|
||||
f"Hidden by \u201c{rule.get('name')}\u201d - {reason}",
|
||||
detail="Edit or disable this in Settings.",
|
||||
)
|
||||
|
||||
# Prometheus has not committed to this alert yet.
|
||||
if alert.state == "pending":
|
||||
remaining = ""
|
||||
if alert.for_seconds and alert.age_minutes is not None:
|
||||
remaining = f" It needs {alert.for_seconds // 60} min of continuous firing; it has {alert.age_minutes} min."
|
||||
return _result(PENDING, f"Prometheus still has this pending, not firing.{remaining}")
|
||||
|
||||
if snap is None or not getattr(snap, "loaded", False):
|
||||
return _result(UNVERIFIED, "Current-state snapshot unavailable, so the condition could not be re-checked.")
|
||||
|
||||
screener = _KIND_SCREENS.get(alert.kind)
|
||||
result = screener(alert, snap) if screener else _screen_state_alert(alert, snap)
|
||||
|
||||
# A still-valid alert can still be the wrong thing to spend time on.
|
||||
if result["actionable"]:
|
||||
# Uses the recovered duration: activeAt is reset by pipeline dips, which
|
||||
# would make every chronic alert look hours old.
|
||||
age = alert.effective_age_minutes
|
||||
sla = SLA_HOURS.get(alert.kind)
|
||||
if sla and age is not None and age > sla * 60 and not alert.is_internal_org:
|
||||
# The runbook commits to contacting the customer inside this window,
|
||||
# so age makes it more urgent, not less. Never demote these to chronic.
|
||||
why = SLA_REASON.get(
|
||||
alert.kind,
|
||||
f"past the {sla}h point where the runbook says to contact the customer",
|
||||
)
|
||||
return _result(
|
||||
OVERDUE,
|
||||
f"Condition has held for {alert.effective_age_text} - {why}. Overdue, not chronic.",
|
||||
current=result["current_state"],
|
||||
detail=result["reason"],
|
||||
)
|
||||
if age is not None and age > CHRONIC_DAYS * 24 * 60:
|
||||
note = result["reason"]
|
||||
if alert.age_is_reset:
|
||||
note += (f" Prometheus reports only {alert.age_text} because a metric-pipeline dip reset "
|
||||
"activeAt; the condition itself has held far longer.")
|
||||
return _result(
|
||||
CHRONIC,
|
||||
f"Condition still holds, but it has held for {alert.effective_age_text} - "
|
||||
"chronic, so it is likely already ticketed rather than new work.",
|
||||
current=result["current_state"],
|
||||
detail=note,
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
def screen_all(items: list[Alert], snap: Any, user_settings: Any = None) -> None:
|
||||
for alert in items:
|
||||
alert.screen = screen(alert, snap, user_settings)
|
||||
|
||||
|
||||
def summarize(items: list[Alert]) -> dict[str, Any]:
|
||||
counts: dict[str, int] = {}
|
||||
for alert in items:
|
||||
verdict = alert.screen.get("verdict", UNVERIFIED)
|
||||
counts[verdict] = counts.get(verdict, 0) + 1
|
||||
return {
|
||||
"counts": counts,
|
||||
"actionable": sum(1 for a in items if a.screen.get("actionable")),
|
||||
"screened_out": sum(1 for a in items if not a.screen.get("actionable")),
|
||||
"labels": VERDICT_LABELS,
|
||||
}
|
||||
|
||||
|
||||
def health_warnings(snap: Any) -> list[str]:
|
||||
"""Rules whose input metrics are empty are broken, not quiet."""
|
||||
warnings: list[str] = []
|
||||
for metric in getattr(snap, "broken_inputs", []) or []:
|
||||
if metric == "openstack_nova_server_status":
|
||||
warnings.append(
|
||||
"The OpenStack server metric (openstack_nova_server_status) is currently empty. Any rule built on "
|
||||
"it is unreliable: 'Exists in Infrahub but does not exist in OpenStack' matches every VM (which is "
|
||||
"why it is excluded here), and 'Suspected Orphan VM' cannot fire at all. Worth raising with whoever "
|
||||
"owns the exporter."
|
||||
)
|
||||
else:
|
||||
warnings.append(
|
||||
f"The metric '{metric}' is currently empty, so alert rules that depend on it are unreliable."
|
||||
)
|
||||
|
||||
dips = getattr(snap, "pipeline_dips", []) or []
|
||||
if dips:
|
||||
latest = max(dips, key=lambda d: d["end"])
|
||||
mins_ago = max(0, int((time.time() - latest["end"]) // 60))
|
||||
warnings.append(
|
||||
f"The Infrahub 'Resources' metric dropped most of its series {len(dips)} time(s) in the last 24h "
|
||||
f"(most recently {mins_ago} min ago: {latest['low']} of ~{latest['normal']} series for "
|
||||
f"{latest['minutes']} min). Every alert live during a dip resolves and re-fires, so Prometheus' own "
|
||||
"alert ages all reset together. Ages shown here are recovered from ALERTS history instead."
|
||||
)
|
||||
return warnings
|
||||
Reference in New Issue
Block a user