CX Triage: alert diagnosis over the CX-Tools collectors
Read-only triage for the Infrahub error alerts. Pulls the Prometheus alert queue, re-checks each alert's condition against live state to separate real work from noise, diagnoses it using the CX runbooks, and drafts the customer comms with contacts resolved from Infrahub. Findings from validating against production: - "Suspected Rogue VM" fires on spare GPU capacity, not rogue VMs: In_Use_Gpus equals the physical count on 71 of 75 firing hosts, so the rule reduces to "this host has a free GPU". Verified against OpenStack on 10 hosts. - "Exists in Infrahub but does not exist in OpenStack" matches every VM because openstack_nova_server_status returns no series; excluded as a rule defect. - Prometheus activeAt is reset several times a day by dips in the Resources metric, so alert ages are recovered from ALERTS history instead. Takes ~2,650 firing alerts down to ~20 that need a decision. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
290
tests/test_screening.py
Normal file
290
tests/test_screening.py
Normal file
@@ -0,0 +1,290 @@
|
||||
"""Screening, exclusion, categorisation and ordering tests.
|
||||
|
||||
These cover the noise-vs-real decisions, which are what keeps the queue small.
|
||||
No network and no CX-Tools: snapshots are synthetic.
|
||||
"""
|
||||
import datetime as dt
|
||||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
|
||||
|
||||
from triagelib import alerts as A, screening
|
||||
|
||||
|
||||
class Snap:
|
||||
"""Stands in for prometheus.StateSnapshot."""
|
||||
|
||||
loaded = True
|
||||
|
||||
def __init__(self, **kw):
|
||||
self.by_openstack_id = {}
|
||||
self.by_instance_name = {}
|
||||
self.fip_counts = {}
|
||||
self.rogue_delta = {}
|
||||
self.total_gpus = {}
|
||||
self.in_use_gpus = {}
|
||||
self.resources_by_host = {}
|
||||
self.broken_inputs = []
|
||||
self.unattributed_active = 0
|
||||
self.unattributed_active_gpus = 0
|
||||
self.__dict__.update(kw)
|
||||
|
||||
|
||||
def al(name, **labels):
|
||||
return A.from_labels({"alertname": name, **labels})
|
||||
|
||||
|
||||
def aged(alert, minutes):
|
||||
alert.active_at = dt.datetime.now(dt.timezone.utc) - dt.timedelta(minutes=minutes)
|
||||
return alert
|
||||
|
||||
|
||||
FAILS = []
|
||||
|
||||
|
||||
def expect(label, cond, got=""):
|
||||
print((" PASS " if cond else " FAIL ") + label + ("" if cond else f" <- {got}"))
|
||||
if not cond:
|
||||
FAILS.append(label)
|
||||
|
||||
|
||||
OSID = "abc-123"
|
||||
ERROR_ALERT = "Instance in ERROR state in :flag-ca:CA-1"
|
||||
SHUTOFF_ALERT = "Instance in SHUTOFF state in :flag-ca:CA-1 for greater than 30 min"
|
||||
|
||||
print("\nSTATE ALERTS - does the condition still hold?")
|
||||
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", region="CANADA-1")
|
||||
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ERROR"}}))
|
||||
expect("Infrahub still ERROR -> real", a.screen["verdict"] == screening.REAL, a.screen)
|
||||
|
||||
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", region="CANADA-1")
|
||||
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ACTIVE"}}))
|
||||
expect("recovered to ACTIVE -> resolved",
|
||||
a.screen["verdict"] == screening.RESOLVED and "ACTIVE" in a.screen["reason"], a.screen)
|
||||
|
||||
a = al(SHUTOFF_ALERT, openstack_id=OSID, status="SHUTOFF")
|
||||
a.screen = screening.screen(a, Snap())
|
||||
expect("record gone from Infrahub -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
|
||||
|
||||
a = al("Instance in CREATING state in :flag-ca:CA-1 for greater than 30min",
|
||||
openstack_id="None", instance_name="vm-x", status="CREATING")
|
||||
a.screen = screening.screen(a, Snap())
|
||||
expect("CREATING with no OpenStack ID -> real, not 'resolved'",
|
||||
a.screen["verdict"] == screening.REAL, a.screen)
|
||||
|
||||
a = al(ERROR_ALERT, openstack_id=OSID)
|
||||
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ERROR"}}))
|
||||
expect("no status label -> unverified but kept", a.screen["verdict"] == screening.UNVERIFIED, a.screen)
|
||||
|
||||
print("\nDUPLICATED IPs")
|
||||
a = al(":awkward:Duplicated IPs", floating_ip="1.2.3.4")
|
||||
a.screen = screening.screen(a, Snap(fip_counts={"1.2.3.4": 1}))
|
||||
expect("one claimant left -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
|
||||
a.screen = screening.screen(a, Snap(fip_counts={"1.2.3.4": 3}))
|
||||
expect("three claimants -> real", a.screen["verdict"] == screening.REAL and "3 VMs" in a.screen["reason"], a.screen)
|
||||
a.screen = screening.screen(a, Snap(fip_counts={"9.9.9.9": 2}))
|
||||
expect("IP held by nobody -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
|
||||
|
||||
print("\nSUSPECTED ROGUE VM - per-host GPU accounting gap")
|
||||
a = al(":ninja:Suspected Rogue VM", instance="CA1-ESC8-068")
|
||||
a.screen = screening.screen(a, Snap(rogue_delta={"CA1-ESC8-068": 0.0}))
|
||||
expect("gap closed -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
|
||||
a.screen = screening.screen(a, Snap(rogue_delta={"CA1-ESC8-068": 4.0},
|
||||
resources_by_host={"CA1-ESC8-068": [{}] * 4}))
|
||||
expect("gap of 4 GPUs -> real", a.screen["verdict"] == screening.REAL, a.screen)
|
||||
expect("reason quantifies the gap", "4 GPU(s)" in a.screen["reason"], a.screen["reason"])
|
||||
a.screen = screening.screen(a, Snap(rogue_delta={"other-host": 4.0}, total_gpus={"x": 8}))
|
||||
expect("no data for the host -> unverified, still actionable",
|
||||
a.screen["verdict"] == screening.UNVERIFIED and a.screen["actionable"], a.screen)
|
||||
|
||||
print("\nTOTAL GPUs")
|
||||
a = al("Problem with Total GPUs in a System", instance="h1", gpu_name="B200-SXM")
|
||||
a.screen = screening.screen(a, Snap(total_gpus={"h1": 8}))
|
||||
expect("full complement of 8 -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
|
||||
a.screen = screening.screen(a, Snap(total_gpus={"h1": 6}, in_use_gpus={"h1": 6}))
|
||||
expect("6 GPUs -> real", a.screen["verdict"] == screening.REAL, a.screen)
|
||||
|
||||
print("\nSUPPRESSION RULES - what used to be hardcoded is now user-editable")
|
||||
from triagelib import settings as settings_mod
|
||||
import tempfile, os as _os
|
||||
|
||||
_tmp = _os.path.join(tempfile.mkdtemp(), "settings.json")
|
||||
CFG = settings_mod.Settings(_tmp)
|
||||
|
||||
a = al(SHUTOFF_ALERT, openstack_id=OSID, status="SHUTOFF",
|
||||
organization="3491 - luis.sarabando+runpod@nexgencloud.coms-Organization")
|
||||
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "SHUTOFF"}}), CFG)
|
||||
expect("default rule hides internal nexgencloud orgs",
|
||||
a.screen["verdict"] == screening.SUPPRESSED, a.screen)
|
||||
expect("suppression names the rule that did it", "Internal NexGen" in a.screen["reason"], a.screen["reason"])
|
||||
|
||||
a = al("Instance in SHUTOFF state in :flag-no:NO-1 for greater than 30 min", openstack_id=OSID,
|
||||
status="SHUTOFF", instance="no1-stor-runpod03", instance_name="no1-stor-runpod03",
|
||||
organization="99 - Real Customer")
|
||||
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "SHUTOFF"}}), CFG)
|
||||
expect("default rule hides runpod storage nodes", a.screen["verdict"] == screening.SUPPRESSED, a.screen)
|
||||
|
||||
# The combinational case the team asked for: type AND organisation.
|
||||
CFG.upsert_rule({"name": "Modal ERROR churn", "reason": "known batch churn",
|
||||
"conditions": {"kind": ["error"], "organization": ["modal"]}})
|
||||
hit = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", organization="19417 - colin@modal.coms-Organization")
|
||||
hit.screen = screening.screen(hit, Snap(by_openstack_id={OSID: {"status": "ERROR"}}), CFG)
|
||||
expect("error + modal is suppressed", hit.screen["verdict"] == screening.SUPPRESSED, hit.screen)
|
||||
|
||||
miss = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", organization="123 - Someone Else")
|
||||
miss.screen = screening.screen(miss, Snap(by_openstack_id={OSID: {"status": "ERROR"}}), CFG)
|
||||
expect("error from another org is NOT suppressed", miss.screen["verdict"] != screening.SUPPRESSED, miss.screen)
|
||||
|
||||
other = al(SHUTOFF_ALERT, openstack_id=OSID, status="SHUTOFF",
|
||||
organization="19417 - colin@modal.coms-Organization")
|
||||
other.screen = screening.screen(other, Snap(by_openstack_id={OSID: {"status": "SHUTOFF"}}), CFG)
|
||||
expect("modal SHUTOFF is NOT suppressed - both conditions must match",
|
||||
other.screen["verdict"] != screening.SUPPRESSED, other.screen)
|
||||
|
||||
empty = {"name": "catch all", "conditions": {}}
|
||||
expect("a rule with no conditions never matches", not settings_mod.rule_matches(
|
||||
settings_mod._normalize_rule(empty), hit))
|
||||
|
||||
expect("rules survive a reload", settings_mod.Settings(_tmp).rules and any(
|
||||
r["name"] == "Modal ERROR churn" for r in settings_mod.Settings(_tmp).rules))
|
||||
|
||||
print("\nAGE DEMOTIONS")
|
||||
a = aged(al("Problem with Total GPUs in a System", instance="h1"), 7 * 24 * 60)
|
||||
a.screen = screening.screen(a, Snap(total_gpus={"h1": 6}))
|
||||
expect("firing 7 days -> chronic", a.screen["verdict"] == screening.CHRONIC, a.screen)
|
||||
|
||||
a = aged(al("Problem with Total GPUs in a System", instance="h1"), 60)
|
||||
a.screen = screening.screen(a, Snap(total_gpus={"h1": 6}))
|
||||
expect("firing 1 hour -> stays real", a.screen["verdict"] == screening.REAL, a.screen)
|
||||
|
||||
print("\nFAIL-SAFE BEHAVIOUR")
|
||||
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR")
|
||||
a.state = "pending"
|
||||
a.for_seconds = 1800
|
||||
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ERROR"}}))
|
||||
expect("pending -> screened out", a.screen["verdict"] == screening.PENDING and not a.screen["actionable"], a.screen)
|
||||
|
||||
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR")
|
||||
a.screen = screening.screen(a, None)
|
||||
expect("no snapshot -> unverified but NOT hidden",
|
||||
a.screen["verdict"] == screening.UNVERIFIED and a.screen["actionable"], a.screen)
|
||||
|
||||
warnings = screening.health_warnings(Snap(broken_inputs=["openstack_nova_server_status"]))
|
||||
expect("empty nova metric raises a monitoring warning",
|
||||
len(warnings) == 1 and "openstack_nova_server_status" in warnings[0], warnings)
|
||||
|
||||
print("\nEXCLUSION AND TAB ROUTING")
|
||||
ex = al("Exists in Infrahub but does not exist in OpenStack", openstack_id=OSID)
|
||||
expect("orphan spam excluded outright", A.is_excluded(ex) and not A.cx_relevant(ex))
|
||||
expect("node-exporter -> Infrastructure tab",
|
||||
A.category("HostSwapIsFillingUp", "node-exporter-rules.yml") == "node")
|
||||
expect("ceph -> Infrastructure tab", A.category("CephOsdDown", "ceph-rules.yml") == "infra")
|
||||
expect("regional Infrahub rule -> CX tab",
|
||||
A.category(ERROR_ALERT, "infrahub-rules-CA1.yml") == "cx")
|
||||
expect("main Infrahub rule -> CX tab",
|
||||
A.category(":ninja:Suspected Rogue VM", "infrahub-rules.yml") == "cx")
|
||||
expect("status-mismatch rule classified",
|
||||
A.classify("Openstack status=ACTIVE and Infrahub status!=ACTIVE in :flag-ca:CANADA-1 for greater "
|
||||
"than 30min") == "status_mismatch")
|
||||
expect("hibernation-failure rule maps to HIBERNATING",
|
||||
A.classify("Failure of Hibernation on Infrahub in :flag-ca:CANADA-1 for greater than 30min") == "hibernating")
|
||||
expect("orphan VM rule classified", A.classify(":pirate_flag:Suspected Orphan VM") == "orphan_vm")
|
||||
|
||||
print("\nORDERING AND GROUPING")
|
||||
|
||||
|
||||
def real(minutes, name=ERROR_ALERT, **labels):
|
||||
x = aged(al(name, openstack_id="o%d" % minutes, status="ERROR", **labels), minutes)
|
||||
x.screen = {"actionable": True, "verdict": "real", "label": "needs action", "reason": ""}
|
||||
return x
|
||||
|
||||
|
||||
ages = [x["age_minutes"] for x in A.group_alerts([real(500), real(10), real(100), real(9331)])[0]["alerts"]]
|
||||
expect("newest first, oldest at the bottom", ages == [10, 100, 500, 9331], ages)
|
||||
|
||||
unknown = real(50)
|
||||
unknown.active_at = None
|
||||
ages = [x["age_minutes"] for x in A.group_alerts([unknown, real(200), real(5)])[0]["alerts"]]
|
||||
expect("unknown start time sorts last", ages == [5, 200, None], ages)
|
||||
|
||||
groups = A.group_alerts([real(5), real(6, ":ninja:Suspected Rogue VM", instance="h1")])
|
||||
expect("focus order puts rogue VM before ERROR", [g["kind"] for g in groups][0] == "rogue_vm",
|
||||
[g["kind"] for g in groups])
|
||||
|
||||
noisy = real(7)
|
||||
noisy.screen = {"actionable": False, "verdict": "resolved", "label": "already resolved", "reason": ""}
|
||||
group = A.group_alerts([real(5), noisy])[0]
|
||||
expect("group counts action vs noise separately",
|
||||
group["actionable"] == 1 and group["noise"] == 1, group)
|
||||
|
||||
expect("age_text renders days", real(9331).age_text == "6d 11h", real(9331).age_text)
|
||||
expect("age_text renders hours", real(431).age_text == "7h 11m", real(431).age_text)
|
||||
expect("age_text renders minutes", real(7).age_text == "7m", real(7).age_text)
|
||||
|
||||
print("\nTRUE AGE - activeAt reset by pipeline dips")
|
||||
# activeAt says 7h; ALERTS history says 7 days. The true value must win.
|
||||
a = real(431)
|
||||
a.true_age_minutes, a.true_age_capped = 7 * 24 * 60, False
|
||||
expect("effective age prefers the recovered duration", a.effective_age_minutes == 10080, a.effective_age_minutes)
|
||||
expect("reset is detected", a.age_is_reset)
|
||||
expect("raw activeAt still reported", a.age_text == "7h 11m", a.age_text)
|
||||
expect("effective text renders days", a.effective_age_text == "7d", a.effective_age_text)
|
||||
a.screen = screening.screen(a, Snap(by_openstack_id={"o431": {"status": "ERROR"}}))
|
||||
expect("7-day ERROR -> overdue (runbook says contact within 24h), not chronic",
|
||||
a.screen["verdict"] == screening.OVERDUE, a.screen)
|
||||
expect("overdue stays in the actionable queue", a.screen["actionable"])
|
||||
|
||||
# A kind with no runbook SLA still demotes to chronic, and explains the reset.
|
||||
g = al("Problem with Total GPUs in a System", instance="h9")
|
||||
g.active_at = dt.datetime.now(dt.timezone.utc) - dt.timedelta(minutes=431)
|
||||
g.true_age_minutes, g.true_age_capped = 7 * 24 * 60, False
|
||||
g.screen = screening.screen(g, Snap(total_gpus={"h9": 6}))
|
||||
expect("no-SLA kind, 7 days -> chronic", g.screen["verdict"] == screening.CHRONIC, g.screen)
|
||||
expect("chronic reason explains the activeAt reset", "pipeline dip" in g.screen["detail"], g.screen["detail"])
|
||||
|
||||
print("\nVALIDATION FINDINGS - rogue VM rule defect")
|
||||
r = al(":ninja:Suspected Rogue VM", instance="CA1-ESC8-057")
|
||||
r.screen = screening.screen(r, Snap(rogue_delta={"CA1-ESC8-057": 1.0},
|
||||
in_use_gpus={"CA1-ESC8-057": 8.0},
|
||||
total_gpus={"CA1-ESC8-057": 8.0},
|
||||
resources_by_host={"CA1-ESC8-057": [{}] * 5}))
|
||||
expect("In_Use == Total -> rule defect, not a rogue VM",
|
||||
r.screen["verdict"] == screening.RULE_DEFECT, r.screen)
|
||||
expect("rule defect is screened out of the queue", not r.screen["actionable"])
|
||||
expect("reason names it as spare capacity", "free GPU" in r.screen["reason"], r.screen["reason"])
|
||||
|
||||
r2 = al(":ninja:Suspected Rogue VM", instance="CA1-ESC812-289")
|
||||
r2.screen = screening.screen(r2, Snap(rogue_delta={"CA1-ESC812-289": 2.0},
|
||||
in_use_gpus={"CA1-ESC812-289": 7.0},
|
||||
resources_by_host={"CA1-ESC812-289": [{}] * 4}))
|
||||
expect("In_Use with no Total reading -> still a real gap",
|
||||
r2.screen["verdict"] == screening.REAL, r2.screen)
|
||||
|
||||
r3 = al(":ninja:Suspected Rogue VM", instance="h3")
|
||||
r3.screen = screening.screen(r3, Snap(rogue_delta={"h3": 3.0}, in_use_gpus={"h3": 9.0},
|
||||
total_gpus={"h3": 8.0}, resources_by_host={"h3": [{}]}))
|
||||
expect("In_Use != Total -> real gap", r3.screen["verdict"] == screening.REAL, r3.screen)
|
||||
|
||||
b = real(431)
|
||||
b.true_age_minutes, b.true_age_capped = 7 * 24 * 60, True
|
||||
expect("window-capped age marked with +", b.effective_age_text == "7d+", b.effective_age_text)
|
||||
|
||||
c = real(120)
|
||||
c.true_age_minutes, c.true_age_capped = 130, False
|
||||
expect("small drift is not flagged as a reset", not c.age_is_reset)
|
||||
expect("no true age -> falls back to activeAt", real(90).effective_age_minutes == 90)
|
||||
|
||||
# Ordering must use the recovered duration, not activeAt.
|
||||
old, new = real(431), real(430)
|
||||
old.true_age_minutes = 7 * 24 * 60
|
||||
new.true_age_minutes = 30
|
||||
order = [x["true_age_minutes"] for x in A.group_alerts([old, new])[0]["alerts"]]
|
||||
expect("true age drives ordering, not activeAt", order == [30, 10080], order)
|
||||
|
||||
dips = screening.health_warnings(Snap(pipeline_dips=[
|
||||
{"start": 0, "end": __import__("time").time() - 600, "minutes": 6, "low": 1359, "normal": 4374}]))
|
||||
expect("pipeline dip raises a warning", len(dips) == 1 and "1359 of ~4374" in dips[0], dips)
|
||||
|
||||
print("\n" + ("ALL CHECKS PASSED" if not FAILS else f"{len(FAILS)} CHECK(S) FAILED: {FAILS}"))
|
||||
sys.exit(1 if FAILS else 0)
|
||||
Reference in New Issue
Block a user