Files
cx-ui/tests/test_screening.py
Parham Monfared a039e0b5fd CX Triage: alert diagnosis over the CX-Tools collectors
Read-only triage for the Infrahub error alerts. Pulls the Prometheus alert
queue, re-checks each alert's condition against live state to separate real
work from noise, diagnoses it using the CX runbooks, and drafts the customer
comms with contacts resolved from Infrahub.

Findings from validating against production:
- "Suspected Rogue VM" fires on spare GPU capacity, not rogue VMs: In_Use_Gpus
  equals the physical count on 71 of 75 firing hosts, so the rule reduces to
  "this host has a free GPU". Verified against OpenStack on 10 hosts.
- "Exists in Infrahub but does not exist in OpenStack" matches every VM because
  openstack_nova_server_status returns no series; excluded as a rule defect.
- Prometheus activeAt is reset several times a day by dips in the Resources
  metric, so alert ages are recovered from ALERTS history instead.

Takes ~2,650 firing alerts down to ~20 that need a decision.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-06 06:48:34 +01:00

291 lines
15 KiB
Python

"""Screening, exclusion, categorisation and ordering tests.
These cover the noise-vs-real decisions, which are what keeps the queue small.
No network and no CX-Tools: snapshots are synthetic.
"""
import datetime as dt
import os
import sys
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
from triagelib import alerts as A, screening
class Snap:
"""Stands in for prometheus.StateSnapshot."""
loaded = True
def __init__(self, **kw):
self.by_openstack_id = {}
self.by_instance_name = {}
self.fip_counts = {}
self.rogue_delta = {}
self.total_gpus = {}
self.in_use_gpus = {}
self.resources_by_host = {}
self.broken_inputs = []
self.unattributed_active = 0
self.unattributed_active_gpus = 0
self.__dict__.update(kw)
def al(name, **labels):
return A.from_labels({"alertname": name, **labels})
def aged(alert, minutes):
alert.active_at = dt.datetime.now(dt.timezone.utc) - dt.timedelta(minutes=minutes)
return alert
FAILS = []
def expect(label, cond, got=""):
print((" PASS " if cond else " FAIL ") + label + ("" if cond else f" <- {got}"))
if not cond:
FAILS.append(label)
OSID = "abc-123"
ERROR_ALERT = "Instance in ERROR state in :flag-ca:CA-1"
SHUTOFF_ALERT = "Instance in SHUTOFF state in :flag-ca:CA-1 for greater than 30 min"
print("\nSTATE ALERTS - does the condition still hold?")
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", region="CANADA-1")
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ERROR"}}))
expect("Infrahub still ERROR -> real", a.screen["verdict"] == screening.REAL, a.screen)
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", region="CANADA-1")
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ACTIVE"}}))
expect("recovered to ACTIVE -> resolved",
a.screen["verdict"] == screening.RESOLVED and "ACTIVE" in a.screen["reason"], a.screen)
a = al(SHUTOFF_ALERT, openstack_id=OSID, status="SHUTOFF")
a.screen = screening.screen(a, Snap())
expect("record gone from Infrahub -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
a = al("Instance in CREATING state in :flag-ca:CA-1 for greater than 30min",
openstack_id="None", instance_name="vm-x", status="CREATING")
a.screen = screening.screen(a, Snap())
expect("CREATING with no OpenStack ID -> real, not 'resolved'",
a.screen["verdict"] == screening.REAL, a.screen)
a = al(ERROR_ALERT, openstack_id=OSID)
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ERROR"}}))
expect("no status label -> unverified but kept", a.screen["verdict"] == screening.UNVERIFIED, a.screen)
print("\nDUPLICATED IPs")
a = al(":awkward:Duplicated IPs", floating_ip="1.2.3.4")
a.screen = screening.screen(a, Snap(fip_counts={"1.2.3.4": 1}))
expect("one claimant left -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
a.screen = screening.screen(a, Snap(fip_counts={"1.2.3.4": 3}))
expect("three claimants -> real", a.screen["verdict"] == screening.REAL and "3 VMs" in a.screen["reason"], a.screen)
a.screen = screening.screen(a, Snap(fip_counts={"9.9.9.9": 2}))
expect("IP held by nobody -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
print("\nSUSPECTED ROGUE VM - per-host GPU accounting gap")
a = al(":ninja:Suspected Rogue VM", instance="CA1-ESC8-068")
a.screen = screening.screen(a, Snap(rogue_delta={"CA1-ESC8-068": 0.0}))
expect("gap closed -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
a.screen = screening.screen(a, Snap(rogue_delta={"CA1-ESC8-068": 4.0},
resources_by_host={"CA1-ESC8-068": [{}] * 4}))
expect("gap of 4 GPUs -> real", a.screen["verdict"] == screening.REAL, a.screen)
expect("reason quantifies the gap", "4 GPU(s)" in a.screen["reason"], a.screen["reason"])
a.screen = screening.screen(a, Snap(rogue_delta={"other-host": 4.0}, total_gpus={"x": 8}))
expect("no data for the host -> unverified, still actionable",
a.screen["verdict"] == screening.UNVERIFIED and a.screen["actionable"], a.screen)
print("\nTOTAL GPUs")
a = al("Problem with Total GPUs in a System", instance="h1", gpu_name="B200-SXM")
a.screen = screening.screen(a, Snap(total_gpus={"h1": 8}))
expect("full complement of 8 -> resolved", a.screen["verdict"] == screening.RESOLVED, a.screen)
a.screen = screening.screen(a, Snap(total_gpus={"h1": 6}, in_use_gpus={"h1": 6}))
expect("6 GPUs -> real", a.screen["verdict"] == screening.REAL, a.screen)
print("\nSUPPRESSION RULES - what used to be hardcoded is now user-editable")
from triagelib import settings as settings_mod
import tempfile, os as _os
_tmp = _os.path.join(tempfile.mkdtemp(), "settings.json")
CFG = settings_mod.Settings(_tmp)
a = al(SHUTOFF_ALERT, openstack_id=OSID, status="SHUTOFF",
organization="3491 - luis.sarabando+runpod@nexgencloud.coms-Organization")
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "SHUTOFF"}}), CFG)
expect("default rule hides internal nexgencloud orgs",
a.screen["verdict"] == screening.SUPPRESSED, a.screen)
expect("suppression names the rule that did it", "Internal NexGen" in a.screen["reason"], a.screen["reason"])
a = al("Instance in SHUTOFF state in :flag-no:NO-1 for greater than 30 min", openstack_id=OSID,
status="SHUTOFF", instance="no1-stor-runpod03", instance_name="no1-stor-runpod03",
organization="99 - Real Customer")
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "SHUTOFF"}}), CFG)
expect("default rule hides runpod storage nodes", a.screen["verdict"] == screening.SUPPRESSED, a.screen)
# The combinational case the team asked for: type AND organisation.
CFG.upsert_rule({"name": "Modal ERROR churn", "reason": "known batch churn",
"conditions": {"kind": ["error"], "organization": ["modal"]}})
hit = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", organization="19417 - colin@modal.coms-Organization")
hit.screen = screening.screen(hit, Snap(by_openstack_id={OSID: {"status": "ERROR"}}), CFG)
expect("error + modal is suppressed", hit.screen["verdict"] == screening.SUPPRESSED, hit.screen)
miss = al(ERROR_ALERT, openstack_id=OSID, status="ERROR", organization="123 - Someone Else")
miss.screen = screening.screen(miss, Snap(by_openstack_id={OSID: {"status": "ERROR"}}), CFG)
expect("error from another org is NOT suppressed", miss.screen["verdict"] != screening.SUPPRESSED, miss.screen)
other = al(SHUTOFF_ALERT, openstack_id=OSID, status="SHUTOFF",
organization="19417 - colin@modal.coms-Organization")
other.screen = screening.screen(other, Snap(by_openstack_id={OSID: {"status": "SHUTOFF"}}), CFG)
expect("modal SHUTOFF is NOT suppressed - both conditions must match",
other.screen["verdict"] != screening.SUPPRESSED, other.screen)
empty = {"name": "catch all", "conditions": {}}
expect("a rule with no conditions never matches", not settings_mod.rule_matches(
settings_mod._normalize_rule(empty), hit))
expect("rules survive a reload", settings_mod.Settings(_tmp).rules and any(
r["name"] == "Modal ERROR churn" for r in settings_mod.Settings(_tmp).rules))
print("\nAGE DEMOTIONS")
a = aged(al("Problem with Total GPUs in a System", instance="h1"), 7 * 24 * 60)
a.screen = screening.screen(a, Snap(total_gpus={"h1": 6}))
expect("firing 7 days -> chronic", a.screen["verdict"] == screening.CHRONIC, a.screen)
a = aged(al("Problem with Total GPUs in a System", instance="h1"), 60)
a.screen = screening.screen(a, Snap(total_gpus={"h1": 6}))
expect("firing 1 hour -> stays real", a.screen["verdict"] == screening.REAL, a.screen)
print("\nFAIL-SAFE BEHAVIOUR")
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR")
a.state = "pending"
a.for_seconds = 1800
a.screen = screening.screen(a, Snap(by_openstack_id={OSID: {"status": "ERROR"}}))
expect("pending -> screened out", a.screen["verdict"] == screening.PENDING and not a.screen["actionable"], a.screen)
a = al(ERROR_ALERT, openstack_id=OSID, status="ERROR")
a.screen = screening.screen(a, None)
expect("no snapshot -> unverified but NOT hidden",
a.screen["verdict"] == screening.UNVERIFIED and a.screen["actionable"], a.screen)
warnings = screening.health_warnings(Snap(broken_inputs=["openstack_nova_server_status"]))
expect("empty nova metric raises a monitoring warning",
len(warnings) == 1 and "openstack_nova_server_status" in warnings[0], warnings)
print("\nEXCLUSION AND TAB ROUTING")
ex = al("Exists in Infrahub but does not exist in OpenStack", openstack_id=OSID)
expect("orphan spam excluded outright", A.is_excluded(ex) and not A.cx_relevant(ex))
expect("node-exporter -> Infrastructure tab",
A.category("HostSwapIsFillingUp", "node-exporter-rules.yml") == "node")
expect("ceph -> Infrastructure tab", A.category("CephOsdDown", "ceph-rules.yml") == "infra")
expect("regional Infrahub rule -> CX tab",
A.category(ERROR_ALERT, "infrahub-rules-CA1.yml") == "cx")
expect("main Infrahub rule -> CX tab",
A.category(":ninja:Suspected Rogue VM", "infrahub-rules.yml") == "cx")
expect("status-mismatch rule classified",
A.classify("Openstack status=ACTIVE and Infrahub status!=ACTIVE in :flag-ca:CANADA-1 for greater "
"than 30min") == "status_mismatch")
expect("hibernation-failure rule maps to HIBERNATING",
A.classify("Failure of Hibernation on Infrahub in :flag-ca:CANADA-1 for greater than 30min") == "hibernating")
expect("orphan VM rule classified", A.classify(":pirate_flag:Suspected Orphan VM") == "orphan_vm")
print("\nORDERING AND GROUPING")
def real(minutes, name=ERROR_ALERT, **labels):
x = aged(al(name, openstack_id="o%d" % minutes, status="ERROR", **labels), minutes)
x.screen = {"actionable": True, "verdict": "real", "label": "needs action", "reason": ""}
return x
ages = [x["age_minutes"] for x in A.group_alerts([real(500), real(10), real(100), real(9331)])[0]["alerts"]]
expect("newest first, oldest at the bottom", ages == [10, 100, 500, 9331], ages)
unknown = real(50)
unknown.active_at = None
ages = [x["age_minutes"] for x in A.group_alerts([unknown, real(200), real(5)])[0]["alerts"]]
expect("unknown start time sorts last", ages == [5, 200, None], ages)
groups = A.group_alerts([real(5), real(6, ":ninja:Suspected Rogue VM", instance="h1")])
expect("focus order puts rogue VM before ERROR", [g["kind"] for g in groups][0] == "rogue_vm",
[g["kind"] for g in groups])
noisy = real(7)
noisy.screen = {"actionable": False, "verdict": "resolved", "label": "already resolved", "reason": ""}
group = A.group_alerts([real(5), noisy])[0]
expect("group counts action vs noise separately",
group["actionable"] == 1 and group["noise"] == 1, group)
expect("age_text renders days", real(9331).age_text == "6d 11h", real(9331).age_text)
expect("age_text renders hours", real(431).age_text == "7h 11m", real(431).age_text)
expect("age_text renders minutes", real(7).age_text == "7m", real(7).age_text)
print("\nTRUE AGE - activeAt reset by pipeline dips")
# activeAt says 7h; ALERTS history says 7 days. The true value must win.
a = real(431)
a.true_age_minutes, a.true_age_capped = 7 * 24 * 60, False
expect("effective age prefers the recovered duration", a.effective_age_minutes == 10080, a.effective_age_minutes)
expect("reset is detected", a.age_is_reset)
expect("raw activeAt still reported", a.age_text == "7h 11m", a.age_text)
expect("effective text renders days", a.effective_age_text == "7d", a.effective_age_text)
a.screen = screening.screen(a, Snap(by_openstack_id={"o431": {"status": "ERROR"}}))
expect("7-day ERROR -> overdue (runbook says contact within 24h), not chronic",
a.screen["verdict"] == screening.OVERDUE, a.screen)
expect("overdue stays in the actionable queue", a.screen["actionable"])
# A kind with no runbook SLA still demotes to chronic, and explains the reset.
g = al("Problem with Total GPUs in a System", instance="h9")
g.active_at = dt.datetime.now(dt.timezone.utc) - dt.timedelta(minutes=431)
g.true_age_minutes, g.true_age_capped = 7 * 24 * 60, False
g.screen = screening.screen(g, Snap(total_gpus={"h9": 6}))
expect("no-SLA kind, 7 days -> chronic", g.screen["verdict"] == screening.CHRONIC, g.screen)
expect("chronic reason explains the activeAt reset", "pipeline dip" in g.screen["detail"], g.screen["detail"])
print("\nVALIDATION FINDINGS - rogue VM rule defect")
r = al(":ninja:Suspected Rogue VM", instance="CA1-ESC8-057")
r.screen = screening.screen(r, Snap(rogue_delta={"CA1-ESC8-057": 1.0},
in_use_gpus={"CA1-ESC8-057": 8.0},
total_gpus={"CA1-ESC8-057": 8.0},
resources_by_host={"CA1-ESC8-057": [{}] * 5}))
expect("In_Use == Total -> rule defect, not a rogue VM",
r.screen["verdict"] == screening.RULE_DEFECT, r.screen)
expect("rule defect is screened out of the queue", not r.screen["actionable"])
expect("reason names it as spare capacity", "free GPU" in r.screen["reason"], r.screen["reason"])
r2 = al(":ninja:Suspected Rogue VM", instance="CA1-ESC812-289")
r2.screen = screening.screen(r2, Snap(rogue_delta={"CA1-ESC812-289": 2.0},
in_use_gpus={"CA1-ESC812-289": 7.0},
resources_by_host={"CA1-ESC812-289": [{}] * 4}))
expect("In_Use with no Total reading -> still a real gap",
r2.screen["verdict"] == screening.REAL, r2.screen)
r3 = al(":ninja:Suspected Rogue VM", instance="h3")
r3.screen = screening.screen(r3, Snap(rogue_delta={"h3": 3.0}, in_use_gpus={"h3": 9.0},
total_gpus={"h3": 8.0}, resources_by_host={"h3": [{}]}))
expect("In_Use != Total -> real gap", r3.screen["verdict"] == screening.REAL, r3.screen)
b = real(431)
b.true_age_minutes, b.true_age_capped = 7 * 24 * 60, True
expect("window-capped age marked with +", b.effective_age_text == "7d+", b.effective_age_text)
c = real(120)
c.true_age_minutes, c.true_age_capped = 130, False
expect("small drift is not flagged as a reset", not c.age_is_reset)
expect("no true age -> falls back to activeAt", real(90).effective_age_minutes == 90)
# Ordering must use the recovered duration, not activeAt.
old, new = real(431), real(430)
old.true_age_minutes = 7 * 24 * 60
new.true_age_minutes = 30
order = [x["true_age_minutes"] for x in A.group_alerts([old, new])[0]["alerts"]]
expect("true age drives ordering, not activeAt", order == [30, 10080], order)
dips = screening.health_warnings(Snap(pipeline_dips=[
{"start": 0, "end": __import__("time").time() - 600, "minutes": 6, "low": 1359, "normal": 4374}]))
expect("pipeline dip raises a warning", len(dips) == 1 and "1359 of ~4374" in dips[0], dips)
print("\n" + ("ALL CHECKS PASSED" if not FAILS else f"{len(FAILS)} CHECK(S) FAILED: {FAILS}"))
sys.exit(1 if FAILS else 0)