Split into a FastAPI backend and a React frontend, add case state and SSO
Some checks failed
build-and-deploy / test (push) Has been cancelled
build-and-deploy / image (push) Has been cancelled
build-and-deploy / deploy (push) Has been cancelled

The single-file stdlib server became the limit: no way to track what had been
done about an alert, no accounts, and a UI that had to be hand-rolled in
template strings. This restructures it into something deployable.

Backend (FastAPI)
- app/ holds config, database, auth, delivery and the routers; triagelib keeps
  the triage engine unchanged, so the validated screening and runbook logic is
  untouched.
- Cases persist per alert fingerprint with a status workflow (investigating,
  customer contacted, escalated to Infra, waiting, remediated, resolved, won't
  fix, false positive), an assignee, notes and an append-only history. An alert
  that stops and re-fires lands back on the same case and counts as a reopen.
- Suppression rules move from a JSON file into the database.

Auth
- Signed session cookies over PBKDF2 local accounts, plus an OIDC flow ready for
  Authentik: users are created on first login and admin follows a group claim.
  Local login can be switched off entirely once SSO is live.

Zendesk and Jira
- Delivery is now implemented, behind three gates: the integration must be
  configured, its feature flag on, and CX_FEATURE_SEND_ENABLED on. A demo
  instance leaves the last off and cannot mail anyone. Both search before
  creating, so re-diagnosing an alert updates one ticket rather than opening
  several, and a rolling daily cap stops a loop mailing everybody.

Deployment
- Multi-stage Dockerfile builds the bundle and serves it from the API origin.
- docker-compose for local and single-host use; Gitea Actions runs the tests,
  builds the image and renders deploy/k8s with envsubst.

Two fixes found while testing: assigning a case returned a null assignee, and
add_event could leave an already-loaded history collection stale.

Known gap: the engine reaches OpenStack via `docker exec <region>-osc`, which
does not work in a pod without the CX-Tools containers alongside it.
docs/DEPLOYMENT.md sets out the three ways to close that.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-06 07:11:28 +01:00
parent a039e0b5fd
commit 1262690276
68 changed files with 3839 additions and 2223 deletions

200
backend/app/services.py Normal file
View File

@@ -0,0 +1,200 @@
"""Glue between the triage engine, the database and the outside world."""
from __future__ import annotations
import datetime as dt
import threading
from typing import Any, Optional
from sqlalchemy.orm import Session
from triagelib import alerts as alertlib, screening, settings as legacy_settings
from triagelib.prometheus import (AlertCache, PrometheusClient, PrometheusError,
RuleIndex, StateSnapshot, TrueAgeIndex)
from triagelib import linkage as linkage_mod
from .config import get_settings
from .models import AppSetting, Case, CaseEvent, CaseStatus, SuppressionRule, User
settings = get_settings()
class RuleAdapter:
"""Presents DB-backed suppression rules the way the engine expects."""
def __init__(self, db: Session):
self._rules = [r.to_json() for r in db.query(SuppressionRule).all()]
row = db.get(AppSetting, "general")
general = (row.value if row else {}) or {}
self._agent = str(general.get("agent_name") or "")
self._chronic = int(general.get("chronic_days") or 3)
@property
def rules(self) -> list[dict[str, Any]]:
return self._rules
@property
def agent_name(self) -> str:
return self._agent
@property
def chronic_days(self) -> int:
return self._chronic
class Engine:
"""Process-wide caches over Prometheus. Cheap to share, expensive to rebuild."""
def __init__(self):
self.prom = PrometheusClient(settings.prometheus_base)
self.cache = AlertCache(self.prom)
self.rules = RuleIndex(self.prom)
self.snapshot = StateSnapshot(self.prom)
self.true_age = TrueAgeIndex(self.prom)
self.scan = linkage_mod.Scan()
self._lock = threading.Lock()
def warm(self, log=print) -> None:
try:
self.rules.ensure()
snap = self.snapshot.get()
log(f" state snapshot: {len(snap.by_openstack_id)} VMs, {len(snap.total_gpus)} hosts")
ages = self.true_age.get()
log(f" alert history: {ages.count} alerts indexed over {ages.WINDOW_DAYS} days")
except PrometheusError as exc:
log(f" WARN: Prometheus caches not warmed: {exc}")
def queue(self, db: Session, force: bool = False) -> dict[str, Any]:
raw, error, age = self.cache.get(force=force)
snap = self.snapshot.get()
ages = self.true_age.get()
parsed = [alertlib.from_prometheus(a, self.rules, ages) for a in raw]
excluded = [a for a in parsed if alertlib.is_excluded(a)]
candidates = [a for a in parsed if not alertlib.is_excluded(a)]
cx = [a for a in candidates if a.category == "cx" and alertlib.cx_relevant(a)]
screening.screen_all(cx, snap, RuleAdapter(db))
cases = {c.fingerprint: c for c in
db.query(Case).filter(Case.fingerprint.in_([a.fingerprint() for a in cx])).all()}
groups = alertlib.group_alerts(cx)
for group in groups:
for item in group["alerts"]:
case = cases.get(item["id"])
item["case"] = case.to_json() if case else None
in_cx = {id(a) for a in cx}
infra = sorted([a for a in candidates if id(a) not in in_cx], key=alertlib.sort_key)
return {
"error": error or snap.error or ages.error,
"warnings": screening.health_warnings(snap),
"totals": {"prometheus": len(parsed), "cx": len(cx),
"infrastructure": len(infra), "excluded": len(excluded)},
"excluded_note": (f"{len(excluded)} '{', '.join(sorted({a.alertname for a in excluded}))}' alerts hidden"
if excluded else ""),
"summary": screening.summarize(cx),
"groups": groups,
"infrastructure": _infra_sections(infra),
"cache_age_seconds": round(age, 1),
}
def find_alert(self, db: Session, fingerprint: str) -> Optional[Any]:
raw, _error, _age = self.cache.get()
ages = self.true_age.get()
for item in raw:
alert = alertlib.from_prometheus(item, self.rules, ages)
if alert.fingerprint() == fingerprint:
alert.screen = screening.screen(alert, self.snapshot.get(), RuleAdapter(db))
return alert
return None
SOURCE_LABELS = {
"node-exporter-rules.yml": "Node exporter (hosts)",
"ceph-rules.yml": "Ceph", "mysql-rules.yml": "MySQL",
"mysql-performance-rules.yml": "MySQL performance", "galera-rules.yml": "Galera",
"openstack-rules.yml": "OpenStack services", "blackbox.yml": "Blackbox / OOB",
"infrahub-rules.yml": "Infrahub (no CX runbook)",
}
def _infra_sections(items: list[Any]) -> list[dict[str, Any]]:
buckets: dict[str, list[Any]] = {}
for alert in items:
buckets.setdefault(alert.rule_file or "unknown", []).append(alert)
sections = []
for source, members in buckets.items():
by_name: dict[str, int] = {}
for alert in members:
name = alertlib.clean_alertname(alert.alertname)
by_name[name] = by_name.get(name, 0) + 1
sections.append({
"source": source, "label": SOURCE_LABELS.get(source, source), "total": len(members),
"by_alertname": sorted(({"name": k, "count": v} for k, v in by_name.items()),
key=lambda x: (-x["count"], x["name"])),
})
sections.sort(key=lambda s: (s["source"] != alertlib.NODE_RULE_FILE, -s["total"]))
return sections
# --- case bookkeeping -------------------------------------------------------
def get_or_create_case(db: Session, alert: Any, actor: Optional[User] = None) -> Case:
case = db.query(Case).filter(Case.fingerprint == alert.fingerprint()).one_or_none()
now = dt.datetime.now(dt.timezone.utc)
subject = (alert.floating_ip if alert.kind == "duplicate_ip"
else alert.host if alert.kind in ("rogue_vm", "total_gpus", "orphan_vm")
else alert.instance_name or alert.openstack_id)
if case is None:
case = Case(
fingerprint=alert.fingerprint(), kind=alert.kind, title=alert.title,
subject=subject or "", openstack_id=alert.openstack_id,
instance_name=alert.instance_name, host=alert.host, region=alert.region,
org_id=alert.org_id, org_name=alert.org_name,
)
db.add(case)
db.flush()
add_event(db, case, actor, "opened", f"Case opened for {alert.title}")
else:
# A closed case whose alert has come back is new work again.
if not case.is_open and case.closed_at:
case.reopen_count += 1
case.status = CaseStatus.NEW
case.closed_at = None
add_event(db, case, None, "reopened",
f"Alert fired again after being {case.status.value}")
case.last_seen_at = now
case.title = alert.title
case.subject = subject or case.subject
db.commit()
return case
def add_event(db: Session, case: Case, actor: Optional[User], action: str,
detail: str = "", payload: Optional[dict[str, Any]] = None) -> CaseEvent:
event = CaseEvent(
actor_id=actor.id if actor else None,
actor_label=(actor.email if actor else "system"),
action=action, detail=detail, payload=payload,
)
# Appended through the relationship rather than inserted by id: sessions use
# expire_on_commit=False, so a collection already loaded would otherwise stay
# stale and the new event would be missing from the response.
case.events.append(event)
db.add(event)
return event
def set_status(db: Session, case: Case, status: CaseStatus, actor: Optional[User],
note: str = "") -> Case:
previous = case.status
case.status = status
if status in (CaseStatus.RESOLVED, CaseStatus.WONT_FIX, CaseStatus.FALSE_POSITIVE):
case.closed_at = dt.datetime.now(dt.timezone.utc)
else:
case.closed_at = None
add_event(db, case, actor, "status_changed",
f"{previous.value} -> {status.value}" + (f": {note}" if note else ""),
{"from": previous.value, "to": status.value})
db.commit()
return case