"""Turning RunPod machine state into tracked history. The API only ever reports what is true now. Everything CX cares about - how often a machine has dropped out, when it drained, who relisted it - only exists if each sync writes down what changed. """ from __future__ import annotations import datetime as dt from typing import Any, Optional from sqlalchemy import select from sqlalchemy.orm import Session from ..models import RunpodColour, RunpodEvent, RunpodEventType, RunpodHost def record_event(db: Session, host: RunpodHost, event_type: RunpodEventType, *, actor: str = "system", detail: str = "", error_hint: str = "", zendesk_ticket: str = "", jira_key: str = "", gpu_reserved: Optional[int] = None, payload: Optional[dict[str, Any]] = None) -> RunpodEvent: event = RunpodEvent( event_type=event_type, actor=actor, detail=detail, error_hint=error_hint, zendesk_ticket=zendesk_ticket, jira_key=jira_key, gpu_reserved=host.gpu_reserved if gpu_reserved is None else gpu_reserved, payload=payload, ) # Through the relationship so an already-loaded history stays correct. host.events.append(event) db.add(event) return event def sync_machines(db: Session, client: Any, actor: str = "system") -> dict[str, Any]: """Pull current machines and write down every transition since last time.""" machines = client.machines() existing = {h.machine_id: h for h in db.scalars(select(RunpodHost)).all()} now = dt.datetime.now(dt.timezone.utc) created = unlisted = relisted = drained = 0 for machine in machines: machine_id = str(machine.get("id") or "") if not machine_id: continue listed = bool(machine.get("listed")) reserved = int(machine.get("gpuReserved") or 0) host = existing.get(machine_id) if host is None: host = RunpodHost(machine_id=machine_id, name=str(machine.get("name") or "")) db.add(host) db.flush() created += 1 record_event(db, host, RunpodEventType.LISTED if listed else RunpodEventType.UNLISTED, actor="runpod", detail="First seen by CX Triage") if not listed: host.unlisted_at = now was_listed = host.listed was_reserved = host.gpu_reserved if was_listed and not listed: unlisted += 1 host.unlisted_at = now host.unlist_count += 1 # Nothing in the API says why; the email carries the reason and is # attached separately through /ingest-email. record_event(db, host, RunpodEventType.UNLISTED, actor="runpod", detail="Unlisted (detected on sync)", gpu_reserved=reserved) elif not was_listed and listed: relisted += 1 host.last_listed_at = now host.unlisted_at = None record_event(db, host, RunpodEventType.LISTED, actor=actor, detail="Relisted", gpu_reserved=reserved) # Unlisted and the last renter has gone: the machine is now safe to work on. if not listed and was_reserved > 0 and reserved == 0: drained += 1 record_event(db, host, RunpodEventType.DRAINED, actor="runpod", detail="Unlisted machine has drained - no rented GPUs left", gpu_reserved=0) gpu_type = machine.get("gpuType") or {} host.name = str(machine.get("name") or host.name) host.listed = listed host.gpu_reserved = reserved host.gpu_total = int(machine.get("gpuTotal") or 0) host.gpu_type = str(gpu_type.get("displayName") or machine.get("gpuTypeId") or "") host.data_center = str(machine.get("dataCenterId") or "") host.uptime_one_week = machine.get("uptimePercentListedOneWeek") host.maintenance_start = str(machine.get("maintenanceStart") or "") host.maintenance_end = str(machine.get("maintenanceEnd") or "") db.commit() return { "machines": len(machines), "created": created, "unlisted": unlisted, "relisted": relisted, "drained": drained, "synced_at": now.isoformat(), }