A second bus subscriber folds executions, errors, timings and queue lag into per-minute rollups, keeps failures with their traceback and an audit trail of who published what, and records one row per cascade — manual runs and previews included, under an id of their own that writes no idempotency markers. Read back through /observability/*, which always answers 200 so a degraded engine still renders its own health screen. Also fixes two things found on the way: node-health alerts read `status` where the engine publishes `health`, so a device dropping never alerted anyone, and the Redis queue reported `parked: 0` whatever was held. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_017MeiWk3Yq12n2pTvnQWYvt
105 lines
3.3 KiB
Python
105 lines
3.3 KiB
Python
from datetime import datetime, timedelta, timezone
|
|
|
|
from fastapi.testclient import TestClient
|
|
from sqlmodel import Session
|
|
|
|
from app.core.config import settings
|
|
from app.models import EngineEvent, FlowRun, MetricBucket
|
|
|
|
PREFIX = f"{settings.API_V1_STR}/observability"
|
|
FLOW = "observability-test"
|
|
|
|
|
|
def _seed(db: Session) -> None:
|
|
now = datetime.now(timezone.utc).replace(second=0, microsecond=0)
|
|
db.add(
|
|
MetricBucket(
|
|
flow=FLOW,
|
|
node=f"{FLOW}.calc",
|
|
bucket=now - timedelta(minutes=1),
|
|
executions=4,
|
|
errors=1,
|
|
messages=6,
|
|
duration_sum_ms=40.0,
|
|
duration_max_ms=25.0,
|
|
lag_sum_ms=100.0,
|
|
lag_max_ms=60.0,
|
|
items=4,
|
|
)
|
|
)
|
|
db.add(
|
|
EngineEvent(
|
|
ts=now,
|
|
type="node_error",
|
|
flow=FLOW,
|
|
node=f"{FLOW}.calc",
|
|
detail="ValueError: bad input\nTraceback",
|
|
)
|
|
)
|
|
db.add(
|
|
EngineEvent(
|
|
ts=now, type="audit", flow=FLOW, detail="published", actor="a@example.com"
|
|
)
|
|
)
|
|
db.add(
|
|
FlowRun(
|
|
id="9-0",
|
|
flow=FLOW,
|
|
source="external",
|
|
started_at=now,
|
|
finished_at=now,
|
|
status="ok",
|
|
nodes=3,
|
|
duration_ms=12.5,
|
|
)
|
|
)
|
|
db.commit()
|
|
|
|
|
|
def test_observability_requires_authentication(client: TestClient) -> None:
|
|
assert client.get(f"{PREFIX}/summary").status_code == 401
|
|
|
|
|
|
def test_the_summary_answers_even_when_degraded(
|
|
client: TestClient, superuser_token_headers: dict[str, str]
|
|
) -> None:
|
|
response = client.get(f"{PREFIX}/summary", headers=superuser_token_headers)
|
|
|
|
assert response.status_code == 200
|
|
body = response.json()
|
|
assert body["status"] in {"ok", "degraded"}
|
|
assert set(body["flows"]) == {"total", "running", "paused", "quarantined"}
|
|
assert "error" in body["nodes"]
|
|
|
|
|
|
def test_the_history_reads_back(
|
|
client: TestClient, superuser_token_headers: dict[str, str], db: Session
|
|
) -> None:
|
|
_seed(db)
|
|
|
|
points = client.get(f"{PREFIX}/timeseries", headers=superuser_token_headers).json()
|
|
mine = [point for point in points if point["executions"]]
|
|
assert mine and mine[-1]["avg_ms"] > 0
|
|
|
|
flows = client.get(f"{PREFIX}/flows", headers=superuser_token_headers).json()
|
|
row = next(entry for entry in flows if entry["flow"] == FLOW)
|
|
assert (row["executions"], row["errors"], row["messages"]) == (4, 1, 6)
|
|
assert len(row["spark"]) == 60
|
|
assert row["last_error_ts"] is not None
|
|
|
|
runs = client.get(f"{PREFIX}/runs", headers=superuser_token_headers).json()
|
|
assert any(run["id"] == "9-0" and run["status"] == "ok" for run in runs)
|
|
|
|
failures = client.get(f"{PREFIX}/events", headers=superuser_token_headers).json()
|
|
assert any("Traceback" in event["detail"] for event in failures)
|
|
assert all(event["type"] != "audit" for event in failures)
|
|
|
|
audit = client.get(
|
|
f"{PREFIX}/events", headers=superuser_token_headers, params={"kind": "audit"}
|
|
).json()
|
|
assert any(event["actor"] == "a@example.com" for event in audit)
|
|
|
|
dead = client.get(f"{PREFIX}/dead-letter", headers=superuser_token_headers)
|
|
assert dead.status_code == 200
|
|
assert isinstance(dead.json(), list)
|