Docs / docs (push) Successful in 23s
Playwright Tests / test-playwright (1, 2) (push) Successful in 3m1s
Playwright Tests / test-playwright (2, 2) (push) Successful in 1m44s
pre-commit / pre-commit (push) Failing after 2m50s
Test Backend / test-backend (push) Successful in 2m39s
Compose Smoke Test / test-compose (push) Successful in 31s
Playwright Tests / merge-reports (push) Successful in 1m9s
A house's inverter broker dropped at 04:27 and the power flow was quarantined 20 seconds later. Quarantine was terminal — the supervised task returned and only a publish or an engine restart could bring it back — so five hours of power and battery readings are missing, and what ended it was an unrelated `git pull` restarting uvicorn. Two changes, both in that path: - the failure budget is per task, not per flow. `power` runs an MQTT subscriber and a Victron keepalive publisher against the same broker; they died together and spent one shared budget in 41s, giving up before the 60s backoff step was ever reached. - quarantine is now a rest. The task sits out 5min, then 15, then an hour, and each time gets its budget back and tries again, so a broker that comes back is picked up without anyone watching. `quarantined` reads from whichever tasks are currently resting. The alert for it says when it will try again. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01C5H4uLCCpsbipL1R7WKCee
370 lines
13 KiB
Python
370 lines
13 KiB
Python
"""Failing loudly: engine failures reach a person, not just a log line.
|
|
|
|
Everything that goes wrong already travels the event bus — a node raising, a
|
|
connection dropping, a flow being quarantined, the queue going away. Until now
|
|
the only subscriber was the editor's websocket, so a failure at three in the
|
|
morning was invisible.
|
|
|
|
The alert manager subscribes to the same bus and forwards what matters to a
|
|
channel the operator configured. What it mostly does is *not* send: the same
|
|
node failing every second is one alert, not thirty-six thousand.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
import time
|
|
from collections import deque
|
|
from collections.abc import Callable
|
|
from typing import Any, Literal
|
|
|
|
import httpx
|
|
from pydantic import BaseModel, Field
|
|
|
|
from fluksio.flow.events import EventBus
|
|
from fluksio.flow.secrets import resolve_params
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# What is worth waking someone for. Everything else on the bus is traffic.
|
|
ALERTING_EVENTS = {
|
|
"node_error",
|
|
"node_health",
|
|
"flow_quarantined",
|
|
"task_crashed",
|
|
"engine_degraded",
|
|
"cascade_dropped",
|
|
"queue_unavailable",
|
|
}
|
|
|
|
# The same fault repeating is the same alert.
|
|
DEFAULT_COOLDOWN_S = 900.0
|
|
# A connection flapping is one story, not one alert per transition.
|
|
FLAP_WINDOW_S = 600.0
|
|
FLAP_THRESHOLD = 3
|
|
# However bad it gets, this is the most anyone is told per hour.
|
|
RATE_LIMIT = 10
|
|
RATE_WINDOW_S = 3600.0
|
|
|
|
|
|
class Channel(BaseModel):
|
|
"""Somewhere to send an alert."""
|
|
|
|
name: str
|
|
kind: Literal["ntfy", "smtp", "webhook", "dashboard"]
|
|
enabled: bool = True
|
|
# ntfy: server + topic + optional token. smtp: to. webhook: url.
|
|
# dashboard: message — the record a notification widget reads.
|
|
config: dict[str, Any] = Field(default_factory=dict)
|
|
|
|
|
|
class Rule(BaseModel):
|
|
"""Which events go to which channels."""
|
|
|
|
events: list[str] = Field(default_factory=list)
|
|
channels: list[str] = Field(default_factory=list)
|
|
cooldown_s: float = DEFAULT_COOLDOWN_S
|
|
|
|
def matches(self, event_type: str) -> bool:
|
|
# No list means every alerting event.
|
|
return not self.events or event_type in self.events
|
|
|
|
|
|
class AlertsConfig(BaseModel):
|
|
"""The whole alerting setup, as stored and as the API sees it."""
|
|
|
|
enabled: bool = True
|
|
channels: list[Channel] = Field(default_factory=list)
|
|
rules: list[Rule] = Field(default_factory=list)
|
|
|
|
|
|
class Alert(BaseModel):
|
|
"""What a channel is asked to deliver."""
|
|
|
|
title: str
|
|
body: str
|
|
severity: Literal["warning", "error"] = "error"
|
|
flow: str = ""
|
|
node: str = ""
|
|
|
|
|
|
def describe(event: dict[str, Any]) -> Alert | None:
|
|
"""Turn an engine event into something worth reading, or nothing."""
|
|
kind = event.get("type")
|
|
flow = str(event.get("flow") or "")
|
|
node = str(event.get("node") or "")
|
|
error = str(event.get("error") or event.get("reason") or "")
|
|
|
|
if kind == "node_error":
|
|
return Alert(
|
|
title=f"{node or 'A node'} failed",
|
|
body=error or "The node raised while running.",
|
|
flow=flow,
|
|
node=node,
|
|
)
|
|
if kind == "node_health":
|
|
if event.get("health") != "down":
|
|
return None
|
|
return Alert(
|
|
title=f"{node or 'A node'} lost its connection",
|
|
body=str(event.get("detail") or "Reported itself down."),
|
|
severity="warning",
|
|
flow=flow,
|
|
node=node,
|
|
)
|
|
if kind == "flow_quarantined":
|
|
retry = event.get("retry_in_s")
|
|
again = f" Trying again in {round(float(retry) / 60)} min." if retry else ""
|
|
return Alert(
|
|
title=f"Flow '{flow}' was quarantined",
|
|
body=(
|
|
f"It kept crashing, so the engine stood it down. {error}{again}"
|
|
).strip(),
|
|
flow=flow,
|
|
)
|
|
if kind == "task_crashed":
|
|
return Alert(
|
|
title=f"{event.get('task') or 'A background task'} crashed",
|
|
body=f"{error} Restarting it.",
|
|
severity="warning",
|
|
flow=flow,
|
|
)
|
|
if kind == "engine_degraded":
|
|
return Alert(title="The engine is struggling", body=error or "Degraded.")
|
|
if kind == "cascade_dropped":
|
|
return Alert(
|
|
title="Work was given up on",
|
|
body=(
|
|
f"An item for {node or 'a node'} came back "
|
|
f"{event.get('deliveries')} times and was set aside."
|
|
),
|
|
flow=flow,
|
|
node=node,
|
|
)
|
|
if kind == "queue_unavailable":
|
|
return Alert(title="The work queue is unreachable", body=error or "")
|
|
return None
|
|
|
|
|
|
def dedup_key(event: dict[str, Any]) -> str:
|
|
"""What counts as "the same alert again"."""
|
|
return f"{event.get('type')}:{event.get('node') or event.get('flow') or ''}"
|
|
|
|
|
|
class AlertManager:
|
|
"""Watches the event bus and tells someone when it matters."""
|
|
|
|
def __init__(
|
|
self,
|
|
events: EventBus,
|
|
config: AlertsConfig | None = None,
|
|
now: Any = time.monotonic,
|
|
) -> None:
|
|
self._events = events
|
|
self.config = config or AlertsConfig()
|
|
self._now = now
|
|
#: How a "dashboard" channel puts its alert into the graph. Bound after
|
|
#: construction, because the controller that publishes does not exist
|
|
#: yet when the manager is built.
|
|
self.publish: Callable[[str, Any], None] | None = None
|
|
self._last_sent: dict[str, float] = {}
|
|
self._suppressed: dict[str, int] = {}
|
|
self._health_flips: dict[str, deque[float]] = {}
|
|
self._flapping: dict[str, float] = {}
|
|
self._recent: deque[float] = deque()
|
|
|
|
# -------------------------------------------------------------------------
|
|
# The loop
|
|
# -------------------------------------------------------------------------
|
|
|
|
async def run(self) -> None:
|
|
"""Consume the bus until cancelled."""
|
|
async with self._events.subscribe() as queue:
|
|
while True:
|
|
event = await queue.get()
|
|
try:
|
|
await self.handle(event)
|
|
except Exception:
|
|
logger.exception("Alerting failed for %s", event.get("type"))
|
|
|
|
async def handle(self, event: dict[str, Any]) -> None:
|
|
if not self.config.enabled:
|
|
return
|
|
kind = str(event.get("type") or "")
|
|
if kind not in ALERTING_EVENTS:
|
|
return
|
|
if self._flaps(event):
|
|
return
|
|
|
|
alert = describe(event)
|
|
if alert is None:
|
|
return
|
|
|
|
rules = [r for r in self.config.rules if r.matches(kind)]
|
|
if not rules:
|
|
return
|
|
|
|
key = dedup_key(event)
|
|
cooldown = min(r.cooldown_s for r in rules)
|
|
if not self._due(key, cooldown):
|
|
return
|
|
if not self._within_rate_limit():
|
|
return
|
|
|
|
held = self._suppressed.pop(key, 0)
|
|
if held:
|
|
alert = alert.model_copy(
|
|
update={"body": f"{alert.body} ({held} more since the last alert.)"}
|
|
)
|
|
|
|
names = {name for rule in rules for name in rule.channels}
|
|
for channel in self.config.channels:
|
|
if channel.enabled and channel.name in names:
|
|
await self.send(channel, alert)
|
|
|
|
# -------------------------------------------------------------------------
|
|
# Deciding whether to speak
|
|
# -------------------------------------------------------------------------
|
|
|
|
def _due(self, key: str, cooldown: float) -> bool:
|
|
now = self._now()
|
|
last = self._last_sent.get(key)
|
|
if last is not None and now - last < cooldown:
|
|
self._suppressed[key] = self._suppressed.get(key, 0) + 1
|
|
return False
|
|
self._last_sent[key] = now
|
|
return True
|
|
|
|
def _within_rate_limit(self) -> bool:
|
|
now = self._now()
|
|
while self._recent and self._recent[0] < now - RATE_WINDOW_S:
|
|
self._recent.popleft()
|
|
if len(self._recent) >= RATE_LIMIT:
|
|
logger.warning("Alert rate limit reached; holding back")
|
|
return False
|
|
self._recent.append(now)
|
|
return True
|
|
|
|
def _flaps(self, event: dict[str, Any]) -> bool:
|
|
"""Is this health event part of a connection flapping up and down?
|
|
|
|
A device dropping every ten seconds should produce one alert, then
|
|
silence until it settles.
|
|
"""
|
|
if event.get("type") != "node_health":
|
|
return False
|
|
key = str(event.get("node") or "")
|
|
now = self._now()
|
|
|
|
muted_until = self._flapping.get(key)
|
|
if muted_until is not None:
|
|
if now < muted_until:
|
|
# Still flapping — push the window out and stay quiet.
|
|
if event.get("health") == "down":
|
|
self._flapping[key] = now + FLAP_WINDOW_S
|
|
return True
|
|
del self._flapping[key]
|
|
|
|
flips = self._health_flips.setdefault(key, deque())
|
|
flips.append(now)
|
|
while flips and flips[0] < now - FLAP_WINDOW_S:
|
|
flips.popleft()
|
|
if len(flips) >= FLAP_THRESHOLD * 2:
|
|
self._flapping[key] = now + FLAP_WINDOW_S
|
|
logger.info("Node '%s' is flapping; muting its health alerts", key)
|
|
return False
|
|
|
|
# -------------------------------------------------------------------------
|
|
# Delivery
|
|
# -------------------------------------------------------------------------
|
|
|
|
async def send(
|
|
self, channel: Channel, alert: Alert, raise_on_error: bool = False
|
|
) -> None:
|
|
"""Deliver one alert.
|
|
|
|
A failure is logged and swallowed, because one dead channel must not
|
|
stop the others hearing about the same fault. The test button passes
|
|
``raise_on_error``: telling a working channel from a broken one is the
|
|
only thing it exists for.
|
|
"""
|
|
try:
|
|
config = resolve_params(channel.config)
|
|
except Exception as exc:
|
|
if raise_on_error:
|
|
raise
|
|
logger.error("Channel '%s' has unusable settings: %s", channel.name, exc)
|
|
return
|
|
|
|
try:
|
|
if channel.kind == "ntfy":
|
|
await self._send_ntfy(config, alert)
|
|
elif channel.kind == "webhook":
|
|
await self._send_webhook(config, alert)
|
|
elif channel.kind == "dashboard":
|
|
await self._send_dashboard(config, alert)
|
|
else:
|
|
await self._send_email(config, alert)
|
|
except Exception as exc:
|
|
if raise_on_error:
|
|
raise
|
|
logger.error("Could not alert through '%s': %s", channel.name, exc)
|
|
|
|
async def _send_ntfy(self, config: dict[str, Any], alert: Alert) -> None:
|
|
server = str(config.get("server") or "https://ntfy.sh").rstrip("/")
|
|
topic = config.get("topic")
|
|
if not topic:
|
|
raise ValueError("ntfy needs a topic")
|
|
headers = {
|
|
"Title": alert.title,
|
|
"Priority": "high" if alert.severity == "error" else "default",
|
|
"Tags": "warning" if alert.severity == "warning" else "rotating_light",
|
|
}
|
|
if config.get("token"):
|
|
headers["Authorization"] = f"Bearer {config['token']}"
|
|
async with httpx.AsyncClient(timeout=10) as client:
|
|
response = await client.post(
|
|
f"{server}/{topic}", content=alert.body.encode(), headers=headers
|
|
)
|
|
response.raise_for_status()
|
|
|
|
async def _send_dashboard(self, config: dict[str, Any], alert: Alert) -> None:
|
|
"""Put the alert into the graph, where a notification widget shows it.
|
|
|
|
The message has to be one a flow declares, like any other a dashboard
|
|
writes to — so a panel that shows engine faults says so in a flow
|
|
rather than appearing from nowhere.
|
|
"""
|
|
message = str(config.get("message") or "")
|
|
if not message:
|
|
raise ValueError("a dashboard channel needs a message name")
|
|
if self.publish is None:
|
|
raise RuntimeError("nothing is wired up to publish this")
|
|
# Blocking: it reads the store to find the declared message.
|
|
await asyncio.to_thread(self.publish, message, alert.model_dump())
|
|
|
|
async def _send_webhook(self, config: dict[str, Any], alert: Alert) -> None:
|
|
url = config.get("url")
|
|
if not url:
|
|
raise ValueError("a webhook channel needs a url")
|
|
async with httpx.AsyncClient(timeout=10) as client:
|
|
response = await client.post(str(url), json=alert.model_dump())
|
|
response.raise_for_status()
|
|
|
|
async def _send_email(self, config: dict[str, Any], alert: Alert) -> None:
|
|
from fluksio.core.config import settings
|
|
from fluksio.utils import send_email
|
|
|
|
recipient = config.get("to")
|
|
if not recipient:
|
|
raise ValueError("an smtp channel needs a recipient")
|
|
if not settings.emails_enabled:
|
|
raise RuntimeError("no SMTP configuration")
|
|
await asyncio.to_thread(
|
|
send_email,
|
|
email_to=str(recipient),
|
|
subject=alert.title,
|
|
html_content=f"<p>{alert.body}</p>",
|
|
)
|