Files
app/backend/fluksio/flow/supervision.py
T
stroblmeandClaude Opus 5 8f1e685526
Docs / docs (push) Successful in 23s
Playwright Tests / test-playwright (1, 2) (push) Successful in 3m1s
Playwright Tests / test-playwright (2, 2) (push) Successful in 1m44s
pre-commit / pre-commit (push) Failing after 2m50s
Test Backend / test-backend (push) Successful in 2m39s
Compose Smoke Test / test-compose (push) Successful in 31s
Playwright Tests / merge-reports (push) Successful in 1m9s
Let a quarantine expire, and stop two tasks spending one budget
A house's inverter broker dropped at 04:27 and the power flow was quarantined
20 seconds later. Quarantine was terminal — the supervised task returned and
only a publish or an engine restart could bring it back — so five hours of
power and battery readings are missing, and what ended it was an unrelated
`git pull` restarting uvicorn.

Two changes, both in that path:

- the failure budget is per task, not per flow. `power` runs an MQTT subscriber
  and a Victron keepalive publisher against the same broker; they died together
  and spent one shared budget in 41s, giving up before the 60s backoff step was
  ever reached.
- quarantine is now a rest. The task sits out 5min, then 15, then an hour, and
  each time gets its budget back and tries again, so a broker that comes back
  is picked up without anyone watching. `quarantined` reads from whichever
  tasks are currently resting.

The alert for it says when it will try again.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01C5H4uLCCpsbipL1R7WKCee
2026-08-28 10:35:13 +02:00

216 lines
8.9 KiB
Python

"""Supervision for the long-lived tasks a pipeline starts.
A node's subscription, schedule or poll loop is an `asyncio.Task`, and a task
that raises is simply gone — the node stays listed as running while nothing
listens any more. The supervisor restarts those tasks with a growing delay and
stands a flow down when one keeps failing, because a flow crash-looping every
second is worse than a flow that is visibly stopped.
Standing down is a rest, not a verdict. It used to be the end of the task: the
flow stayed dead until somebody published it or the engine restarted. On a
house that runs unattended that turns a broker rebooting at four in the morning
into a hole in the record until someone notices — so the rest expires, the
budget comes back, and the task tries again on a much longer clock.
Deliberately a plain task registry rather than a `TaskGroup`: a group cancels
its siblings when one member fails, which is the opposite of what supervision
means here.
"""
from __future__ import annotations
import asyncio
import logging
import time
from collections import deque
from collections.abc import Callable, Coroutine
from typing import Any
from fluksio.flow.events import EventBus
logger = logging.getLogger(__name__)
BACKOFF = (1.0, 5.0, 30.0, 60.0)
# A task that burns through this many restarts in the window is not going to
# recover by being restarted again *now*.
#
# Per task rather than per flow: a flow whose subscriber and its keepalive
# publisher both go down with the same broker spends one shared budget twice as
# fast as it was sized for, and gives up before the backoff above has reached
# its longest step. That is what took the house's inverter readings out for
# nearly five hours — six crashes across two tasks inside 41 seconds.
FAILURE_BUDGET = 5
FAILURE_WINDOW = 300.0
# How long a task sits out before its budget is handed back. Minutes rather
# than seconds because a device that is really gone should not be hammered, and
# capped in the hour because nothing here is worth losing a day of readings to.
QUARANTINE_BACKOFF = (300.0, 900.0, 3600.0)
# How long a cancelled task gets to notice. A loop that is still waiting after
# this is not going to stop on its own — a client closing a socket the broker
# no longer answers on is the case seen in the wild — and the rebuild asking
# for it must not wait on that forever.
CANCEL_GRACE = 5.0
TaskFactory = Callable[[], Coroutine[Any, Any, None]]
class Supervisor:
"""Keeps the pipeline's background tasks alive, or admits it cannot."""
def __init__(self, events: EventBus | None = None) -> None:
self._events = events
self._tasks: dict[str, asyncio.Task[None]] = {}
#: Which flow each task belongs to. Recorded rather than read off the
#: task's name, because names are a flow and a node joined by a dot and
#: matching on that prefix would let 'hea' cancel 'heating'.
self._flows: dict[str, str] = {}
#: Crash times per *task*, keyed by task name.
self._failures: dict[str, deque[float]] = {}
#: The tasks sitting out a quarantine, and the flow each belongs to.
self._resting: dict[str, str] = {}
@property
def quarantined(self) -> set[str]:
"""The flows with at least one task currently standing down."""
return set(self._resting.values())
def spawn(self, name: str, flow: str, factory: TaskFactory) -> None:
"""Run `factory()` and keep running it until told to stop."""
if name in self._tasks:
return
self._flows[name] = flow
self._tasks[name] = asyncio.create_task(
self._supervise(name, flow, factory), name=f"supervised:{name}"
)
async def _supervise(self, name: str, flow: str, factory: TaskFactory) -> None:
attempt = 0
rests = 0
while True:
try:
await factory()
except asyncio.CancelledError:
raise
except Exception as exc:
if not self._record_failure(name, flow, exc):
await self._rest(name, flow, exc, rests)
rests += 1
attempt = 0
continue
else:
# A clean return means the loop decided it was done.
return
delay = BACKOFF[min(attempt, len(BACKOFF) - 1)]
attempt += 1
await asyncio.sleep(delay)
async def _rest(self, name: str, flow: str, exc: Exception, rests: int) -> None:
"""Stand the task down, then hand its budget back and let it retry."""
delay = QUARANTINE_BACKOFF[min(rests, len(QUARANTINE_BACKOFF) - 1)]
logger.error(
"Flow '%s' crashed %d times in %.0fs — quarantined, retrying in %.0fs",
flow,
FAILURE_BUDGET,
FAILURE_WINDOW,
delay,
)
self._resting[name] = flow
self._publish(
{
"type": "flow_quarantined",
"flow": flow,
"task": name,
"error": f"{type(exc).__name__}: {exc}",
"retry_in_s": delay,
"ts": time.time(),
}
)
try:
await asyncio.sleep(delay)
finally:
# Both, and in this order: a task woken by a cancellation is being
# torn down, and must not leave the flow looking quarantined.
self._resting.pop(name, None)
self._failures.pop(name, None)
def _record_failure(self, name: str, flow: str, exc: Exception) -> bool:
"""Note the crash; False when this task has spent its budget."""
logger.warning("Supervised task '%s' crashed: %s", name, exc, exc_info=True)
self._publish(
{
"type": "task_crashed",
"task": name,
"flow": flow,
"error": f"{type(exc).__name__}: {exc}",
"ts": time.time(),
}
)
now = time.monotonic()
window = self._failures.setdefault(name, deque())
window.append(now)
while window and window[0] < now - FAILURE_WINDOW:
window.popleft()
return len(window) < FAILURE_BUDGET
async def cancel_all(self) -> None:
"""Stop supervising. Idempotent, and safe to call mid-restart."""
tasks = list(self._tasks.values())
self._tasks.clear()
self._flows.clear()
self._resting.clear()
self._failures.clear()
await self._cancel(tasks)
async def cancel_flow(self, flow: str) -> None:
"""Stop one flow's supervised tasks and give it a clean slate.
Rebuilding the whole pipeline throws the supervisor away and builds
another, so a flow quarantined by the last build gets another chance.
Rebuilding one flow has to say the same thing about that flow alone, or
every other flow's quarantine would go with it.
"""
names = [name for name, owner in self._flows.items() if owner == flow]
tasks = [self._tasks.pop(name) for name in names if name in self._tasks]
for name in names:
del self._flows[name]
# The clean slate: whatever these tasks spent before, the build
# that follows starts their budgets again — and a rebuild is not
# made to wait out a rest somebody has just fixed the cause of.
self._failures.pop(name, None)
self._resting.pop(name, None)
await self._cancel(tasks)
async def _cancel(self, tasks: list[asyncio.Task[None]]) -> None:
"""Ask these tasks to stop, and wait no longer than the grace period."""
for task in tasks:
task.cancel()
if not tasks:
return
# `wait` hands back what is still going instead of waiting on it, and
# lets a cancellation aimed at *this* coroutine through — the
# `except CancelledError` it replaces swallowed that, which left
# whoever asked for the teardown holding their lock and unkillable.
done, pending = await asyncio.wait(tasks, timeout=CANCEL_GRACE)
for task in done:
if not task.cancelled():
# Retrieved so a crash on the way out is not reported at exit;
# the supervisor has already said what it was.
task.exception()
for task in pending:
# Cancelled once and still running means its shutdown is waiting on
# something that is not answering. A second cancellation interrupts
# that wait; whether it takes is no longer the rebuild's problem.
task.cancel()
logger.warning(
"Supervised task '%s' did not stop within %.0fs — abandoned",
task.get_name(),
CANCEL_GRACE,
)
def _publish(self, event: dict[str, Any]) -> None:
if self._events is not None:
self._events.publish(event)