`RedisWorkQueue.stats` read XPENDING, which counts entries delivered to a consumer and not yet acknowledged — work in progress. Entries sitting in the stream undelivered were counted nowhere, so an engine hours behind reported itself idle: on the house, `pending: 4` while the group's lag was 1554. The group's own `lag` is the missing number. `backlog` now carries it on both queues (`len(_items)` in memory), leads the health tile, and a sustained one publishes `engine_degraded` from the timer thread — named with the flow most of the waiting work belongs to, sampled from the undelivered tail, since that is the actionable half. It is a summary problem rather than a /utils/health 503: a backlog should not restart the container. Also drops the keyspace `scan_iter` `stats()` did per poll to count parked items — it walked every state and idempotency key twice per ten seconds — for a set the park/unpark path maintains. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01BpfSinyCBfjuieikyfMPbf
434 lines
13 KiB
Python
434 lines
13 KiB
Python
"""The work queue, and what the execution service does with it."""
|
|
|
|
import threading
|
|
import time
|
|
|
|
from fluksio.flow import executor
|
|
from fluksio.flow.executor import ExecutionService
|
|
from fluksio.flow.messages import DType, MessageSpec
|
|
from fluksio.flow.nodes import Node
|
|
from fluksio.flow.pipeline import Pipeline
|
|
from fluksio.flow.queue import MemoryWorkQueue, WorkItem
|
|
from fluksio.flow.state import MemoryState
|
|
|
|
|
|
def test_items_come_back_in_the_order_they_went_in():
|
|
queue = MemoryWorkQueue()
|
|
for i in range(3):
|
|
queue.add(WorkItem(kind="cascade", node=f"f.n{i}", flow="f"))
|
|
|
|
claimed = queue.claim(10, 10)
|
|
|
|
assert [item.node for item in claimed] == ["f.n0", "f.n1", "f.n2"]
|
|
# Every item gets an id, which is what idempotency markers hang off.
|
|
assert all(item.entry_id for item in claimed)
|
|
|
|
|
|
def test_claiming_an_empty_queue_waits_and_gives_up():
|
|
queue = MemoryWorkQueue()
|
|
started = time.monotonic()
|
|
|
|
assert queue.claim(1, 50) == []
|
|
assert time.monotonic() - started >= 0.04
|
|
|
|
|
|
def test_a_delayed_item_stays_put_until_it_is_due():
|
|
queue = MemoryWorkQueue()
|
|
queue.add_delayed(WorkItem(kind="cascade", node="f.n", flow="f"), time.time() + 60)
|
|
|
|
assert queue.claim(1, 10) == []
|
|
assert queue.move_due(time.time()) == 0
|
|
|
|
assert queue.move_due(time.time() + 61) == 1
|
|
assert [i.node for i in queue.claim(1, 10)] == ["f.n"]
|
|
|
|
|
|
def test_parked_work_comes_back_oldest_first():
|
|
queue = MemoryWorkQueue()
|
|
for i in range(3):
|
|
queue.park("heating", WorkItem(kind="cascade", node=f"f.n{i}", flow="heating"))
|
|
|
|
assert [i.node for i in queue.unpark("heating")] == ["f.n0", "f.n1", "f.n2"]
|
|
# Unparking empties it, so a second resume does not replay the same work.
|
|
assert queue.unpark("heating") == []
|
|
|
|
|
|
def test_a_deleted_flow_leaves_nothing_parked():
|
|
queue = MemoryWorkQueue()
|
|
queue.park("gone", WorkItem(kind="cascade", node="gone.n", flow="gone"))
|
|
|
|
queue.clear_flow("gone")
|
|
|
|
assert queue.unpark("gone") == []
|
|
|
|
|
|
def test_claimed_work_counts_as_in_flight_until_it_is_acknowledged():
|
|
"""The health tile's "in flight" reads zero without this."""
|
|
queue = MemoryWorkQueue()
|
|
queue.add(WorkItem(kind="cascade", node="f.n", flow="f"))
|
|
|
|
assert queue.stats()["pending"] == 0
|
|
# The stream length was never a backlog, so the key is gone from both queues.
|
|
assert "depth" not in queue.stats()
|
|
|
|
(item,) = queue.claim(1, 10)
|
|
assert queue.stats()["pending"] == 1
|
|
|
|
queue.ack(item)
|
|
assert queue.stats()["pending"] == 0
|
|
|
|
|
|
def test_work_waiting_to_be_claimed_is_the_backlog():
|
|
"""`pending` is what is running; an engine hours behind reports it as idle."""
|
|
queue = MemoryWorkQueue()
|
|
queue.add(WorkItem(kind="cascade", node="f.n", flow="f"))
|
|
|
|
assert queue.stats()["backlog"] == 1
|
|
assert queue.stats()["pending"] == 0
|
|
|
|
(item,) = queue.claim(1, 10)
|
|
assert queue.stats()["backlog"] == 0
|
|
assert queue.stats()["pending"] == 1
|
|
|
|
queue.ack(item)
|
|
assert queue.stats()["backlog"] == 0
|
|
|
|
|
|
def test_a_sustained_backlog_says_the_engine_is_behind():
|
|
"""A flow enqueuing faster than the pool drains produced no signal at all."""
|
|
events: list[dict] = []
|
|
queue = MemoryWorkQueue()
|
|
service = ExecutionService(queue)
|
|
service._publish = events.append # type: ignore[method-assign]
|
|
for _ in range(executor.BACKLOG_DEGRADED):
|
|
queue.add(WorkItem(kind="cascade", node="f.n", flow="f"))
|
|
|
|
for _ in range(executor.BACKLOG_STRIKES - 1):
|
|
service._check_backlog()
|
|
assert events == []
|
|
|
|
service._check_backlog()
|
|
assert [e["type"] for e in events] == ["engine_degraded"]
|
|
assert service.stats()["behind"] is True
|
|
|
|
# Said once, not once every five seconds for as long as it lasts.
|
|
service._check_backlog()
|
|
assert len(events) == 1
|
|
|
|
# And a drained queue clears it, so the next backlog is announced again.
|
|
queue.claim(executor.BACKLOG_DEGRADED, 10)
|
|
service._check_backlog()
|
|
assert service.stats()["behind"] is False
|
|
|
|
|
|
def _pipeline_with_a_consumer() -> tuple[Pipeline, Node, MemoryState, list]:
|
|
"""A source whose message a consumer records."""
|
|
seen: list[float] = []
|
|
|
|
def consume(reading, params):
|
|
seen.append(reading)
|
|
return {"doubled": reading * 2}
|
|
|
|
source = Node(
|
|
f=lambda params: None,
|
|
provides=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
name="source",
|
|
)
|
|
consumer = Node(
|
|
f=consume,
|
|
requires=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
provides=[MessageSpec(name="doubled", port="doubled", dtype=DType.FLOAT)],
|
|
name="consumer",
|
|
)
|
|
source.assign_flow("f", "source")
|
|
consumer.assign_flow("f", "consumer")
|
|
|
|
state = MemoryState()
|
|
queue = MemoryWorkQueue()
|
|
pipeline = Pipeline(nodes=[source, consumer], state=state, work_queue=queue)
|
|
return pipeline, source, state, seen
|
|
|
|
|
|
def test_a_trigger_is_journaled_rather_than_run_on_the_spot():
|
|
pipeline, source, state, seen = _pipeline_with_a_consumer()
|
|
|
|
source.inject({"reading": 3.0})
|
|
|
|
# Nothing ran yet: the value is in the queue, not in state.
|
|
assert seen == []
|
|
assert "f.reading" not in state
|
|
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == [3.0]
|
|
assert state["f.doubled"] == 6.0
|
|
|
|
|
|
def test_work_for_a_paused_flow_is_held_and_released_on_resume():
|
|
pipeline, source, state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
pipeline.pause("f")
|
|
source.inject({"reading": 1.0})
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == []
|
|
|
|
pipeline.resume("f")
|
|
for item in pipeline._queue.unpark("f"):
|
|
pipeline._queue.add(item)
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == [1.0]
|
|
|
|
|
|
def test_a_step_runs_one_held_item_and_leaves_the_flow_paused():
|
|
pipeline, source, _state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
pipeline.pause("f")
|
|
source.inject({"reading": 1.0})
|
|
source.inject({"reading": 2.0})
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
assert seen == []
|
|
|
|
assert service.step("f") == "f.source"
|
|
|
|
assert seen == [1.0]
|
|
# Still paused, and the second value is still waiting for the next step.
|
|
assert pipeline.is_paused("f")
|
|
assert [i.outputs for i in pipeline._queue.unpark("f")] == [{"f.reading": 2.0}]
|
|
|
|
|
|
def test_stepping_a_flow_with_nothing_held_says_so_rather_than_failing():
|
|
pipeline, _source, _state, _seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
pipeline.pause("f")
|
|
|
|
assert service.step("f") is None
|
|
|
|
|
|
def test_work_for_a_stopped_flow_is_dropped():
|
|
pipeline, source, state, seen = _pipeline_with_a_consumer()
|
|
stopped = Pipeline(
|
|
nodes=pipeline.nodes,
|
|
state=pipeline.state,
|
|
work_queue=pipeline._queue,
|
|
disabled_flows={"f"},
|
|
)
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(stopped)
|
|
|
|
# Reaching the queue at all takes a direct add: trigger drops it earlier.
|
|
stopped._queue.add(
|
|
WorkItem(kind="cascade", node="f.source", flow="f", outputs={"f.reading": 1.0})
|
|
)
|
|
for item in stopped._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == []
|
|
|
|
|
|
def test_an_item_that_keeps_coming_back_is_dead_lettered():
|
|
pipeline, _source, _state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
item = WorkItem(
|
|
kind="cascade",
|
|
node="f.source",
|
|
flow="f",
|
|
outputs={"f.reading": 1.0},
|
|
deliveries=4,
|
|
)
|
|
service._run_item(item)
|
|
|
|
# Given up on rather than run again, so a poison item cannot loop forever.
|
|
assert seen == []
|
|
|
|
|
|
def test_an_item_for_a_node_that_no_longer_exists_is_dropped():
|
|
pipeline, _source, _state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
service._run_item(WorkItem(kind="cascade", node="f.removed", flow="f"))
|
|
|
|
assert seen == []
|
|
|
|
|
|
def test_a_replayed_item_does_not_repeat_a_side_effect():
|
|
"""At-least-once delivery must not mean two of the same outgoing request."""
|
|
calls: list[float] = []
|
|
|
|
def send(reading, params):
|
|
calls.append(reading)
|
|
return None
|
|
|
|
source = Node(
|
|
f=lambda params: None,
|
|
provides=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
name="source",
|
|
)
|
|
|
|
class SendingNode(Node):
|
|
"""Stands in for the built-ins that reach outside."""
|
|
|
|
idempotent = False
|
|
|
|
sender = SendingNode(
|
|
f=send,
|
|
requires=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
name="sender",
|
|
)
|
|
source.assign_flow("f", "source")
|
|
sender.assign_flow("f", "sender")
|
|
|
|
queue = MemoryWorkQueue()
|
|
pipeline = Pipeline(nodes=[source, sender], state=MemoryState(), work_queue=queue)
|
|
service = ExecutionService(queue)
|
|
service.bind(pipeline)
|
|
|
|
source.inject({"reading": 5.0})
|
|
(item,) = queue.claim(10, 10)
|
|
service._run_item(item)
|
|
assert calls == [5.0]
|
|
|
|
# The same item again, as a reaper would hand it back after a crash.
|
|
item.deliveries = 2
|
|
service._run_item(item)
|
|
|
|
assert calls == [5.0]
|
|
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# Telling the queue that a long node is working, not lost
|
|
#
|
|
# What marks an item abandoned is nobody touching it. A node with no timeout
|
|
# may run for hours, so the engine holding it says so on a timer — and an
|
|
# engine that died says nothing, which is the distinction the reaper needs.
|
|
# -----------------------------------------------------------------------------
|
|
|
|
|
|
class RecordingQueue(MemoryWorkQueue):
|
|
"""A memory queue that writes down what it was asked to hold on to."""
|
|
|
|
def __init__(self) -> None:
|
|
super().__init__()
|
|
self.touched: list[list[str]] = []
|
|
|
|
def touch(self, entry_ids: list[str]) -> None:
|
|
self.touched.append(list(entry_ids))
|
|
|
|
|
|
def test_work_in_flight_is_touched_until_it_finishes(monkeypatch):
|
|
monkeypatch.setattr(executor, "TOUCH_INTERVAL_S", 0.0)
|
|
monkeypatch.setattr(executor, "DELAYED_INTERVAL_S", 0.05)
|
|
running = threading.Event()
|
|
release = threading.Event()
|
|
|
|
def slow(reading, params):
|
|
running.set()
|
|
release.wait(5)
|
|
return {"doubled": reading * 2}
|
|
|
|
source = Node(
|
|
f=lambda params: None,
|
|
provides=[MessageSpec(name="reading", dtype=DType.FLOAT)],
|
|
name="source",
|
|
)
|
|
consumer = Node(
|
|
f=slow,
|
|
requires=[MessageSpec(name="reading", dtype=DType.FLOAT)],
|
|
provides=[MessageSpec(name="doubled", dtype=DType.FLOAT)],
|
|
name="consumer",
|
|
)
|
|
source.assign_flow("f", "source")
|
|
consumer.assign_flow("f", "consumer")
|
|
|
|
queue = RecordingQueue()
|
|
pipeline = Pipeline(nodes=[source, consumer], state=MemoryState(), work_queue=queue)
|
|
service = ExecutionService(queue)
|
|
service.bind(pipeline)
|
|
service.start()
|
|
try:
|
|
source.inject({"reading": 3.0})
|
|
assert running.wait(5)
|
|
# Give the timer a couple of passes while the node is still in there.
|
|
time.sleep(0.2)
|
|
held = [ids for ids in queue.touched if ids]
|
|
assert held, "a running item was never touched"
|
|
|
|
release.set()
|
|
time.sleep(0.3)
|
|
# Once it is done it is acknowledged, so there is nothing to hold.
|
|
assert queue.touched[-1] == []
|
|
finally:
|
|
release.set()
|
|
service.stop()
|
|
|
|
|
|
def test_no_more_is_claimed_than_the_pool_can_run():
|
|
"""A backlog belongs in the queue, not inside the process.
|
|
|
|
Claiming ahead of the pool used to leave every waiting item counted as a
|
|
busy cascade and holding its journal entry open, so four cascade threads
|
|
reported hundreds in flight on an engine that was merely behind.
|
|
"""
|
|
release = threading.Event()
|
|
|
|
def slow(reading, params):
|
|
release.wait(5)
|
|
return {"doubled": reading * 2}
|
|
|
|
source = Node(
|
|
f=lambda params: None,
|
|
provides=[MessageSpec(name="reading", dtype=DType.FLOAT)],
|
|
name="source",
|
|
)
|
|
consumer = Node(
|
|
f=slow,
|
|
requires=[MessageSpec(name="reading", dtype=DType.FLOAT)],
|
|
provides=[MessageSpec(name="doubled", dtype=DType.FLOAT)],
|
|
name="consumer",
|
|
)
|
|
source.assign_flow("f", "source")
|
|
consumer.assign_flow("f", "consumer")
|
|
|
|
queue = MemoryWorkQueue()
|
|
pipeline = Pipeline(nodes=[source, consumer], state=MemoryState(), work_queue=queue)
|
|
service = ExecutionService(queue)
|
|
service.bind(pipeline)
|
|
service.start()
|
|
try:
|
|
for i in range(40):
|
|
queue.add(
|
|
WorkItem(
|
|
kind="cascade",
|
|
node="f.source",
|
|
flow="f",
|
|
outputs={"f.reading": float(i)},
|
|
)
|
|
)
|
|
time.sleep(0.5)
|
|
|
|
stats = service.stats()
|
|
assert stats["cascades_busy"] <= executor.MAX_CASCADES
|
|
# And the journal entries of what is only waiting are still free.
|
|
assert stats["pending"] <= executor.MAX_CASCADES
|
|
# Waiting is not idle: the rest of the forty is the backlog.
|
|
assert stats["backlog"] >= 30
|
|
finally:
|
|
release.set()
|
|
service.stop()
|