Files
app/backend/tests/flow/test_queue.py
T
stroblmeandClaude Opus 5 93374a310e Refuse what a node cannot publish, and stop timing out work that is fine
Four things the python SDK turned up, each fixed where every client sees it.

A key no port declares is now an error rather than a silent drop, on the
return, the yield and the emit alike — the contract the docs already stated.
The SDK reads literal yields at sync time, so a typo fails before anything
runs, and an emission of one fails the call rather than being logged where
nobody looks.

NaN and infinity are refused at the port. JSON cannot spell either, so one
that travelled came back as a 500, a socket frame that stopped the canvas, or
a metric batch the database dropped whole.

An artifact input takes `@run:<id>.<output>` or a bare digest, resolved on the
engine — so the CLI, the run dialog and a python caller mean the same thing,
and a sweep can pass one at all.

Node timeouts are off by default. The clock measured silence, which a training
node is full of, and remote workers had already stopped enforcing it — their
heartbeat reset it. Now a heartbeat proves the agent rather than the node,
ninety seconds of nothing fails the call either way, and the engine touches
work it is still running so a long node is not redelivered at sixty seconds.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019V5bsYGNxcgPs4xXmTPx69
2026-08-25 07:30:14 +02:00

338 lines
10 KiB
Python

"""The work queue, and what the execution service does with it."""
import threading
import time
from fluksio.flow import executor
from fluksio.flow.executor import ExecutionService
from fluksio.flow.messages import DType, MessageSpec
from fluksio.flow.nodes import Node
from fluksio.flow.pipeline import Pipeline
from fluksio.flow.queue import MemoryWorkQueue, WorkItem
from fluksio.flow.state import MemoryState
def test_items_come_back_in_the_order_they_went_in():
queue = MemoryWorkQueue()
for i in range(3):
queue.add(WorkItem(kind="cascade", node=f"f.n{i}", flow="f"))
claimed = queue.claim(10, 10)
assert [item.node for item in claimed] == ["f.n0", "f.n1", "f.n2"]
# Every item gets an id, which is what idempotency markers hang off.
assert all(item.entry_id for item in claimed)
def test_claiming_an_empty_queue_waits_and_gives_up():
queue = MemoryWorkQueue()
started = time.monotonic()
assert queue.claim(1, 50) == []
assert time.monotonic() - started >= 0.04
def test_a_delayed_item_stays_put_until_it_is_due():
queue = MemoryWorkQueue()
queue.add_delayed(WorkItem(kind="cascade", node="f.n", flow="f"), time.time() + 60)
assert queue.claim(1, 10) == []
assert queue.move_due(time.time()) == 0
assert queue.move_due(time.time() + 61) == 1
assert [i.node for i in queue.claim(1, 10)] == ["f.n"]
def test_parked_work_comes_back_oldest_first():
queue = MemoryWorkQueue()
for i in range(3):
queue.park("heating", WorkItem(kind="cascade", node=f"f.n{i}", flow="heating"))
assert [i.node for i in queue.unpark("heating")] == ["f.n0", "f.n1", "f.n2"]
# Unparking empties it, so a second resume does not replay the same work.
assert queue.unpark("heating") == []
def test_a_deleted_flow_leaves_nothing_parked():
queue = MemoryWorkQueue()
queue.park("gone", WorkItem(kind="cascade", node="gone.n", flow="gone"))
queue.clear_flow("gone")
assert queue.unpark("gone") == []
def test_claimed_work_counts_as_in_flight_until_it_is_acknowledged():
"""The health tile's "in flight" reads zero without this."""
queue = MemoryWorkQueue()
queue.add(WorkItem(kind="cascade", node="f.n", flow="f"))
assert queue.stats()["pending"] == 0
# The stream length was never a backlog, so the key is gone from both queues.
assert "depth" not in queue.stats()
(item,) = queue.claim(1, 10)
assert queue.stats()["pending"] == 1
queue.ack(item)
assert queue.stats()["pending"] == 0
def _pipeline_with_a_consumer() -> tuple[Pipeline, Node, MemoryState, list]:
"""A source whose message a consumer records."""
seen: list[float] = []
def consume(reading, params):
seen.append(reading)
return {"doubled": reading * 2}
source = Node(
f=lambda params: None,
provides=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
name="source",
)
consumer = Node(
f=consume,
requires=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
provides=[MessageSpec(name="doubled", port="doubled", dtype=DType.FLOAT)],
name="consumer",
)
source.assign_flow("f", "source")
consumer.assign_flow("f", "consumer")
state = MemoryState()
queue = MemoryWorkQueue()
pipeline = Pipeline(nodes=[source, consumer], state=state, work_queue=queue)
return pipeline, source, state, seen
def test_a_trigger_is_journaled_rather_than_run_on_the_spot():
pipeline, source, state, seen = _pipeline_with_a_consumer()
source.inject({"reading": 3.0})
# Nothing ran yet: the value is in the queue, not in state.
assert seen == []
assert "f.reading" not in state
service = ExecutionService(pipeline._queue)
service.bind(pipeline)
for item in pipeline._queue.claim(10, 10):
service._run_item(item)
assert seen == [3.0]
assert state["f.doubled"] == 6.0
def test_work_for_a_paused_flow_is_held_and_released_on_resume():
pipeline, source, state, seen = _pipeline_with_a_consumer()
service = ExecutionService(pipeline._queue)
service.bind(pipeline)
pipeline.pause("f")
source.inject({"reading": 1.0})
for item in pipeline._queue.claim(10, 10):
service._run_item(item)
assert seen == []
pipeline.resume("f")
for item in pipeline._queue.unpark("f"):
pipeline._queue.add(item)
for item in pipeline._queue.claim(10, 10):
service._run_item(item)
assert seen == [1.0]
def test_a_step_runs_one_held_item_and_leaves_the_flow_paused():
pipeline, source, _state, seen = _pipeline_with_a_consumer()
service = ExecutionService(pipeline._queue)
service.bind(pipeline)
pipeline.pause("f")
source.inject({"reading": 1.0})
source.inject({"reading": 2.0})
for item in pipeline._queue.claim(10, 10):
service._run_item(item)
assert seen == []
assert service.step("f") == "f.source"
assert seen == [1.0]
# Still paused, and the second value is still waiting for the next step.
assert pipeline.is_paused("f")
assert [i.outputs for i in pipeline._queue.unpark("f")] == [{"f.reading": 2.0}]
def test_stepping_a_flow_with_nothing_held_says_so_rather_than_failing():
pipeline, _source, _state, _seen = _pipeline_with_a_consumer()
service = ExecutionService(pipeline._queue)
service.bind(pipeline)
pipeline.pause("f")
assert service.step("f") is None
def test_work_for_a_stopped_flow_is_dropped():
pipeline, source, state, seen = _pipeline_with_a_consumer()
stopped = Pipeline(
nodes=pipeline.nodes,
state=pipeline.state,
work_queue=pipeline._queue,
disabled_flows={"f"},
)
service = ExecutionService(pipeline._queue)
service.bind(stopped)
# Reaching the queue at all takes a direct add: trigger drops it earlier.
stopped._queue.add(
WorkItem(kind="cascade", node="f.source", flow="f", outputs={"f.reading": 1.0})
)
for item in stopped._queue.claim(10, 10):
service._run_item(item)
assert seen == []
def test_an_item_that_keeps_coming_back_is_dead_lettered():
pipeline, _source, _state, seen = _pipeline_with_a_consumer()
service = ExecutionService(pipeline._queue)
service.bind(pipeline)
item = WorkItem(
kind="cascade",
node="f.source",
flow="f",
outputs={"f.reading": 1.0},
deliveries=4,
)
service._run_item(item)
# Given up on rather than run again, so a poison item cannot loop forever.
assert seen == []
def test_an_item_for_a_node_that_no_longer_exists_is_dropped():
pipeline, _source, _state, seen = _pipeline_with_a_consumer()
service = ExecutionService(pipeline._queue)
service.bind(pipeline)
service._run_item(WorkItem(kind="cascade", node="f.removed", flow="f"))
assert seen == []
def test_a_replayed_item_does_not_repeat_a_side_effect():
"""At-least-once delivery must not mean two of the same outgoing request."""
calls: list[float] = []
def send(reading, params):
calls.append(reading)
return None
source = Node(
f=lambda params: None,
provides=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
name="source",
)
class SendingNode(Node):
"""Stands in for the built-ins that reach outside."""
idempotent = False
sender = SendingNode(
f=send,
requires=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
name="sender",
)
source.assign_flow("f", "source")
sender.assign_flow("f", "sender")
queue = MemoryWorkQueue()
pipeline = Pipeline(nodes=[source, sender], state=MemoryState(), work_queue=queue)
service = ExecutionService(queue)
service.bind(pipeline)
source.inject({"reading": 5.0})
(item,) = queue.claim(10, 10)
service._run_item(item)
assert calls == [5.0]
# The same item again, as a reaper would hand it back after a crash.
item.deliveries = 2
service._run_item(item)
assert calls == [5.0]
# -----------------------------------------------------------------------------
# Telling the queue that a long node is working, not lost
#
# What marks an item abandoned is nobody touching it. A node with no timeout
# may run for hours, so the engine holding it says so on a timer — and an
# engine that died says nothing, which is the distinction the reaper needs.
# -----------------------------------------------------------------------------
class RecordingQueue(MemoryWorkQueue):
"""A memory queue that writes down what it was asked to hold on to."""
def __init__(self) -> None:
super().__init__()
self.touched: list[list[str]] = []
def touch(self, entry_ids: list[str]) -> None:
self.touched.append(list(entry_ids))
def test_work_in_flight_is_touched_until_it_finishes(monkeypatch):
monkeypatch.setattr(executor, "TOUCH_INTERVAL_S", 0.0)
monkeypatch.setattr(executor, "DELAYED_INTERVAL_S", 0.05)
running = threading.Event()
release = threading.Event()
def slow(reading, params):
running.set()
release.wait(5)
return {"doubled": reading * 2}
source = Node(
f=lambda params: None,
provides=[MessageSpec(name="reading", dtype=DType.FLOAT)],
name="source",
)
consumer = Node(
f=slow,
requires=[MessageSpec(name="reading", dtype=DType.FLOAT)],
provides=[MessageSpec(name="doubled", dtype=DType.FLOAT)],
name="consumer",
)
source.assign_flow("f", "source")
consumer.assign_flow("f", "consumer")
queue = RecordingQueue()
pipeline = Pipeline(
nodes=[source, consumer], state=MemoryState(), work_queue=queue
)
service = ExecutionService(queue)
service.bind(pipeline)
service.start()
try:
source.inject({"reading": 3.0})
assert running.wait(5)
# Give the timer a couple of passes while the node is still in there.
time.sleep(0.2)
held = [ids for ids in queue.touched if ids]
assert held, "a running item was never touched"
release.set()
time.sleep(0.3)
# Once it is done it is acknowledged, so there is nothing to hold.
assert queue.touched[-1] == []
finally:
release.set()
service.stop()