The queue shared the `pipeline:` prefix with flow state, so `RedisState.clear()` could DEL the queue stream and `keys()` enumerated queue entries — only callers filtering `__`-prefixed names kept it safe. It moves to `queue:` without a migration: whatever is in flight at the upgrade is dropped once, documented in DEPLOY.md rather than papered over. Alongside it: `pool_pre_ping`, so a connection idle across a Postgres restart costs a round trip instead of a failed request; the test suite pins ENVIRONMENT=local and DOMAIN=localhost itself rather than inheriting a deployment's .env; and `depth` leaves the queue stats, where it reported the capped journal length as if it were a backlog. ALERTS_FILE and PANELS_FILE now point at /data. They defaulted to a path on no volume, so alert routing and every wall-panel pairing were living in the container's writable layer and vanishing on each rebuild. Carrying the existing files across is a manual step; DEPLOY.md has it. development.md was still the upstream template — compose.override.yml, localhost.tiangolo.com, `docker compose watch` as the dev flow — and said nothing about the Playwright suite. Rewritten against what the Makefiles actually do. deployment.md was template text too, duplicating the root DEPLOY.md, and is gone. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Uq8mtNb97A7praJLyeEYgs
267 lines
7.9 KiB
Python
267 lines
7.9 KiB
Python
"""The work queue, and what the execution service does with it."""
|
|
|
|
import time
|
|
|
|
from app.flow.executor import ExecutionService
|
|
from app.flow.messages import DType, MessageSpec
|
|
from app.flow.nodes import Node
|
|
from app.flow.pipeline import Pipeline
|
|
from app.flow.queue import MemoryWorkQueue, WorkItem
|
|
from app.flow.state import MemoryState
|
|
|
|
|
|
def test_items_come_back_in_the_order_they_went_in():
|
|
queue = MemoryWorkQueue()
|
|
for i in range(3):
|
|
queue.add(WorkItem(kind="cascade", node=f"f.n{i}", flow="f"))
|
|
|
|
claimed = queue.claim(10, 10)
|
|
|
|
assert [item.node for item in claimed] == ["f.n0", "f.n1", "f.n2"]
|
|
# Every item gets an id, which is what idempotency markers hang off.
|
|
assert all(item.entry_id for item in claimed)
|
|
|
|
|
|
def test_claiming_an_empty_queue_waits_and_gives_up():
|
|
queue = MemoryWorkQueue()
|
|
started = time.monotonic()
|
|
|
|
assert queue.claim(1, 50) == []
|
|
assert time.monotonic() - started >= 0.04
|
|
|
|
|
|
def test_a_delayed_item_stays_put_until_it_is_due():
|
|
queue = MemoryWorkQueue()
|
|
queue.add_delayed(WorkItem(kind="cascade", node="f.n", flow="f"), time.time() + 60)
|
|
|
|
assert queue.claim(1, 10) == []
|
|
assert queue.move_due(time.time()) == 0
|
|
|
|
assert queue.move_due(time.time() + 61) == 1
|
|
assert [i.node for i in queue.claim(1, 10)] == ["f.n"]
|
|
|
|
|
|
def test_parked_work_comes_back_oldest_first():
|
|
queue = MemoryWorkQueue()
|
|
for i in range(3):
|
|
queue.park("heating", WorkItem(kind="cascade", node=f"f.n{i}", flow="heating"))
|
|
|
|
assert [i.node for i in queue.unpark("heating")] == ["f.n0", "f.n1", "f.n2"]
|
|
# Unparking empties it, so a second resume does not replay the same work.
|
|
assert queue.unpark("heating") == []
|
|
|
|
|
|
def test_a_deleted_flow_leaves_nothing_parked():
|
|
queue = MemoryWorkQueue()
|
|
queue.park("gone", WorkItem(kind="cascade", node="gone.n", flow="gone"))
|
|
|
|
queue.clear_flow("gone")
|
|
|
|
assert queue.unpark("gone") == []
|
|
|
|
|
|
def test_claimed_work_counts_as_in_flight_until_it_is_acknowledged():
|
|
"""The health tile's "in flight" reads zero without this."""
|
|
queue = MemoryWorkQueue()
|
|
queue.add(WorkItem(kind="cascade", node="f.n", flow="f"))
|
|
|
|
assert queue.stats()["pending"] == 0
|
|
# The stream length was never a backlog, so the key is gone from both queues.
|
|
assert "depth" not in queue.stats()
|
|
|
|
(item,) = queue.claim(1, 10)
|
|
assert queue.stats()["pending"] == 1
|
|
|
|
queue.ack(item)
|
|
assert queue.stats()["pending"] == 0
|
|
|
|
|
|
def _pipeline_with_a_consumer() -> tuple[Pipeline, Node, MemoryState, list]:
|
|
"""A source whose message a consumer records."""
|
|
seen: list[float] = []
|
|
|
|
def consume(reading, params):
|
|
seen.append(reading)
|
|
return {"doubled": reading * 2}
|
|
|
|
source = Node(
|
|
f=lambda params: None,
|
|
provides=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
name="source",
|
|
)
|
|
consumer = Node(
|
|
f=consume,
|
|
requires=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
provides=[MessageSpec(name="doubled", port="doubled", dtype=DType.FLOAT)],
|
|
name="consumer",
|
|
)
|
|
source.assign_flow("f", "source")
|
|
consumer.assign_flow("f", "consumer")
|
|
|
|
state = MemoryState()
|
|
queue = MemoryWorkQueue()
|
|
pipeline = Pipeline(nodes=[source, consumer], state=state, work_queue=queue)
|
|
return pipeline, source, state, seen
|
|
|
|
|
|
def test_a_trigger_is_journaled_rather_than_run_on_the_spot():
|
|
pipeline, source, state, seen = _pipeline_with_a_consumer()
|
|
|
|
source.inject({"reading": 3.0})
|
|
|
|
# Nothing ran yet: the value is in the queue, not in state.
|
|
assert seen == []
|
|
assert "f.reading" not in state
|
|
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == [3.0]
|
|
assert state["f.doubled"] == 6.0
|
|
|
|
|
|
def test_work_for_a_paused_flow_is_held_and_released_on_resume():
|
|
pipeline, source, state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
pipeline.pause("f")
|
|
source.inject({"reading": 1.0})
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == []
|
|
|
|
pipeline.resume("f")
|
|
for item in pipeline._queue.unpark("f"):
|
|
pipeline._queue.add(item)
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == [1.0]
|
|
|
|
|
|
def test_a_step_runs_one_held_item_and_leaves_the_flow_paused():
|
|
pipeline, source, _state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
pipeline.pause("f")
|
|
source.inject({"reading": 1.0})
|
|
source.inject({"reading": 2.0})
|
|
for item in pipeline._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
assert seen == []
|
|
|
|
assert service.step("f") == "f.source"
|
|
|
|
assert seen == [1.0]
|
|
# Still paused, and the second value is still waiting for the next step.
|
|
assert pipeline.is_paused("f")
|
|
assert [i.outputs for i in pipeline._queue.unpark("f")] == [{"f.reading": 2.0}]
|
|
|
|
|
|
def test_stepping_a_flow_with_nothing_held_says_so_rather_than_failing():
|
|
pipeline, _source, _state, _seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
pipeline.pause("f")
|
|
|
|
assert service.step("f") is None
|
|
|
|
|
|
def test_work_for_a_stopped_flow_is_dropped():
|
|
pipeline, source, state, seen = _pipeline_with_a_consumer()
|
|
stopped = Pipeline(
|
|
nodes=pipeline.nodes,
|
|
state=pipeline.state,
|
|
work_queue=pipeline._queue,
|
|
disabled_flows={"f"},
|
|
)
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(stopped)
|
|
|
|
# Reaching the queue at all takes a direct add: trigger drops it earlier.
|
|
stopped._queue.add(
|
|
WorkItem(kind="cascade", node="f.source", flow="f", outputs={"f.reading": 1.0})
|
|
)
|
|
for item in stopped._queue.claim(10, 10):
|
|
service._run_item(item)
|
|
|
|
assert seen == []
|
|
|
|
|
|
def test_an_item_that_keeps_coming_back_is_dead_lettered():
|
|
pipeline, _source, _state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
item = WorkItem(
|
|
kind="cascade",
|
|
node="f.source",
|
|
flow="f",
|
|
outputs={"f.reading": 1.0},
|
|
deliveries=4,
|
|
)
|
|
service._run_item(item)
|
|
|
|
# Given up on rather than run again, so a poison item cannot loop forever.
|
|
assert seen == []
|
|
|
|
|
|
def test_an_item_for_a_node_that_no_longer_exists_is_dropped():
|
|
pipeline, _source, _state, seen = _pipeline_with_a_consumer()
|
|
service = ExecutionService(pipeline._queue)
|
|
service.bind(pipeline)
|
|
|
|
service._run_item(WorkItem(kind="cascade", node="f.removed", flow="f"))
|
|
|
|
assert seen == []
|
|
|
|
|
|
def test_a_replayed_item_does_not_repeat_a_side_effect():
|
|
"""At-least-once delivery must not mean two of the same outgoing request."""
|
|
calls: list[float] = []
|
|
|
|
def send(reading, params):
|
|
calls.append(reading)
|
|
return None
|
|
|
|
source = Node(
|
|
f=lambda params: None,
|
|
provides=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
name="source",
|
|
)
|
|
|
|
class SendingNode(Node):
|
|
"""Stands in for the built-ins that reach outside."""
|
|
|
|
idempotent = False
|
|
|
|
sender = SendingNode(
|
|
f=send,
|
|
requires=[MessageSpec(name="reading", port="reading", dtype=DType.FLOAT)],
|
|
name="sender",
|
|
)
|
|
source.assign_flow("f", "source")
|
|
sender.assign_flow("f", "sender")
|
|
|
|
queue = MemoryWorkQueue()
|
|
pipeline = Pipeline(nodes=[source, sender], state=MemoryState(), work_queue=queue)
|
|
service = ExecutionService(queue)
|
|
service.bind(pipeline)
|
|
|
|
source.inject({"reading": 5.0})
|
|
(item,) = queue.claim(10, 10)
|
|
service._run_item(item)
|
|
assert calls == [5.0]
|
|
|
|
# The same item again, as a reaper would hand it back after a crash.
|
|
item.deliveries = 2
|
|
service._run_item(item)
|
|
|
|
assert calls == [5.0]
|