Refuse what a node cannot publish, and stop timing out work that is fine
Four things the python SDK turned up, each fixed where every client sees it. A key no port declares is now an error rather than a silent drop, on the return, the yield and the emit alike — the contract the docs already stated. The SDK reads literal yields at sync time, so a typo fails before anything runs, and an emission of one fails the call rather than being logged where nobody looks. NaN and infinity are refused at the port. JSON cannot spell either, so one that travelled came back as a 500, a socket frame that stopped the canvas, or a metric batch the database dropped whole. An artifact input takes `@run:<id>.<output>` or a bare digest, resolved on the engine — so the CLI, the run dialog and a python caller mean the same thing, and a sweep can pass one at all. Node timeouts are off by default. The clock measured silence, which a training node is full of, and remote workers had already stopped enforcing it — their heartbeat reset it. Now a heartbeat proves the agent rather than the node, ninety seconds of nothing fails the call either way, and the engine touches work it is still running so a long node is not redelivered at sixty seconds. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_019V5bsYGNxcgPs4xXmTPx69
This commit is contained in:
@@ -93,10 +93,11 @@ class Settings(BaseSettings):
|
||||
MCP_TOKEN_EXPIRE_MINUTES: int = 60
|
||||
MCP_REFRESH_EXPIRE_DAYS: int = 30
|
||||
FLOW_MAX_WORKERS: int = 4
|
||||
# How long a python node may run before its worker is killed, unless the
|
||||
# node sets its own. Long enough for a slow HTTP call, short enough that a
|
||||
# runaway loop is not a wedged flow.
|
||||
FLOW_NODE_TIMEOUT: float = 30.0
|
||||
# How long a python node may be silent before its worker is killed, unless
|
||||
# the node sets its own. 0, the default, disables it: a dead worker still
|
||||
# fails fast, and a slow one is left to finish. Set it where silence means
|
||||
# stuck rather than working.
|
||||
FLOW_NODE_TIMEOUT: float = 0.0
|
||||
# How long the engine's own metrics, events and run records are kept.
|
||||
OBS_RETENTION_DAYS: int = 30
|
||||
# Without a Redis host the engine keeps its state in memory.
|
||||
|
||||
@@ -877,7 +877,13 @@ class FlowController:
|
||||
entry.status = NodeStatus.ERROR
|
||||
entry.error = problem
|
||||
return entry
|
||||
timeout = node_def.timeout or settings.FLOW_NODE_TIMEOUT
|
||||
# Not ``or``: 0 is a node saying it has no limit, which is
|
||||
# exactly the value that would fall through to the default.
|
||||
timeout = (
|
||||
node_def.timeout
|
||||
if node_def.timeout is not None
|
||||
else settings.FLOW_NODE_TIMEOUT
|
||||
)
|
||||
function = self.workers.proxy(
|
||||
owner,
|
||||
local,
|
||||
|
||||
@@ -31,6 +31,10 @@ CLAIM_BLOCK_MS = 1000
|
||||
# Long enough that a busy cascade is not mistaken for a dead one.
|
||||
RECLAIM_IDLE_MS = 60_000
|
||||
RECLAIM_INTERVAL_S = 30.0
|
||||
# How often to tell the queue that what we hold is still being worked on. A
|
||||
# node may run for as long as it likes, so what marks an item abandoned is this
|
||||
# stopping — which is what an engine that died does.
|
||||
TOUCH_INTERVAL_S = 20.0
|
||||
DELAYED_INTERVAL_S = 1.0
|
||||
MAX_CASCADES = 4
|
||||
# How long a reload waits for claimed work to finish before rebuilding anyway.
|
||||
@@ -54,6 +58,8 @@ class ExecutionService:
|
||||
self._intake.set()
|
||||
self._inflight = 0
|
||||
self._inflight_lock = threading.Condition()
|
||||
# Entry ids claimed and still running, under _inflight_lock.
|
||||
self._active: set[str] = set()
|
||||
self.node_pool = ThreadPoolExecutor(
|
||||
max_workers=max_workers or 4, thread_name_prefix="node"
|
||||
)
|
||||
@@ -145,6 +151,7 @@ class ExecutionService:
|
||||
def _tick(self) -> None:
|
||||
"""Promote delayed items, and take back what a dead engine dropped."""
|
||||
last_reclaim = 0.0
|
||||
last_touch = 0.0
|
||||
while not self._stop.is_set():
|
||||
self._stop.wait(DELAYED_INTERVAL_S)
|
||||
if self._stop.is_set():
|
||||
@@ -155,6 +162,16 @@ class ExecutionService:
|
||||
logger.error("Could not promote delayed work: %s", exc)
|
||||
|
||||
now = time.monotonic()
|
||||
if now - last_touch >= TOUCH_INTERVAL_S:
|
||||
last_touch = now
|
||||
with self._inflight_lock:
|
||||
running = list(self._active)
|
||||
try:
|
||||
self.queue.touch(running)
|
||||
except Exception as exc:
|
||||
logger.error("Could not touch claimed work: %s", exc)
|
||||
|
||||
|
||||
if now - last_reclaim < RECLAIM_INTERVAL_S:
|
||||
continue
|
||||
last_reclaim = now
|
||||
@@ -172,15 +189,18 @@ class ExecutionService:
|
||||
def _dispatch(self, item: WorkItem) -> None:
|
||||
with self._inflight_lock:
|
||||
self._inflight += 1
|
||||
if item.entry_id:
|
||||
self._active.add(item.entry_id)
|
||||
try:
|
||||
self._cascade_pool.submit(self._handle, item)
|
||||
except RuntimeError:
|
||||
# Pool already shutting down.
|
||||
self._done()
|
||||
self._done(item)
|
||||
|
||||
def _done(self) -> None:
|
||||
def _done(self, item: WorkItem) -> None:
|
||||
with self._inflight_lock:
|
||||
self._inflight -= 1
|
||||
self._active.discard(item.entry_id)
|
||||
self._inflight_lock.notify_all()
|
||||
|
||||
# -------------------------------------------------------------------------
|
||||
@@ -222,7 +242,7 @@ class ExecutionService:
|
||||
logger.error(
|
||||
"Could not acknowledge work for '%s': %s", item.node, exc
|
||||
)
|
||||
self._done()
|
||||
self._done(item)
|
||||
|
||||
def _run_item(self, item: WorkItem) -> bool:
|
||||
"""Run one item. False means it was not handled and must come back."""
|
||||
|
||||
@@ -11,6 +11,7 @@ consume each other's messages.
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
from enum import Enum
|
||||
from typing import Any
|
||||
|
||||
@@ -65,11 +66,42 @@ _ITEM_TYPES = frozenset(
|
||||
)
|
||||
|
||||
|
||||
#: How deep to look for a non-finite number. Deeper than any payload that
|
||||
#: reads well on a canvas, and a bound on a value that refers to itself.
|
||||
_WALK_DEPTH = 8
|
||||
|
||||
|
||||
def _is_number(value: Any) -> bool:
|
||||
"""A measurement. bool is an int subclass; a flag is not a number here."""
|
||||
return isinstance(value, (int, float)) and not isinstance(value, bool)
|
||||
|
||||
|
||||
def _nonfinite(value: Any, depth: int = 0) -> float | None:
|
||||
"""The first NaN or infinity in a value, however deeply it sits.
|
||||
|
||||
JSON cannot spell either: ``json.dumps`` writes a bare ``NaN``, which a
|
||||
strict parser refuses. So a metric that goes non-finite leaves the engine
|
||||
as a response nobody can read, a socket frame that stops a canvas, or a row
|
||||
the database rejects — all of them a long way from the node that produced
|
||||
it. Naming it here costs one walk of a value already about to be encoded.
|
||||
"""
|
||||
if isinstance(value, float) and not math.isfinite(value):
|
||||
return value
|
||||
if depth >= _WALK_DEPTH:
|
||||
return None
|
||||
if isinstance(value, dict):
|
||||
items: Any = value.values()
|
||||
elif isinstance(value, (list, tuple)):
|
||||
items = value
|
||||
else:
|
||||
return None
|
||||
for item in items:
|
||||
found = _nonfinite(item, depth + 1)
|
||||
if found is not None:
|
||||
return found
|
||||
return None
|
||||
|
||||
|
||||
def _is_record(value: Any) -> bool:
|
||||
return isinstance(value, dict) and all(
|
||||
isinstance(key, str) and (item is None or isinstance(item, _SCALARS))
|
||||
@@ -188,6 +220,13 @@ class MessageSpec(BaseModel):
|
||||
def check(self, value: Any) -> None:
|
||||
"""Raise if ``value`` does not match this port's declared type."""
|
||||
where = self.name or self.port
|
||||
stray = _nonfinite(value)
|
||||
if stray is not None:
|
||||
raise TypeError(
|
||||
f"{where}: {stray} cannot travel as JSON. A subset with nothing "
|
||||
"in it, or a division that had no denominator, is what usually "
|
||||
"produces one — publish None, or a number that says so."
|
||||
)
|
||||
if self.dtype is DType.SERIES:
|
||||
ok = _is_series(value)
|
||||
elif self.dtype is DType.LIST:
|
||||
|
||||
@@ -287,7 +287,9 @@ class Node:
|
||||
|
||||
Only ``None`` means "nothing to publish". A falsy value of the wrong
|
||||
shape — ``0``, ``""``, an empty list — is a mistake worth naming rather
|
||||
than silence someone has to debug from an empty canvas.
|
||||
than silence someone has to debug from an empty canvas. So is a key no
|
||||
port declares: a mistyped metric name is how a training curve goes
|
||||
missing, and it costs nothing to say so at the first yield.
|
||||
"""
|
||||
if retval is None:
|
||||
return None
|
||||
@@ -302,8 +304,18 @@ class Node:
|
||||
for key, value in retval.items():
|
||||
spec = by_port.get(key) or self.provides.get(key)
|
||||
if spec is None:
|
||||
continue
|
||||
spec.check(value)
|
||||
declared = sorted(by_port) or ["none"]
|
||||
raise NodeOutputError(
|
||||
f"'{self.local_id}' produced '{key}', which no port "
|
||||
f"declares. Its ports are: {', '.join(declared)}."
|
||||
)
|
||||
try:
|
||||
spec.check(value)
|
||||
except TypeError as exc:
|
||||
# Same category as an undeclared key: what the node produced is
|
||||
# wrong. Naming it as one is what lets an emission fail the
|
||||
# call rather than being logged where nobody reads it.
|
||||
raise NodeOutputError(f"'{self.local_id}' {exc}") from exc
|
||||
outputs[spec.name] = value
|
||||
return outputs or None
|
||||
|
||||
|
||||
@@ -126,6 +126,15 @@ class WorkQueue(ABC):
|
||||
def ack(self, item: WorkItem) -> None:
|
||||
"""Mark an item done, so it is never redelivered."""
|
||||
|
||||
def touch(self, entry_ids: list[str]) -> None:
|
||||
"""Say these items are still being worked on, not abandoned.
|
||||
|
||||
A node with no timeout may run far longer than the reclaim window, and
|
||||
nothing else distinguishes that from an engine that died holding the
|
||||
item. Concrete rather than abstract: a queue with no redelivery has
|
||||
nothing to answer here.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def reclaim_stale(self, min_idle_ms: int) -> list[WorkItem]:
|
||||
"""Take back items claimed by a consumer that never acknowledged them."""
|
||||
@@ -361,6 +370,23 @@ class RedisWorkQueue(WorkQueue):
|
||||
if item.entry_id:
|
||||
self._redis.xack(self._stream, GROUP, item.entry_id)
|
||||
|
||||
def touch(self, entry_ids: list[str]) -> None:
|
||||
if not entry_ids:
|
||||
return
|
||||
# Claiming an entry we already hold resets how long it has been idle,
|
||||
# which is the only thing `reclaim_stale` reads. `justid` keeps the
|
||||
# delivery count where it is, so a long node neither redelivers nor
|
||||
# spends its way towards the dead-letter cap. An entry already
|
||||
# acknowledged is simply not there, and this is a no-op for it.
|
||||
self._redis.xclaim(
|
||||
self._stream,
|
||||
GROUP,
|
||||
self._consumer,
|
||||
min_idle_time=0,
|
||||
message_ids=entry_ids,
|
||||
justid=True,
|
||||
)
|
||||
|
||||
def reclaim_stale(self, min_idle_ms: int) -> list[WorkItem]:
|
||||
"""Take over entries a dead consumer never acknowledged."""
|
||||
_cursor, entries, _deleted = cast(
|
||||
|
||||
@@ -30,14 +30,17 @@ import time
|
||||
from collections.abc import Callable
|
||||
from typing import Any
|
||||
|
||||
from fluksio.flow.nodes.base import NodeOutputError
|
||||
from fluksio.flow.workers import NodeTimeout, RemoteError, _remote_class
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
#: How long the loop is given to accept a frame we are handing it.
|
||||
SEND_TIMEOUT_S = 30.0
|
||||
#: A worker that has said nothing for this long is treated as gone. It sends a
|
||||
#: heartbeat while it is executing, so this only ever catches a dead socket.
|
||||
#: A worker that has said nothing at all for this long — not even a heartbeat —
|
||||
#: is treated as gone. It beats every ten seconds while it is executing, so this
|
||||
#: catches a dead socket rather than a slow node, and it is what bounds a call
|
||||
#: whose node has no timeout of its own.
|
||||
SILENCE_S = 90.0
|
||||
#: Protocol version this engine speaks. A worker announcing anything else is
|
||||
#: refused rather than half-understood.
|
||||
@@ -125,15 +128,27 @@ class RemoteWorker:
|
||||
future = asyncio.run_coroutine_threadsafe(self._send(payload), self._loop)
|
||||
future.result(timeout=SEND_TIMEOUT_S)
|
||||
|
||||
# The node's own deadline measures silence, so a node reporting its
|
||||
# progress is never mistaken for a hung one. A heartbeat is not
|
||||
# progress: it says the agent is alive, which is what SILENCE_S
|
||||
# asks, and says nothing about the node — so it feeds the liveness
|
||||
# bound below and never the node's own.
|
||||
deadline = time.monotonic() + timeout if timeout > 0 else None
|
||||
while True:
|
||||
wait = SILENCE_S
|
||||
if deadline is not None:
|
||||
wait = min(SILENCE_S, deadline - time.monotonic())
|
||||
try:
|
||||
# Reset per frame: the deadline measures silence, so a node
|
||||
# reporting its progress is never mistaken for a hung one.
|
||||
message = inbox.get(timeout=timeout)
|
||||
message = inbox.get(timeout=max(wait, 0.0))
|
||||
except queue.Empty:
|
||||
self.cancel(call_id)
|
||||
raise NodeTimeout(
|
||||
f"'{self.name}' was silent for {timeout}s"
|
||||
if deadline is not None and time.monotonic() >= deadline:
|
||||
raise NodeTimeout(
|
||||
f"'{self.name}' was silent for {timeout}s"
|
||||
) from None
|
||||
raise RemoteError(
|
||||
f"worker '{self.name}' sent nothing for "
|
||||
f"{SILENCE_S:.0f}s and is presumed gone"
|
||||
) from None
|
||||
if message is None:
|
||||
raise RemoteError(f"worker '{self.name}' went away mid-call")
|
||||
@@ -144,12 +159,19 @@ class RemoteWorker:
|
||||
if on_event is not None:
|
||||
try:
|
||||
on_event(message)
|
||||
except NodeOutputError:
|
||||
# A port the node never declared. Stop the call
|
||||
# rather than let the rest of its emissions arrive.
|
||||
self.cancel(call_id)
|
||||
raise
|
||||
except Exception:
|
||||
logger.exception("Could not record a worker event")
|
||||
if deadline is not None:
|
||||
deadline = time.monotonic() + timeout
|
||||
continue
|
||||
return message
|
||||
except Exception as exc:
|
||||
if isinstance(exc, (NodeTimeout, RemoteError)):
|
||||
if isinstance(exc, (NodeTimeout, RemoteError, NodeOutputError)):
|
||||
raise
|
||||
raise RemoteError(f"worker '{self.name}': {exc}") from exc
|
||||
finally:
|
||||
|
||||
@@ -28,6 +28,7 @@ from __future__ import annotations
|
||||
import hashlib
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import os
|
||||
import socket
|
||||
import threading
|
||||
@@ -43,9 +44,9 @@ from sqlalchemy.dialects.sqlite import insert as upsert
|
||||
from sqlmodel import Session, col, select
|
||||
|
||||
from fluksio.core.db import engine as db_engine
|
||||
from fluksio.flow.artifacts import ArtifactStore, is_reference
|
||||
from fluksio.flow.artifacts import ArtifactStore, is_reference, valid_digest
|
||||
from fluksio.flow.controller import FlowController, RunContext
|
||||
from fluksio.flow.messages import qualify
|
||||
from fluksio.flow.messages import DType, qualify
|
||||
from fluksio.flow.pipeline import NodeOutcome, Pipeline
|
||||
from fluksio.flow.queue import WorkItem, WorkQueue
|
||||
from fluksio.flow.schemas import FlowDef
|
||||
@@ -144,6 +145,112 @@ def required_labels(flow: FlowDef) -> list[str]:
|
||||
)
|
||||
|
||||
|
||||
#: What a run's output is called from outside it: ``@run:<id>.<output>``.
|
||||
RUN_REF_PREFIX = "@run:"
|
||||
|
||||
|
||||
def resolve_references(
|
||||
flow: FlowDef, params: dict[str, Any], artifacts: ArtifactStore | None = None
|
||||
) -> dict[str, Any]:
|
||||
"""Turn the text spellings of an artifact input into the reference itself.
|
||||
|
||||
A python caller hands one run's output straight to the next, because it has
|
||||
the reference in its hand. A shell does not, and pasting the whole object
|
||||
is not a command anyone wants to type — so an artifact input also takes
|
||||
``@run:<id>.<output>``, naming what a run produced, or a bare
|
||||
``sha256:...`` digest naming the bytes. Resolved here rather than in each
|
||||
client, so the CLI, the browser and a python caller all mean the same
|
||||
thing by the same string.
|
||||
"""
|
||||
wanted = {
|
||||
declared.spec.name
|
||||
for declared in flow.inputs
|
||||
if declared.spec.dtype is DType.ARTIFACT
|
||||
}
|
||||
pending = {
|
||||
key: value
|
||||
for key, value in params.items()
|
||||
if key in wanted and isinstance(value, str)
|
||||
}
|
||||
if not pending:
|
||||
return params
|
||||
|
||||
resolved = dict(params)
|
||||
with Session(db_engine) as session:
|
||||
for key, text in pending.items():
|
||||
if text.startswith(RUN_REF_PREFIX):
|
||||
resolved[key] = _from_run(session, key, text[len(RUN_REF_PREFIX) :])
|
||||
elif valid_digest(text):
|
||||
resolved[key] = _from_digest(session, key, text, artifacts)
|
||||
# Anything else is left alone: the type check names it better than
|
||||
# a guess about what was meant would.
|
||||
return resolved
|
||||
|
||||
|
||||
def _from_run(session: Session, key: str, spelling: str) -> dict[str, Any]:
|
||||
"""``<run id>.<output>`` as the reference that run produced."""
|
||||
run_id, _, output = spelling.partition(".")
|
||||
if not run_id or not output:
|
||||
raise RunRejected(
|
||||
f"Parameter '{key}': '{RUN_REF_PREFIX}{spelling}' names no output — "
|
||||
f"write {RUN_REF_PREFIX}<run id>.<output>"
|
||||
)
|
||||
run = session.get(Run, run_id)
|
||||
if run is None:
|
||||
raise RunRejected(f"Parameter '{key}': there is no run '{run_id}'")
|
||||
|
||||
# The run's own result first: that is the reference as its producer made
|
||||
# it, file name and all. The rows are the fallback, and they carry the
|
||||
# message name instead — which loads the same bytes either way.
|
||||
candidate = (run.result or {}).get(output)
|
||||
if is_reference(candidate):
|
||||
return dict(candidate)
|
||||
|
||||
rows = session.exec(
|
||||
select(RunArtifact).where(col(RunArtifact.run_id) == run_id)
|
||||
).all()
|
||||
for row in rows:
|
||||
if output in (row.name, row.name.rsplit(".", 1)[-1]):
|
||||
return {
|
||||
"digest": row.digest,
|
||||
"size": row.size,
|
||||
"media_type": row.media_type or "application/octet-stream",
|
||||
"name": row.name,
|
||||
}
|
||||
known = ", ".join(sorted(row.name for row in rows)) or "none"
|
||||
raise RunRejected(
|
||||
f"Parameter '{key}': run '{run_id}' has no artifact '{output}' "
|
||||
f"(it made: {known})"
|
||||
)
|
||||
|
||||
|
||||
def _from_digest(
|
||||
session: Session, key: str, digest: str, artifacts: ArtifactStore | None
|
||||
) -> dict[str, Any]:
|
||||
"""A bare digest as a reference, with the size the store needs."""
|
||||
row = session.exec(
|
||||
select(RunArtifact)
|
||||
.where(col(RunArtifact.digest) == digest)
|
||||
.order_by(col(RunArtifact.run_id).desc())
|
||||
).first()
|
||||
if row is None:
|
||||
raise RunRejected(
|
||||
f"Parameter '{key}': no run has produced '{digest}', so there is "
|
||||
"nothing here under that digest"
|
||||
)
|
||||
if artifacts is not None and artifacts.path(digest) is None:
|
||||
raise RunRejected(
|
||||
f"Parameter '{key}': '{digest}' is known but its bytes are gone "
|
||||
"from this installation's store"
|
||||
)
|
||||
return {
|
||||
"digest": row.digest,
|
||||
"size": row.size,
|
||||
"media_type": row.media_type or "application/octet-stream",
|
||||
"name": row.name,
|
||||
}
|
||||
|
||||
|
||||
def seed_values(
|
||||
flow: FlowDef, params: dict[str, Any], seed: int | None = None
|
||||
) -> dict[str, Any]:
|
||||
@@ -230,6 +337,11 @@ class MetricSink:
|
||||
# A checkpoint or a record is on the run some other way —
|
||||
# as an artifact, or as its result. Only numbers are series.
|
||||
continue
|
||||
if not math.isfinite(value):
|
||||
# The ports refuse these, so one here came from somewhere
|
||||
# that does not go through them. A column that cannot hold
|
||||
# it would take the whole batch down with it.
|
||||
continue
|
||||
step = self._steps.get(name, -1) + 1
|
||||
self._steps[name] = step
|
||||
row = RunMetric(
|
||||
@@ -368,6 +480,7 @@ class RunService:
|
||||
) -> None:
|
||||
self.controller = controller
|
||||
self.queue = queue
|
||||
self._artifacts = artifacts
|
||||
self._cache = RunCache(artifacts)
|
||||
# Without one, a run gets a private in-memory state — which is exactly
|
||||
# the isolation it wants, minus surviving the process.
|
||||
@@ -433,6 +546,10 @@ class RunService:
|
||||
if issues:
|
||||
raise RunRejected(" ".join(issues))
|
||||
params = params or {}
|
||||
# Resolved before it is stored, so what the run records is the same
|
||||
# reference a python caller would have passed and every later reader —
|
||||
# the digest, the cache, the run detail — sees one spelling.
|
||||
params = resolve_references(flow, params, self._artifacts)
|
||||
# Checked here rather than in the driver: a caller who mistyped a
|
||||
# parameter should be told now, not by a run that fails in a minute.
|
||||
seed_values(flow, params, seed)
|
||||
|
||||
@@ -46,13 +46,13 @@ class NodeDef(BaseModel):
|
||||
source_ref: str | None = None
|
||||
timeout: float | None = Field(
|
||||
default=None,
|
||||
gt=0,
|
||||
ge=0,
|
||||
description=(
|
||||
"Seconds this node's code may run before it is stopped. This "
|
||||
"covers the first call's imports, which can be much slower than "
|
||||
"the body. Above 60 the engine may deliver its work again while "
|
||||
"it is still running — in a batch run, which never redelivers, "
|
||||
"it is an idle timeout instead: silence this long is a kill."
|
||||
"Seconds this node's code may be silent before it is stopped. A "
|
||||
"yield or an emit resets the clock, and the first call's imports "
|
||||
"are not charged to it. 0 disables the limit: the node runs until "
|
||||
"it finishes, and only a dead worker fails the call. Empty "
|
||||
"inherits the engine default."
|
||||
),
|
||||
)
|
||||
device: str | None = Field(
|
||||
|
||||
@@ -31,6 +31,7 @@ from typing import Any
|
||||
from fluksio_worker import worker_main as _worker_main
|
||||
|
||||
from fluksio.flow.events import EventBus
|
||||
from fluksio.flow.nodes.base import NodeOutputError
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -42,6 +43,12 @@ WORKER_MAIN = Path(_worker_main.__file__)
|
||||
#: something a person is watching a spinner for.
|
||||
COMPILE_TIMEOUT = 60.0
|
||||
|
||||
#: How often to wake up while waiting on a node with no timeout. Nothing is
|
||||
#: checked on a schedule — a worker that dies closes its pipe and wakes the
|
||||
#: read immediately — so this is only here to notice a pipe held open by
|
||||
#: something that outlived the worker it belonged to.
|
||||
POLL_S = 30.0
|
||||
|
||||
#: Environment the worker is not given. ``SECRET_KEY`` decrypts every stored
|
||||
#: secret, not only the ones bound to the node asking.
|
||||
ENV_DENY_PREFIXES = ("DATABASE_URL", "FIRST_SUPERUSER", "SENTRY_DSN")
|
||||
@@ -402,7 +409,9 @@ class PythonWorkerPool:
|
||||
# been silent rather than how long it has been working. A node
|
||||
# that reports nothing is still held to it, which is what keeps
|
||||
# the deadline meaningful for the ones that never report.
|
||||
line = worker.read_line(time.monotonic() + timeout)
|
||||
line = worker.read_line(
|
||||
time.monotonic() + (timeout if timeout > 0 else POLL_S)
|
||||
)
|
||||
if line:
|
||||
try:
|
||||
message = dict(json.loads(line))
|
||||
@@ -418,6 +427,14 @@ class PythonWorkerPool:
|
||||
if on_event is not None:
|
||||
try:
|
||||
on_event(message)
|
||||
except NodeOutputError:
|
||||
# An emission on a port the node never declared.
|
||||
# The call is already wrong, and its generator has
|
||||
# more frames coming down this pipe — so retire the
|
||||
# worker and let the author see what they emitted.
|
||||
worker.cancelled = True
|
||||
worker.kill()
|
||||
raise
|
||||
except Exception:
|
||||
logger.exception("Could not record a worker event")
|
||||
continue
|
||||
@@ -435,8 +452,18 @@ class PythonWorkerPool:
|
||||
if worker.cancelled:
|
||||
raise NodeCancelled("cancelled while it was running")
|
||||
if line is None:
|
||||
worker.kill()
|
||||
raise NodeTimeout(f"was silent for {timeout}s and was killed")
|
||||
if timeout > 0:
|
||||
worker.kill()
|
||||
raise NodeTimeout(f"was silent for {timeout}s and was killed")
|
||||
# No limit, so silence is the node working. What ends the call
|
||||
# is the worker dying, and that arrives as the pipe closing
|
||||
# rather than as a deadline — unless something else inherited
|
||||
# the write end and is holding it open, which is what this
|
||||
# asks about.
|
||||
if worker.alive():
|
||||
continue
|
||||
worker.cancelled = True
|
||||
raise RemoteError("worker died")
|
||||
worker.cancelled = True
|
||||
raise RemoteError("worker died")
|
||||
|
||||
|
||||
@@ -13,9 +13,11 @@ have to resolve inside a worker, where ``fluksio`` is the reporter module.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import inspect
|
||||
import json
|
||||
import re
|
||||
import textwrap
|
||||
import warnings
|
||||
from collections.abc import Callable, Sequence
|
||||
from pathlib import Path
|
||||
@@ -265,9 +267,15 @@ def node(
|
||||
A batch run skips this node when an earlier one already ran the same source
|
||||
with the same settings and the same input values, and restores what it
|
||||
returned. ``cache=False`` says not to: the answer can change on its own.
|
||||
|
||||
``timeout`` is seconds of silence — a yield or an emit resets it — after
|
||||
which the node is stopped. Left out it inherits the engine's default, which
|
||||
is no limit; set one where silence means stuck rather than working.
|
||||
"""
|
||||
if device_policy not in ("require", "prefer"):
|
||||
raise SyncError("device_policy is 'require' or 'prefer'")
|
||||
if timeout is not None and timeout < 0:
|
||||
raise SyncError(f"node timeout must be 0 or more (0 disables it), got {timeout}")
|
||||
|
||||
def decorate(fn: F) -> F:
|
||||
required, _ = _ports(requires)
|
||||
@@ -366,6 +374,68 @@ def _check_signature(spec: NodeSpec) -> None:
|
||||
f"add Port('{name}', ...) to requires=, or give it a default"
|
||||
)
|
||||
|
||||
_check_emissions(spec)
|
||||
|
||||
|
||||
def _check_emissions(spec: NodeSpec) -> None:
|
||||
"""Whether what the function yields is what it says it provides.
|
||||
|
||||
The engine refuses an undeclared key at the first yield, which is right but
|
||||
late: a sweep can be an hour in. Only literal keys are read here — a yield
|
||||
of a name the code computes is left to the engine — so this catches the
|
||||
mistake it is for, which is a typo.
|
||||
"""
|
||||
provided = {port.port for port in spec.provides}
|
||||
try:
|
||||
tree = ast.parse(textwrap.dedent(inspect.getsource(spec.fn)))
|
||||
except (OSError, TypeError, SyntaxError):
|
||||
# No source to read: a function built at runtime, or one from a
|
||||
# console. Not a reason to refuse it.
|
||||
return
|
||||
|
||||
for key in _emitted_keys(tree):
|
||||
if key not in provided:
|
||||
declared = ", ".join(sorted(provided)) or "none"
|
||||
raise SyncError(
|
||||
f"node '{spec.id}': {spec.fn.__name__}() emits '{key}', which no "
|
||||
f"port declares. Its ports are: {declared}."
|
||||
)
|
||||
|
||||
|
||||
def _emitted_keys(tree: ast.AST) -> list[str]:
|
||||
"""Literal port names a function yields or emits, its own only.
|
||||
|
||||
A nested function is somebody else's ports — a helper generator yielding
|
||||
pairs is not this node emitting them — so the walk stops at one.
|
||||
"""
|
||||
keys: list[str] = []
|
||||
# Each branch carries whether it is already inside the node's own function,
|
||||
# so the first def is the node and any def below it is a helper.
|
||||
stack: list[tuple[ast.AST, bool]] = [(tree, False)]
|
||||
while stack:
|
||||
node, inside = stack.pop()
|
||||
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.Lambda)):
|
||||
if inside:
|
||||
continue
|
||||
inside = True
|
||||
if isinstance(node, ast.Yield) and isinstance(node.value, ast.Dict):
|
||||
keys += [
|
||||
key.value
|
||||
for key in node.value.keys
|
||||
if isinstance(key, ast.Constant) and isinstance(key.value, str)
|
||||
]
|
||||
elif isinstance(node, ast.Call) and _is_emit(node.func):
|
||||
keys += [kw.arg for kw in node.keywords if kw.arg]
|
||||
stack += [(child, inside) for child in ast.iter_child_nodes(node)]
|
||||
return keys
|
||||
|
||||
|
||||
def _is_emit(func: ast.expr) -> bool:
|
||||
"""``emit(...)`` or ``fluksio.emit(...)``, however it was imported."""
|
||||
if isinstance(func, ast.Name):
|
||||
return func.id == "emit"
|
||||
return isinstance(func, ast.Attribute) and func.attr == "emit"
|
||||
|
||||
|
||||
def _check(spec: NodeSpec, mode: str) -> dict[str, Any]:
|
||||
"""The node as stored, with the settings and wiring of one flow.
|
||||
|
||||
@@ -290,6 +290,13 @@ def _coerce(value: str, dtype: str) -> Any:
|
||||
return value.lower() in ("true", "1", "yes", "on")
|
||||
if dtype == "str":
|
||||
return value
|
||||
if dtype == "artifact" and (
|
||||
value.startswith("@run:") or value.startswith("sha256:")
|
||||
):
|
||||
# The engine turns these into the reference itself. Passing the whole
|
||||
# object as JSON still works, and is what a script that already has one
|
||||
# would do.
|
||||
return value
|
||||
return json.loads(value)
|
||||
|
||||
|
||||
|
||||
@@ -5,14 +5,26 @@ The pipeline half — what a hit restores and what a key is made of — is in
|
||||
"""
|
||||
|
||||
import json
|
||||
from datetime import UTC, datetime
|
||||
|
||||
import pytest
|
||||
from sqlmodel import Session
|
||||
|
||||
from fluksio.core.db import engine as db_engine
|
||||
from fluksio.flow.artifacts import ArtifactStore
|
||||
from fluksio.flow.messages import DType, MessageSpec
|
||||
from fluksio.flow.pipeline import NodeOutcome
|
||||
from fluksio.flow.runs import OUTPUT_CAP, RunCache, _cacheable
|
||||
from fluksio.models import RunNode
|
||||
from fluksio.flow.runs import (
|
||||
OUTPUT_CAP,
|
||||
RunCache,
|
||||
RunRejected,
|
||||
_cacheable,
|
||||
new_run_id,
|
||||
resolve_references,
|
||||
seed_values,
|
||||
)
|
||||
from fluksio.flow.schemas import FlowDef, FlowInput, NodeDef
|
||||
from fluksio.models import Run, RunArtifact, RunNode
|
||||
|
||||
|
||||
def test_a_run_cache_finds_what_an_earlier_run_recorded(tmp_path):
|
||||
@@ -75,3 +87,110 @@ def test_what_may_be_stored_as_a_cache_entry():
|
||||
assert _cacheable(ok.model_copy(update={"cache_key": ""})) is None
|
||||
big = {"study.data": "x" * (OUTPUT_CAP + 1)}
|
||||
assert _cacheable(ok.model_copy(update={"output_values": big})) is None
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Naming an artifact from outside the process that made it
|
||||
#
|
||||
# A python caller passes the reference it holds. A shell holds nothing, so the
|
||||
# same input also takes `@run:<id>.<output>` or a bare digest, resolved here
|
||||
# rather than in each client.
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def artifact_flow() -> FlowDef:
|
||||
"""A flow taking a dataset somebody else's run produced."""
|
||||
dataset = MessageSpec(name="dataset", dtype=DType.ARTIFACT)
|
||||
return FlowDef(
|
||||
name="study",
|
||||
mode="batch",
|
||||
inputs=[FlowInput(spec=dataset)],
|
||||
nodes=[NodeDef(id="train", requires=[dataset])],
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def made_artifact():
|
||||
"""A finished run with one artifact, as a later run would find it."""
|
||||
digest = "sha256:" + "a1" * 32
|
||||
reference = {
|
||||
"digest": digest,
|
||||
"size": 12,
|
||||
"media_type": "text/csv",
|
||||
"name": "cities.csv",
|
||||
}
|
||||
run_id = new_run_id()
|
||||
with Session(db_engine) as session:
|
||||
session.add(
|
||||
Run(
|
||||
id=run_id,
|
||||
flow="prepare",
|
||||
status="ok",
|
||||
result={"dataset": reference},
|
||||
created_at=datetime.now(UTC),
|
||||
)
|
||||
)
|
||||
session.add(
|
||||
RunArtifact(
|
||||
run_id=run_id,
|
||||
name="prepare.dataset",
|
||||
node="load",
|
||||
digest=digest,
|
||||
size=12,
|
||||
)
|
||||
)
|
||||
session.commit()
|
||||
yield run_id, reference
|
||||
with Session(db_engine) as session:
|
||||
session.delete(session.get(RunArtifact, (run_id, "prepare.dataset")))
|
||||
session.delete(session.get(Run, run_id))
|
||||
session.commit()
|
||||
|
||||
|
||||
def test_a_run_reference_resolves_to_what_that_run_produced(made_artifact):
|
||||
run_id, reference = made_artifact
|
||||
resolved = resolve_references(artifact_flow(), {"dataset": f"@run:{run_id}.dataset"})
|
||||
|
||||
# The producer's own reference, file name and all — not one rebuilt from
|
||||
# the row, which carries the message name instead.
|
||||
assert resolved["dataset"] == reference
|
||||
|
||||
|
||||
def test_a_bare_digest_resolves_to_the_bytes_under_it(made_artifact):
|
||||
_run_id, reference = made_artifact
|
||||
resolved = resolve_references(artifact_flow(), {"dataset": reference["digest"]})
|
||||
|
||||
assert resolved["dataset"]["digest"] == reference["digest"]
|
||||
assert resolved["dataset"]["size"] == 12
|
||||
|
||||
|
||||
def test_a_resolved_reference_passes_the_input_check(made_artifact):
|
||||
run_id, _reference = made_artifact
|
||||
flow = artifact_flow()
|
||||
resolved = resolve_references(flow, {"dataset": f"@run:{run_id}.dataset"})
|
||||
|
||||
assert "study.dataset" in seed_values(flow, resolved)
|
||||
|
||||
|
||||
def test_an_output_a_run_never_made_says_what_it_did(made_artifact):
|
||||
run_id, _reference = made_artifact
|
||||
with pytest.raises(RunRejected, match="prepare.dataset"):
|
||||
resolve_references(artifact_flow(), {"dataset": f"@run:{run_id}.weights"})
|
||||
|
||||
|
||||
def test_a_reference_to_no_run_at_all_is_refused():
|
||||
with pytest.raises(RunRejected, match="no run"):
|
||||
resolve_references(artifact_flow(), {"dataset": "@run:nothing.dataset"})
|
||||
|
||||
|
||||
def test_an_unknown_digest_is_refused():
|
||||
with pytest.raises(RunRejected, match="nothing here"):
|
||||
resolve_references(artifact_flow(), {"dataset": "sha256:" + "b2" * 32})
|
||||
|
||||
|
||||
def test_a_reference_passed_whole_is_left_alone(made_artifact):
|
||||
"""A python caller already has the object, and hands it over as one."""
|
||||
_run_id, reference = made_artifact
|
||||
assert resolve_references(artifact_flow(), {"dataset": reference}) == {
|
||||
"dataset": reference
|
||||
}
|
||||
|
||||
@@ -114,3 +114,41 @@ def test_qualify_scopes_bare_names_only():
|
||||
assert qualify("heating", "solar.power") == "solar.power"
|
||||
assert qualify("heating", "") == ""
|
||||
assert flow_of("heating.temp") == "heating"
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# NaN and infinity
|
||||
#
|
||||
# JSON cannot spell either, so one travelling through a port would come back as
|
||||
# a response nobody can parse, a socket frame that stops a canvas, or a row the
|
||||
# database rejects — a long way from the node that produced it.
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_a_float_port_refuses_nan_and_infinity():
|
||||
spec = MessageSpec(name="score", dtype=DType.FLOAT)
|
||||
spec.check(0.5)
|
||||
for value in (float("nan"), float("inf"), float("-inf")):
|
||||
with pytest.raises(TypeError, match="score"):
|
||||
spec.check(value)
|
||||
|
||||
|
||||
def test_a_json_port_refuses_a_nan_nested_in_it():
|
||||
spec = MessageSpec(name="report", dtype=DType.JSON)
|
||||
spec.check({"groups": [{"mean": 1.0}]})
|
||||
with pytest.raises(TypeError, match="JSON"):
|
||||
spec.check({"groups": [{"mean": float("nan")}]})
|
||||
|
||||
|
||||
def test_a_series_refuses_a_nan_point():
|
||||
spec = MessageSpec(name="curve", dtype=DType.SERIES)
|
||||
lines = [{"label": "loss", "points": [[1.0, float("nan")]]}]
|
||||
with pytest.raises(TypeError):
|
||||
spec.check({"lines": lines})
|
||||
|
||||
|
||||
def test_a_value_that_refers_to_itself_does_not_hang_the_check():
|
||||
spec = MessageSpec(name="report", dtype=DType.JSON)
|
||||
loop: dict = {}
|
||||
loop["self"] = loop
|
||||
spec.check(loop)
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
"""The work queue, and what the execution service does with it."""
|
||||
|
||||
import threading
|
||||
import time
|
||||
|
||||
from fluksio.flow import executor
|
||||
from fluksio.flow.executor import ExecutionService
|
||||
from fluksio.flow.messages import DType, MessageSpec
|
||||
from fluksio.flow.nodes import Node
|
||||
@@ -264,3 +266,72 @@ def test_a_replayed_item_does_not_repeat_a_side_effect():
|
||||
service._run_item(item)
|
||||
|
||||
assert calls == [5.0]
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Telling the queue that a long node is working, not lost
|
||||
#
|
||||
# What marks an item abandoned is nobody touching it. A node with no timeout
|
||||
# may run for hours, so the engine holding it says so on a timer — and an
|
||||
# engine that died says nothing, which is the distinction the reaper needs.
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
class RecordingQueue(MemoryWorkQueue):
|
||||
"""A memory queue that writes down what it was asked to hold on to."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
self.touched: list[list[str]] = []
|
||||
|
||||
def touch(self, entry_ids: list[str]) -> None:
|
||||
self.touched.append(list(entry_ids))
|
||||
|
||||
|
||||
def test_work_in_flight_is_touched_until_it_finishes(monkeypatch):
|
||||
monkeypatch.setattr(executor, "TOUCH_INTERVAL_S", 0.0)
|
||||
monkeypatch.setattr(executor, "DELAYED_INTERVAL_S", 0.05)
|
||||
running = threading.Event()
|
||||
release = threading.Event()
|
||||
|
||||
def slow(reading, params):
|
||||
running.set()
|
||||
release.wait(5)
|
||||
return {"doubled": reading * 2}
|
||||
|
||||
source = Node(
|
||||
f=lambda params: None,
|
||||
provides=[MessageSpec(name="reading", dtype=DType.FLOAT)],
|
||||
name="source",
|
||||
)
|
||||
consumer = Node(
|
||||
f=slow,
|
||||
requires=[MessageSpec(name="reading", dtype=DType.FLOAT)],
|
||||
provides=[MessageSpec(name="doubled", dtype=DType.FLOAT)],
|
||||
name="consumer",
|
||||
)
|
||||
source.assign_flow("f", "source")
|
||||
consumer.assign_flow("f", "consumer")
|
||||
|
||||
queue = RecordingQueue()
|
||||
pipeline = Pipeline(
|
||||
nodes=[source, consumer], state=MemoryState(), work_queue=queue
|
||||
)
|
||||
service = ExecutionService(queue)
|
||||
service.bind(pipeline)
|
||||
service.start()
|
||||
try:
|
||||
source.inject({"reading": 3.0})
|
||||
assert running.wait(5)
|
||||
# Give the timer a couple of passes while the node is still in there.
|
||||
time.sleep(0.2)
|
||||
held = [ids for ids in queue.touched if ids]
|
||||
assert held, "a running item was never touched"
|
||||
|
||||
release.set()
|
||||
time.sleep(0.3)
|
||||
# Once it is done it is acknowledged, so there is nothing to hold.
|
||||
assert queue.touched[-1] == []
|
||||
finally:
|
||||
release.set()
|
||||
service.stop()
|
||||
|
||||
@@ -9,10 +9,12 @@ what happened rather than a wait that never ends.
|
||||
|
||||
import asyncio
|
||||
import threading
|
||||
import time
|
||||
from collections.abc import Iterator
|
||||
|
||||
import pytest
|
||||
|
||||
from fluksio.flow import remote
|
||||
from fluksio.flow.remote import NoWorker, RemoteWorker, RemoteWorkerHub
|
||||
from fluksio.flow.workers import NodeTimeout, RemoteError
|
||||
|
||||
@@ -220,3 +222,76 @@ def test_cancelling_a_run_reaches_only_that_run(loop):
|
||||
assert hub.cancel_run("run-a") == 1
|
||||
cancels = [frame for frame in socket.sent if frame.get("op") == "cancel"]
|
||||
assert [frame["call_id"] for frame in cancels] == ["run-a:flow.node"]
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# Liveness, and what a heartbeat is evidence of
|
||||
#
|
||||
# The agent beats every ten seconds while it is executing. That says the agent
|
||||
# is alive; it says nothing about the node, which is why it bounds the silence
|
||||
# deadline and not the node's own timeout.
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_heartbeats_do_not_stave_off_a_nodes_own_timeout(loop):
|
||||
hub = RemoteWorkerHub()
|
||||
worker, socket = attach(hub, loop)
|
||||
caught: list[Exception] = []
|
||||
|
||||
def call() -> None:
|
||||
try:
|
||||
hub.run("gpu", "flow", "node", "src", {}, "flow.node", timeout=0.5)
|
||||
except Exception as exc:
|
||||
caught.append(exc)
|
||||
|
||||
thread = call_in_thread(call)
|
||||
assert socket.arrived.wait(5)
|
||||
call_id = socket.sent[0]["call_id"]
|
||||
# Beating faster than the deadline. A node that reports its progress is
|
||||
# held off; one whose agent is merely alive is not.
|
||||
for _ in range(10):
|
||||
worker.deliver({"call_id": call_id, "event": "heartbeat"})
|
||||
time.sleep(0.1)
|
||||
|
||||
thread.join(timeout=5)
|
||||
assert isinstance(caught[0], NodeTimeout)
|
||||
|
||||
|
||||
def test_with_no_timeout_total_silence_is_still_bounded(loop, monkeypatch):
|
||||
monkeypatch.setattr(remote, "SILENCE_S", 0.3)
|
||||
hub = RemoteWorkerHub()
|
||||
attach(hub, loop)
|
||||
caught: list[Exception] = []
|
||||
|
||||
def call() -> None:
|
||||
try:
|
||||
hub.run("gpu", "flow", "node", "src", {}, "flow.node", timeout=0)
|
||||
except Exception as exc:
|
||||
caught.append(exc)
|
||||
|
||||
call_in_thread(call).join(timeout=5)
|
||||
assert isinstance(caught[0], RemoteError)
|
||||
assert not isinstance(caught[0], NodeTimeout)
|
||||
assert "presumed gone" in str(caught[0])
|
||||
|
||||
|
||||
def test_with_no_timeout_a_beating_worker_is_left_to_finish(loop, monkeypatch):
|
||||
monkeypatch.setattr(remote, "SILENCE_S", 0.3)
|
||||
hub = RemoteWorkerHub()
|
||||
worker, socket = attach(hub, loop)
|
||||
result: dict = {}
|
||||
|
||||
thread = call_in_thread(
|
||||
lambda: result.update(
|
||||
value=hub.run("gpu", "flow", "node", "src", {}, "flow.node", timeout=0)
|
||||
)
|
||||
)
|
||||
assert socket.arrived.wait(5)
|
||||
call_id = socket.sent[0]["call_id"]
|
||||
for _ in range(6):
|
||||
worker.deliver({"call_id": call_id, "event": "heartbeat"})
|
||||
time.sleep(0.1)
|
||||
worker.deliver({"call_id": call_id, "ok": True, "result": {"done": True}})
|
||||
|
||||
thread.join(timeout=5)
|
||||
assert result["value"] == {"done": True}
|
||||
|
||||
@@ -8,6 +8,7 @@ parameters a caller sends are refused before anything runs if they are wrong.
|
||||
|
||||
import pytest
|
||||
|
||||
from fluksio.flow.events import EventBus
|
||||
from fluksio.flow.messages import DType, MessageSpec
|
||||
from fluksio.flow.nodes import Node
|
||||
from fluksio.flow.pipeline import NodeOutcome, Pipeline, run_cache_key
|
||||
@@ -241,17 +242,24 @@ def test_emissions_are_checked_against_the_port_they_name():
|
||||
assert "loss" in seen[0].error
|
||||
|
||||
|
||||
def test_an_emission_that_nothing_declares_is_ignored():
|
||||
def test_an_emission_that_nothing_declares_names_what_was_emitted():
|
||||
"""A mistyped metric name is how a training curve goes missing."""
|
||||
events = []
|
||||
|
||||
def stray(params):
|
||||
yield {"undeclared": 1.0}
|
||||
return {"final_loss": 2.0}
|
||||
|
||||
bus = EventBus()
|
||||
bus.publish = events.append # type: ignore[method-assign]
|
||||
node = make_node("train", "study", stray, provides=[spec("final_loss")])
|
||||
state = MemoryState()
|
||||
Pipeline(nodes=[node], state=state).run()
|
||||
Pipeline(nodes=[node], state=state, events=bus).run()
|
||||
|
||||
(error,) = [e for e in events if e["type"] == "node_error"]
|
||||
assert "undeclared" in error["error"]
|
||||
assert "final_loss" in error["error"]
|
||||
assert "study.undeclared" not in state
|
||||
assert state["study.final_loss"] == 2.0
|
||||
|
||||
|
||||
def test_emissions_reach_the_run_as_a_series_with_a_step_each():
|
||||
@@ -435,3 +443,4 @@ def test_a_node_with_no_fingerprint_is_never_looked_up():
|
||||
assert calls == [1]
|
||||
assert cache.asked == []
|
||||
assert seen[0].cache_key == ""
|
||||
|
||||
|
||||
@@ -373,3 +373,98 @@ def test_a_node_saves_and_loads_an_artifact(tmp_path):
|
||||
assert loaded == {"size": 2048}
|
||||
finally:
|
||||
pool.stop()
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# No timeout at all
|
||||
#
|
||||
# The default: silence is a node working, not a node stuck. What still ends a
|
||||
# call is the worker dying, which arrives as its pipe closing rather than as a
|
||||
# deadline.
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_a_node_with_no_timeout_runs_past_what_the_default_would_have_killed(pool):
|
||||
assert pool.run(
|
||||
"demo",
|
||||
"patient",
|
||||
"import time\n\n\ndef process():\n time.sleep(2)\n return {'out': 1}\n",
|
||||
{},
|
||||
"demo.patient",
|
||||
timeout=0,
|
||||
) == {"out": 1}
|
||||
|
||||
|
||||
def test_a_worker_that_dies_still_fails_promptly_with_no_timeout(pool):
|
||||
started = time.monotonic()
|
||||
with pytest.raises(Exception, match="worker died"):
|
||||
pool.run(
|
||||
"demo",
|
||||
"doomed",
|
||||
"import os\n\n\ndef process():\n os._exit(1)\n",
|
||||
{},
|
||||
"demo.doomed",
|
||||
timeout=0,
|
||||
)
|
||||
# Not waiting out a poll interval: the pipe closing is what wakes the read.
|
||||
assert time.monotonic() - started < 5
|
||||
|
||||
assert run(pool, "def process():\n return {'out': 3}\n") == {"out": 3}
|
||||
|
||||
|
||||
def test_a_node_with_no_timeout_can_still_be_cancelled(pool):
|
||||
def stop_it() -> None:
|
||||
for _ in range(100):
|
||||
if pool.cancel("demo.slow"):
|
||||
return
|
||||
time.sleep(0.05)
|
||||
|
||||
stopper = threading.Thread(target=stop_it)
|
||||
stopper.start()
|
||||
try:
|
||||
with pytest.raises(NodeCancelled):
|
||||
pool.run(
|
||||
"demo",
|
||||
"slow",
|
||||
"import time\n\n\ndef process():\n time.sleep(30)\n",
|
||||
{},
|
||||
"demo.slow",
|
||||
timeout=0,
|
||||
)
|
||||
finally:
|
||||
stopper.join()
|
||||
|
||||
|
||||
def test_an_emission_on_an_undeclared_port_fails_the_call(pool):
|
||||
"""The engine's sink raises, and that has to reach the node's author.
|
||||
|
||||
A yield is held one behind — the last one is the return value when there is
|
||||
no explicit return — so the mistake surfaces on the loop's second pass
|
||||
rather than its first. Which is what a training loop does in a moment, and
|
||||
a long way short of the hours it used to cost.
|
||||
"""
|
||||
from fluksio.flow.nodes.base import NodeOutputError
|
||||
|
||||
def refuse(event):
|
||||
raise NodeOutputError("'demo.gen' produced 'lss', which no port declares")
|
||||
|
||||
started = time.monotonic()
|
||||
with pytest.raises(NodeOutputError, match="lss"):
|
||||
pool.run(
|
||||
"demo",
|
||||
"gen",
|
||||
"import time\n\n\ndef process():\n"
|
||||
" for _ in range(3):\n"
|
||||
" yield {'lss': 1.0}\n"
|
||||
" time.sleep(30)\n"
|
||||
" return {'out': 1}\n",
|
||||
{},
|
||||
"demo.gen",
|
||||
timeout=0,
|
||||
on_event=refuse,
|
||||
)
|
||||
# It did not wait out the node: the failure stopped the call.
|
||||
assert time.monotonic() - started < 10
|
||||
|
||||
# The worker was retired rather than left mid-generator, so the slot works.
|
||||
assert run(pool, "def process():\n return {'out': 4}\n") == {"out": 4}
|
||||
|
||||
@@ -236,3 +236,83 @@ def test_the_decorators_leave_the_function_alone():
|
||||
assert prepare(limit=2)["rows"] == 2
|
||||
assert [step["loss"] for step in fit(None, 0.5, epochs=2)] == [1.0, 0.5]
|
||||
assert evaluate(None) == 0.5
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# What a function emits against what it declares
|
||||
#
|
||||
# The engine refuses an undeclared key at the first yield, which is right but
|
||||
# late — a sweep can be an hour in. A literal one is a typo, and a typo is
|
||||
# readable from the source.
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def trains_with_a_typo(steps=3):
|
||||
for _ in range(steps):
|
||||
yield {"lss": 0.5}
|
||||
return {"final_loss": 0.5}
|
||||
|
||||
|
||||
def trains(steps=3):
|
||||
for _ in range(steps):
|
||||
yield {"loss": 0.5}
|
||||
return {"final_loss": 0.5}
|
||||
|
||||
|
||||
def emits_a_typo(steps=3):
|
||||
import fluksio
|
||||
|
||||
fluksio.emit(lss=0.5)
|
||||
return {"final_loss": 0.5}
|
||||
|
||||
|
||||
def yields_a_name_it_computes(steps=3):
|
||||
for index in range(steps):
|
||||
yield {f"loss_{index}": 0.5}
|
||||
return {"final_loss": 0.5}
|
||||
|
||||
|
||||
def trains_beside_a_helper(steps=3):
|
||||
def every_pair():
|
||||
yield {"internal": 1}
|
||||
|
||||
for _ in range(steps):
|
||||
yield {"loss": 0.5}
|
||||
return {"final_loss": 0.5}
|
||||
|
||||
|
||||
def test_a_yielded_key_no_port_declares_is_refused():
|
||||
with pytest.raises(SyncError, match="lss"):
|
||||
node(provides=[Port("loss", "float"), Port("final_loss", "float")])(
|
||||
trains_with_a_typo
|
||||
)
|
||||
|
||||
|
||||
def test_an_emitted_key_no_port_declares_is_refused():
|
||||
with pytest.raises(SyncError, match="lss"):
|
||||
node(provides=[Port("final_loss", "float")])(emits_a_typo)
|
||||
|
||||
|
||||
def test_declared_keys_pass():
|
||||
assert node(provides=[Port("loss", "float"), Port("final_loss", "float")])(trains)
|
||||
|
||||
|
||||
def test_a_key_the_code_computes_is_left_to_the_engine():
|
||||
"""Only literals are readable here; the rest is checked where it runs."""
|
||||
assert node(provides=[Port("final_loss", "float")])(yields_a_name_it_computes)
|
||||
|
||||
|
||||
def test_a_helper_defined_inside_the_node_is_not_the_nodes_ports():
|
||||
assert node(provides=[Port("loss", "float"), Port("final_loss", "float")])(
|
||||
trains_beside_a_helper
|
||||
)
|
||||
|
||||
|
||||
def test_a_negative_timeout_is_refused():
|
||||
with pytest.raises(SyncError, match="0 or more"):
|
||||
node(requires=["a"], timeout=-1)(one_default)
|
||||
|
||||
|
||||
def test_a_zero_timeout_means_no_limit():
|
||||
decorated = node(requires=["a"], timeout=0)(one_default)
|
||||
assert decorated.__fluksio__.timeout == 0
|
||||
|
||||
Reference in New Issue
Block a user