Say at startup when a flow wants a card, and record one seed rather than two
Docs / docs (push) Successful in 22s
Playwright Tests / test-playwright (1, 2) (push) Failing after 2m46s
Playwright Tests / test-playwright (2, 2) (push) Successful in 1m49s
pre-commit / pre-commit (push) Failing after 1m57s
Test Backend / test-backend (push) Failing after 2m28s
Compose Smoke Test / test-compose (push) Successful in 35s
Playwright Tests / merge-reports (push) Successful in 1m19s

Three things the first pass left.

`serve` now names the flows asking for a GPU when the engine has none
declared. The placer already warned, but into the log, where a fresh install
that forgot `--gpus` does not read it — and the cost of missing it is GPU
nodes running concurrently, which is what the declaration exists to prevent.

The seed was the one field an export still had to coalesce: `--seed 1`
filled the run-level column and left `param.seed` blank, while a declared
seed filled the parameter and left the column blank. It is resolved like
every other input now, and the column carries the seed the run actually used
however it arrived — including when a parameter outranks the run's own,
where the column used to report the one that lost.

And the docs say plainly that declaring the card is what buys the worker
retirement: a node that imports jax without `resources={"gpus": 1}` never
gets CUDA_VISIBLE_DEVICES, so nothing marks its worker as one holding a card.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_019Hra4ndWMCLU5F3KjUuVAc
This commit is contained in:
2026-08-29 15:18:42 +02:00
co-authored by Claude Opus 5
parent 6111b90747
commit 68d2565054
5 changed files with 127 additions and 10 deletions
+34
View File
@@ -82,6 +82,39 @@ def _mention_other_installation(data_dir: Path) -> None:
_say(" `fluksio serve --global` runs that one instead.")
def _mention_undeclared_cards() -> None:
"""Say when a stored flow asks for a card this engine does not have.
Cards are declared rather than detected, so an engine told nothing has
none — and a node asking for one is clamped to zero and runs beside every
other, which on a GPU is the deadlock the declaration exists to prevent.
The placer says so once it happens, into the log; this says it while
somebody is still reading the terminal.
"""
from fluksio.core.config import settings
if settings.FLOW_GPUS:
return
from fluksio.flow.runs import required_resources
from fluksio.flow.store import FlowStore
store = FlowStore(settings.FLOWS_DIR)
asking = []
for name in store.list_flows():
try:
needs = required_resources(store.read_flow(name))
except Exception:
# A flow that will not parse is the engine's to complain about.
continue
if (needs or {}).get("gpus"):
asking.append(name)
if not asking:
return
named = ", ".join(sorted(asking)[:3]) + ("" if len(asking) > 3 else "")
_say(f" Cards 0 declared, but {named} asks for one.")
_say(" Nothing detects them: `--gpus N` says how many are here.")
def _warn_if_networked(path: Path) -> None:
"""A cluster's $HOME is often NFS, and SQLite's WAL does not work there."""
try:
@@ -454,6 +487,7 @@ def cmd_serve(args: argparse.Namespace) -> int:
_say(" fluksio enroll <code>")
_say(f" Signed in as {admin_email}")
_say(f" token in {token_path}")
_mention_undeclared_cards()
_mention_other_installation(data_dir)
# One process: it holds the flow engine, and a second worker would be a
+16 -10
View File
@@ -849,17 +849,23 @@ class RunService:
# What the run actually starts from, not only what was passed: an input
# left out takes its declared value, and a row that records `{}` cannot
# say which. Folded literally — an initial is a value from the
# definition, never a reference to resolve. The run-level seed still
# wins over a declared one, the way `seed_values` has it.
params = {
**{
declared.spec.name: declared.initial
for declared in flow.inputs
if declared.initial is not None
and not (declared.spec.name == "seed" and seed is not None)
},
**params,
# definition, never a reference to resolve.
declared = {
one.spec.name: one.initial for one in flow.inputs if one.initial is not None
}
# The run's own seed fills an input of that name, outranking what the
# flow declares and outranked by one passed as a parameter — the order
# `seed_values` applies, moved to where the record is written.
if seed is not None and any(one.spec.name == "seed" for one in flow.inputs):
declared["seed"] = seed
params = {**declared, **params}
# And back the other way, so the run-level column holds the seed the
# run actually used however it arrived. Otherwise `--seed 1` fills one
# column and a declared seed the other, and that is the single field an
# export still has to coalesce.
resolved_seed = params.get("seed")
if isinstance(resolved_seed, int) and not isinstance(resolved_seed, bool):
seed = resolved_seed
# Checked here rather than in the driver: a caller who mistyped a
# parameter should be told now, not by a run that fails in a minute.
seed_values(flow, params, seed)
+36
View File
@@ -455,6 +455,42 @@ def test_a_run_records_the_inputs_it_actually_starts_from():
session.commit()
def test_the_seed_is_recorded_the_same_way_however_it_arrived():
"""One field an export should not have to coalesce two columns for.
`--seed 1` fills the run's own column; a flow declaring a `seed` input
fills the parameter. Both are the seed the run used, so both are written.
"""
flow = FlowDef(
name="seeded",
mode="batch",
inputs=[FlowInput(spec=MessageSpec(name="seed", dtype=DType.INT), initial=42)],
)
service = RunService(controller=_OneFlow(flow), queue=_Collect())
made = []
try:
passed = service.submit("seeded", {}, seed=1)
made.append(passed.id)
assert (passed.seed, passed.params) == (1, {"seed": 1})
# Nothing passed: the declared value is the seed it ran with, and the
# run-level column says so rather than staying empty.
defaulted = service.submit("seeded", {})
made.append(defaulted.id)
assert (defaulted.seed, defaulted.params) == (42, {"seed": 42})
# A parameter still outranks the run's own seed, as it always has —
# and the column follows it rather than reporting the one that lost.
both = service.submit("seeded", {"seed": 7}, seed=1)
made.append(both.id)
assert (both.seed, both.params) == (7, {"seed": 7})
finally:
with Session(db_engine) as session:
for run in session.exec(select(Run).where(col(Run.id).in_(made))).all():
session.delete(run)
session.commit()
def test_a_key_nobody_used_submits_normally(
client, superuser_token_headers, monkeypatch
):
+35
View File
@@ -538,6 +538,41 @@ def test_how_long_ago_reads_like_a_duration() -> None:
assert _ago(None) == ""
def test_serve_says_when_a_flow_wants_a_card_nobody_declared(
tmp_path, monkeypatch, capsys
) -> None:
"""The clamp warning goes to the log; this is said while someone is reading.
Cards are declared rather than detected, so a fresh install that forgets
`--gpus` clamps a GPU node to zero and runs them all at once.
"""
from fluksio import cli
from fluksio.core.config import settings
from fluksio.flow.schemas import FlowDef, NodeDef, Resources
from fluksio.flow.store import FlowStore
store = FlowStore(tmp_path / "flows")
store.write_flow(
FlowDef(
name="finetune",
mode="batch",
nodes=[NodeDef(id="fit", type="python", resources=Resources(gpus=1))],
)
)
monkeypatch.setattr(settings, "FLOWS_DIR", tmp_path / "flows")
monkeypatch.setattr(settings, "FLOW_GPUS", 0)
cli._mention_undeclared_cards()
said = capsys.readouterr().out
assert "finetune asks for one" in said
assert "--gpus" in said
# Told how many there are, it has nothing to say.
monkeypatch.setattr(settings, "FLOW_GPUS", 1)
cli._mention_undeclared_cards()
assert capsys.readouterr().out == ""
def test_a_serve_limit_is_refused_as_a_flag_not_as_a_traceback(capsys) -> None:
"""These are written into the environment before the settings are built."""
import pytest
+6
View File
@@ -309,6 +309,12 @@ import would hold it after the run finished. The workers that ran on a card
are retired when the run ends, which gives the memory back; what it costs is
the next GPU run paying for its imports again.
**The declaration is what buys that**, not touching the card. A node that
imports jax without `resources={"gpus": 1}` runs on the shared pool, is never
given `CUDA_VISIBLE_DEVICES`, and leaves a warm worker holding whatever it
preallocated. Declaring the card is what makes it a worker the engine knows
to retire — and what stops two such nodes running at once in the first place.
Declaring nothing is the default and is right for most nodes — a poll, a
threshold, a message on its way somewhere. Those share the engine's worker pool
and are given a fair share of `FLOW_CPUS` as a thread cap, which is what stops