Schedule a node across every machine, not just this one
The engine answered "where does this node run" twice, in two ways that could not see each other: a device sent it to a worker carrying that label, and resources were counted against the engine's own cores. Declaring both meant the second answer won and nothing was counted at all — which the data-science getting-started page and the worked example both do. One question now, in flow/placement.py: of every machine attached, which could grant what this node asked for, and which of those has it free. The books move onto each machine — one accountant per worker, built from the inventory it reported — and the waiting moves above them, where one condition variable can be woken by a release anywhere or by a worker attaching. Locks go one way: placer, then a machine's books, never back. So a node asking for a card now finds the box that has one, rather than being clamped down to none and run here. When nothing can grant the ask at all it is still cut down and run — a flow written on a cluster has to work on a laptop — but the ceiling is one real machine now, since taking the largest of each dimension separately can describe a machine nobody has. Two things fixed on the way. A device on a connector node held every batch run of its flow forever, waiting for a worker that could never run an entry point. And `prefer` falling back to the engine skipped the books, so the fallback held nothing. The bench flow's node has taken a `params` argument that with_settings has not forwarded for some time, so the benchmark could not run at all: 62 ms median submit-to-result with this, against the 61 ms on record. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01A6HeySA27EkGANZN95QySW
This commit is contained in:
@@ -1,15 +1,14 @@
|
||||
"""Who gets the machine: accounting, exclusivity, and what the worker is told.
|
||||
"""One machine's books: what it holds, and what is free of it.
|
||||
|
||||
The failure this exists for is not subtle — five concurrent nodes each sizing a
|
||||
thread pool to every core starved the engine's own event loop, and three GPU
|
||||
processes each preallocating most of the card deadlocked at zero utilisation.
|
||||
Both come down to arithmetic nobody was doing, so the arithmetic is what is
|
||||
checked here.
|
||||
checked here. Which machine a node goes to, and the waiting, is
|
||||
``test_placement.py``.
|
||||
"""
|
||||
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
@@ -25,79 +24,70 @@ from fluksio.flow.schemas import NodeDef, Resources
|
||||
|
||||
def test_what_is_free_is_what_was_handed_out():
|
||||
accountant = ResourceAccountant(cpus=4, gpus=0)
|
||||
with accountant.claim(Resources(cpus=3)):
|
||||
assert accountant.snapshot()["cpus"] == {"total": 4, "free": 1}
|
||||
held = accountant.try_take(cpus=3, gpus=0)
|
||||
|
||||
assert accountant.snapshot()["cpus"] == {"total": 4, "free": 1}
|
||||
accountant.give_back(held)
|
||||
assert accountant.snapshot()["cpus"] == {"total": 4, "free": 4}
|
||||
|
||||
|
||||
def test_a_node_waits_until_there_is_room():
|
||||
"""Blocking is the mechanism — the same backpressure a worker slot applies."""
|
||||
def test_asking_for_more_than_is_free_is_answered_not_waited_on():
|
||||
"""The books never block: the placer has other machines to try first."""
|
||||
accountant = ResourceAccountant(cpus=2)
|
||||
running = threading.Event()
|
||||
started = threading.Event()
|
||||
|
||||
def second() -> None:
|
||||
with accountant.claim(Resources(cpus=2), node="study.b"):
|
||||
started.set()
|
||||
|
||||
with accountant.claim(Resources(cpus=2), node="study.a"):
|
||||
thread = threading.Thread(target=second)
|
||||
thread.start()
|
||||
# Long enough to have run if nothing was holding it back.
|
||||
assert not started.wait(0.2)
|
||||
assert accountant.snapshot()["waiting"][0]["node"] == "study.b"
|
||||
running.set()
|
||||
|
||||
thread.join(timeout=5)
|
||||
assert started.is_set()
|
||||
assert accountant.try_take(cpus=2, gpus=0) is not None
|
||||
assert accountant.try_take(cpus=2, gpus=0) is None
|
||||
|
||||
|
||||
def test_a_gpu_is_held_by_one_node_at_a_time():
|
||||
"""The deadlock was three processes each preallocating most of one card."""
|
||||
accountant = ResourceAccountant(cpus=8, gpus=2)
|
||||
with accountant.claim(Resources(cpus=1, gpus=1)) as first:
|
||||
with accountant.claim(Resources(cpus=1, gpus=1)) as second:
|
||||
assert set(first.gpus) & set(second.gpus) == set()
|
||||
assert accountant.snapshot()["gpus"] == {"total": 2, "free": 0}
|
||||
first = accountant.try_take(cpus=1, gpus=1)
|
||||
second = accountant.try_take(cpus=1, gpus=1)
|
||||
|
||||
assert set(first.gpus) & set(second.gpus) == set()
|
||||
assert accountant.snapshot()["gpus"] == {"total": 2, "free": 0}
|
||||
assert accountant.try_take(cpus=1, gpus=1) is None
|
||||
|
||||
accountant.give_back(first)
|
||||
accountant.give_back(second)
|
||||
assert accountant.snapshot()["gpus"] == {"total": 2, "free": 2}
|
||||
|
||||
|
||||
def test_asking_for_more_than_there_is_gets_what_there_is():
|
||||
"""A flow written on a big box still has to run on a laptop."""
|
||||
accountant = ResourceAccountant(cpus=2, gpus=0)
|
||||
with accountant.claim(Resources(cpus=64, gpus=4)) as allocation:
|
||||
assert allocation.cpus == 2
|
||||
assert allocation.gpus == ()
|
||||
|
||||
|
||||
def test_everything_comes_back_when_a_node_fails():
|
||||
def test_what_a_machine_could_ever_grant_is_a_different_question():
|
||||
"""Busy is worth queueing for; too small is not, and reads the same."""
|
||||
accountant = ResourceAccountant(cpus=4, gpus=1)
|
||||
with pytest.raises(ValueError):
|
||||
with accountant.claim(Resources(cpus=4, gpus=1)):
|
||||
raise ValueError("the node raised")
|
||||
assert accountant.fits(cpus=4, gpus=1)
|
||||
assert not accountant.fits(cpus=8, gpus=0)
|
||||
assert not accountant.fits(cpus=1, gpus=2)
|
||||
|
||||
assert accountant.snapshot()["cpus"]["free"] == 4
|
||||
assert accountant.snapshot()["gpus"]["free"] == 1
|
||||
accountant.try_take(cpus=4, gpus=1)
|
||||
# Still true with nothing free: it is about the machine, not the moment.
|
||||
assert accountant.fits(cpus=4, gpus=1)
|
||||
|
||||
|
||||
def test_many_nodes_at_once_all_finish():
|
||||
"""The accountant must not deadlock under contention; it is on every call."""
|
||||
accountant = ResourceAccountant(cpus=4)
|
||||
done = []
|
||||
def test_a_machine_that_said_nothing_about_memory_is_not_a_machine_with_none():
|
||||
said = ResourceAccountant(cpus=4, ram_mb=2048)
|
||||
assert said.fits(cpus=1, gpus=0, ram_mb=2048)
|
||||
assert not said.fits(cpus=1, gpus=0, ram_mb=4096)
|
||||
held = said.try_take(cpus=1, gpus=0, ram_mb=1536)
|
||||
assert said.try_take(cpus=1, gpus=0, ram_mb=1024) is None
|
||||
said.give_back(held)
|
||||
assert said.snapshot()["ram_mb"] == {"total": 2048, "free": 2048}
|
||||
|
||||
def work() -> None:
|
||||
with accountant.claim(Resources(cpus=2), node="study.n"):
|
||||
time.sleep(0.01)
|
||||
done.append(1)
|
||||
quiet = ResourceAccountant(cpus=4)
|
||||
assert quiet.fits(cpus=1, gpus=0, ram_mb=999_999)
|
||||
assert quiet.try_take(cpus=1, gpus=0, ram_mb=999_999) is not None
|
||||
assert quiet.snapshot()["ram_mb"] is None
|
||||
|
||||
threads = [threading.Thread(target=work) for _ in range(8)]
|
||||
for thread in threads:
|
||||
thread.start()
|
||||
for thread in threads:
|
||||
thread.join(timeout=10)
|
||||
|
||||
assert len(done) == 8
|
||||
assert accountant.snapshot()["cpus"]["free"] == 4
|
||||
def test_a_release_says_so_once_it_has_let_go():
|
||||
"""The placer is told outside the lock, which is what keeps the order one-way."""
|
||||
seen: list[dict] = []
|
||||
accountant = ResourceAccountant(cpus=2)
|
||||
accountant.on_release = lambda: seen.append(accountant.snapshot()["cpus"])
|
||||
|
||||
accountant.give_back(accountant.try_take(cpus=2, gpus=0))
|
||||
assert seen == [{"total": 2, "free": 2}]
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
@@ -160,90 +150,3 @@ def test_declaring_nothing_stays_exactly_as_it_was():
|
||||
def test_a_misspelled_resource_is_refused():
|
||||
with pytest.raises(ValueError):
|
||||
Resources(cpu=4)
|
||||
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
# The whole path, through a real worker
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_a_declared_node_is_held_to_its_share(tmp_path):
|
||||
"""Four nodes wanting two cores each, on a machine with four.
|
||||
|
||||
What is checked is the pair: never more than the machine has in flight at
|
||||
once, and every one of them told what it was given — the two halves that
|
||||
together are the oversubscription this is for. Also that it does not
|
||||
deadlock, since the claim is taken before a worker slot and both block.
|
||||
"""
|
||||
import sys
|
||||
|
||||
from fluksio.flow.controller import FlowController, RunContext
|
||||
from fluksio.flow.messages import DType, MessageSpec
|
||||
from fluksio.flow.schemas import FlowDef
|
||||
from fluksio.flow.state import MemoryState
|
||||
from fluksio.flow.store import FlowStore
|
||||
from fluksio.flow.workers import PythonWorkerPool
|
||||
|
||||
store = FlowStore(tmp_path / "flows")
|
||||
store.write_flow(
|
||||
FlowDef(
|
||||
name="study",
|
||||
mode="batch",
|
||||
nodes=[
|
||||
NodeDef(
|
||||
id="fit",
|
||||
provides=[MessageSpec(name="threads", dtype=DType.INT)],
|
||||
resources=Resources(cpus=2, env={"XLA_FLAGS": "--x=false"}),
|
||||
)
|
||||
],
|
||||
)
|
||||
)
|
||||
store.write_node_source(
|
||||
"study",
|
||||
"fit",
|
||||
"import os, time\n\n\ndef process():\n"
|
||||
" time.sleep(0.2)\n"
|
||||
" return {'threads': int(os.environ['OMP_NUM_THREADS'])}\n",
|
||||
)
|
||||
|
||||
pool = PythonWorkerPool(python=sys.executable, size=4)
|
||||
pool.start()
|
||||
accountant = ResourceAccountant(cpus=4)
|
||||
controller = FlowController(store, workers=pool, resources=accountant)
|
||||
answers: list[object] = []
|
||||
in_flight: list[int] = []
|
||||
|
||||
def once(index: int) -> None:
|
||||
pipeline = controller.build_run_pipeline(
|
||||
store.read_flow("study"),
|
||||
state=MemoryState(),
|
||||
run=RunContext(run_id=f"r{index}"),
|
||||
)
|
||||
pipeline.run({})
|
||||
answers.append(pipeline.state.get("study.threads"))
|
||||
|
||||
def watch(until: threading.Event) -> None:
|
||||
while not until.is_set():
|
||||
in_flight.append(4 - int(accountant.snapshot()["cpus"]["free"]))
|
||||
time.sleep(0.01)
|
||||
|
||||
finished = threading.Event()
|
||||
watcher = threading.Thread(target=watch, args=(finished,))
|
||||
watcher.start()
|
||||
threads = [threading.Thread(target=once, args=(i,)) for i in range(4)]
|
||||
try:
|
||||
for thread in threads:
|
||||
thread.start()
|
||||
for thread in threads:
|
||||
thread.join(timeout=60)
|
||||
assert not thread.is_alive(), "a claim and a worker slot deadlocked"
|
||||
finally:
|
||||
finished.set()
|
||||
watcher.join(timeout=5)
|
||||
pool.stop()
|
||||
|
||||
assert answers == [2, 2, 2, 2]
|
||||
assert max(in_flight) <= 4
|
||||
assert accountant.snapshot()["cpus"]["free"] == 4
|
||||
# One environment, so one extra pool however many nodes derived it.
|
||||
assert len(pool._children) == 1
|
||||
|
||||
Reference in New Issue
Block a user