The engine answered "where does this node run" twice, in two ways that could not see each other: a device sent it to a worker carrying that label, and resources were counted against the engine's own cores. Declaring both meant the second answer won and nothing was counted at all — which the data-science getting-started page and the worked example both do. One question now, in flow/placement.py: of every machine attached, which could grant what this node asked for, and which of those has it free. The books move onto each machine — one accountant per worker, built from the inventory it reported — and the waiting moves above them, where one condition variable can be woken by a release anywhere or by a worker attaching. Locks go one way: placer, then a machine's books, never back. So a node asking for a card now finds the box that has one, rather than being clamped down to none and run here. When nothing can grant the ask at all it is still cut down and run — a flow written on a cluster has to work on a laptop — but the ceiling is one real machine now, since taking the largest of each dimension separately can describe a machine nobody has. Two things fixed on the way. A device on a connector node held every batch run of its flow forever, waiting for a worker that could never run an entry point. And `prefer` falling back to the engine skipped the books, so the fallback held nothing. The bench flow's node has taken a `params` argument that with_settings has not forwarded for some time, so the benchmark could not run at all: 62 ms median submit-to-result with this, against the 61 ms on record. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01A6HeySA27EkGANZN95QySW
153 lines
5.4 KiB
Python
153 lines
5.4 KiB
Python
"""One machine's books: what it holds, and what is free of it.
|
|
|
|
The failure this exists for is not subtle — five concurrent nodes each sizing a
|
|
thread pool to every core starved the engine's own event loop, and three GPU
|
|
processes each preallocating most of the card deadlocked at zero utilisation.
|
|
Both come down to arithmetic nobody was doing, so the arithmetic is what is
|
|
checked here. Which machine a node goes to, and the waiting, is
|
|
``test_placement.py``.
|
|
"""
|
|
|
|
import os
|
|
|
|
import pytest
|
|
|
|
from fluksio.flow.resources import (
|
|
THREAD_VARS,
|
|
Allocation,
|
|
ResourceAccountant,
|
|
derive_env,
|
|
fair_share_env,
|
|
)
|
|
from fluksio.flow.schemas import NodeDef, Resources
|
|
|
|
|
|
def test_what_is_free_is_what_was_handed_out():
|
|
accountant = ResourceAccountant(cpus=4, gpus=0)
|
|
held = accountant.try_take(cpus=3, gpus=0)
|
|
|
|
assert accountant.snapshot()["cpus"] == {"total": 4, "free": 1}
|
|
accountant.give_back(held)
|
|
assert accountant.snapshot()["cpus"] == {"total": 4, "free": 4}
|
|
|
|
|
|
def test_asking_for_more_than_is_free_is_answered_not_waited_on():
|
|
"""The books never block: the placer has other machines to try first."""
|
|
accountant = ResourceAccountant(cpus=2)
|
|
assert accountant.try_take(cpus=2, gpus=0) is not None
|
|
assert accountant.try_take(cpus=2, gpus=0) is None
|
|
|
|
|
|
def test_a_gpu_is_held_by_one_node_at_a_time():
|
|
"""The deadlock was three processes each preallocating most of one card."""
|
|
accountant = ResourceAccountant(cpus=8, gpus=2)
|
|
first = accountant.try_take(cpus=1, gpus=1)
|
|
second = accountant.try_take(cpus=1, gpus=1)
|
|
|
|
assert set(first.gpus) & set(second.gpus) == set()
|
|
assert accountant.snapshot()["gpus"] == {"total": 2, "free": 0}
|
|
assert accountant.try_take(cpus=1, gpus=1) is None
|
|
|
|
accountant.give_back(first)
|
|
accountant.give_back(second)
|
|
assert accountant.snapshot()["gpus"] == {"total": 2, "free": 2}
|
|
|
|
|
|
def test_what_a_machine_could_ever_grant_is_a_different_question():
|
|
"""Busy is worth queueing for; too small is not, and reads the same."""
|
|
accountant = ResourceAccountant(cpus=4, gpus=1)
|
|
assert accountant.fits(cpus=4, gpus=1)
|
|
assert not accountant.fits(cpus=8, gpus=0)
|
|
assert not accountant.fits(cpus=1, gpus=2)
|
|
|
|
accountant.try_take(cpus=4, gpus=1)
|
|
# Still true with nothing free: it is about the machine, not the moment.
|
|
assert accountant.fits(cpus=4, gpus=1)
|
|
|
|
|
|
def test_a_machine_that_said_nothing_about_memory_is_not_a_machine_with_none():
|
|
said = ResourceAccountant(cpus=4, ram_mb=2048)
|
|
assert said.fits(cpus=1, gpus=0, ram_mb=2048)
|
|
assert not said.fits(cpus=1, gpus=0, ram_mb=4096)
|
|
held = said.try_take(cpus=1, gpus=0, ram_mb=1536)
|
|
assert said.try_take(cpus=1, gpus=0, ram_mb=1024) is None
|
|
said.give_back(held)
|
|
assert said.snapshot()["ram_mb"] == {"total": 2048, "free": 2048}
|
|
|
|
quiet = ResourceAccountant(cpus=4)
|
|
assert quiet.fits(cpus=1, gpus=0, ram_mb=999_999)
|
|
assert quiet.try_take(cpus=1, gpus=0, ram_mb=999_999) is not None
|
|
assert quiet.snapshot()["ram_mb"] is None
|
|
|
|
|
|
def test_a_release_says_so_once_it_has_let_go():
|
|
"""The placer is told outside the lock, which is what keeps the order one-way."""
|
|
seen: list[dict] = []
|
|
accountant = ResourceAccountant(cpus=2)
|
|
accountant.on_release = lambda: seen.append(accountant.snapshot()["cpus"])
|
|
|
|
accountant.give_back(accountant.try_take(cpus=2, gpus=0))
|
|
assert seen == [{"total": 2, "free": 2}]
|
|
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# What the worker is started with
|
|
# -----------------------------------------------------------------------------
|
|
|
|
|
|
def test_the_share_becomes_the_thread_limit():
|
|
env = derive_env(Resources(cpus=3), Allocation(cpus=3))
|
|
|
|
assert all(env[var] == "3" for var in THREAD_VARS)
|
|
assert "CUDA_VISIBLE_DEVICES" not in env
|
|
|
|
|
|
def test_a_gpu_node_is_told_which_card_is_its():
|
|
env = derive_env(Resources(cpus=1, gpus=2), Allocation(cpus=1, gpus=(1, 3)))
|
|
|
|
assert env["CUDA_VISIBLE_DEVICES"] == "1,3"
|
|
|
|
|
|
def test_what_the_node_asked_for_wins():
|
|
"""A declaration outranks what the allocation would imply — it is deliberate."""
|
|
wanted = Resources(
|
|
cpus=4, env={"OMP_NUM_THREADS": "1", "XLA_FLAGS": "--xla_cpu_x=false"}
|
|
)
|
|
env = derive_env(wanted, Allocation(cpus=4))
|
|
|
|
assert env["OMP_NUM_THREADS"] == "1"
|
|
assert env["XLA_FLAGS"] == "--xla_cpu_x=false"
|
|
assert env["MKL_NUM_THREADS"] == "4"
|
|
|
|
|
|
def test_the_shared_pool_divides_what_it_has():
|
|
env = fair_share_env(cpus=8, workers=4)
|
|
|
|
assert all(env[var] == "2" for var in THREAD_VARS)
|
|
# Never zero, however many workers there are.
|
|
assert fair_share_env(cpus=2, workers=8)["OMP_NUM_THREADS"] == "1"
|
|
|
|
|
|
def test_an_operator_who_set_one_keeps_it(monkeypatch):
|
|
"""An explicit value in the engine's environment is an answer, not a default."""
|
|
monkeypatch.setitem(os.environ, "OMP_NUM_THREADS", "2")
|
|
env = fair_share_env(cpus=16, workers=2)
|
|
|
|
assert "OMP_NUM_THREADS" not in env
|
|
assert env["MKL_NUM_THREADS"] == "8"
|
|
|
|
|
|
# -----------------------------------------------------------------------------
|
|
# Declaration
|
|
# -----------------------------------------------------------------------------
|
|
|
|
|
|
def test_declaring_nothing_stays_exactly_as_it_was():
|
|
"""Every flow that exists today parses unchanged and is not accounted for."""
|
|
assert NodeDef(id="poll").resources is None
|
|
|
|
|
|
def test_a_misspelled_resource_is_refused():
|
|
with pytest.raises(ValueError):
|
|
Resources(cpu=4)
|