Files
stroblmeandClaude Opus 5 6ff56533f5 Schedule a node across every machine, not just this one
The engine answered "where does this node run" twice, in two ways that could
not see each other: a device sent it to a worker carrying that label, and
resources were counted against the engine's own cores. Declaring both meant the
second answer won and nothing was counted at all — which the data-science
getting-started page and the worked example both do.

One question now, in flow/placement.py: of every machine attached, which could
grant what this node asked for, and which of those has it free. The books move
onto each machine — one accountant per worker, built from the inventory it
reported — and the waiting moves above them, where one condition variable can
be woken by a release anywhere or by a worker attaching. Locks go one way:
placer, then a machine's books, never back.

So a node asking for a card now finds the box that has one, rather than being
clamped down to none and run here. When nothing can grant the ask at all it is
still cut down and run — a flow written on a cluster has to work on a laptop —
but the ceiling is one real machine now, since taking the largest of each
dimension separately can describe a machine nobody has.

Two things fixed on the way. A device on a connector node held every batch run
of its flow forever, waiting for a worker that could never run an entry point.
And `prefer` falling back to the engine skipped the books, so the fallback held
nothing.

The bench flow's node has taken a `params` argument that with_settings has not
forwarded for some time, so the benchmark could not run at all: 62 ms median
submit-to-result with this, against the 61 ms on record.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01A6HeySA27EkGANZN95QySW
2026-08-27 08:49:36 +02:00

228 lines
6.9 KiB
Python

"""Remote workers: how one attaches, and what is attached right now.
A worker dials in rather than being dialled: the GPU box and the engine are
usually on different networks, and only one of them can be reached. It presents
a token minted here, says what it can do, and then answers calls on the socket
it opened.
"""
from __future__ import annotations
import asyncio
import logging
from datetime import timedelta
from typing import Any
from fastapi import APIRouter, Depends, HTTPException, Request, WebSocket
from fastapi.responses import PlainTextResponse
from fluksio_worker import worker_main
from jwt.exceptions import InvalidTokenError
from pydantic import BaseModel, Field
from fluksio.api.deps import get_current_active_superuser, get_current_user
from fluksio.core import security
from fluksio.flow.remote import (
PROTOCOL,
RemoteWorker,
RemoteWorkerHub,
WorkerInventory,
)
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/workers", tags=["workers"])
#: Long, because a worker is a machine somebody set up once and left running.
TOKEN_DAYS = 365
class WorkerInfo(BaseModel):
name: str
labels: list[str] = Field(default_factory=list)
max_parallel: int = 1
in_flight: int = 0
attached_at: float = 0.0
last_seen: float = 0.0
python: str = ""
venv_digest: str = ""
cpus: int = 1
gpus: int = 0
ram_mb: int | None = None
class TokenRequest(BaseModel):
name: str
class TokenIssued(BaseModel):
name: str
token: str
expires_days: int = TOKEN_DAYS
def _hub(app: Any) -> RemoteWorkerHub:
hub: RemoteWorkerHub | None = getattr(app.state, "worker_hub", None)
if hub is None:
raise HTTPException(status_code=503, detail="Remote workers are not available")
return hub
@router.get(
"", response_model=list[WorkerInfo], dependencies=[Depends(get_current_user)]
)
def read_workers(request: Request) -> Any:
"""What is attached, and how busy it is."""
return [
WorkerInfo(
name=worker.name,
labels=sorted(worker.labels),
max_parallel=worker.max_parallel,
in_flight=worker.in_flight,
attached_at=worker.attached_at,
last_seen=worker.last_seen,
python=str(worker.info.get("python") or ""),
venv_digest=str(worker.info.get("venv_digest") or ""),
cpus=worker.inventory.cpus,
gpus=worker.inventory.gpus,
ram_mb=worker.inventory.ram_mb,
)
for worker in _hub(request.app).workers()
]
class ResourceLevel(BaseModel):
total: int
free: int
class WaitingNode(BaseModel):
node: str
reason: str
seconds: float
class TargetResources(BaseModel):
"""One machine: this engine, or a worker attached to it."""
target: str
cpus: ResourceLevel
gpus: ResourceLevel
#: Absent where the machine did not say how much memory it has.
ram_mb: ResourceLevel | None = None
labels: list[str] = Field(default_factory=list)
in_flight: int = 0
class ResourcesSnapshot(BaseModel):
#: This engine's own figures, kept where they have always been.
cpus: ResourceLevel
gpus: ResourceLevel
waiting: list[WaitingNode] = Field(default_factory=list)
targets: list[TargetResources] = Field(default_factory=list)
provisioners: list[dict[str, Any]] = Field(default_factory=list)
@router.get(
"/resources",
response_model=ResourcesSnapshot,
dependencies=[Depends(get_current_user)],
)
def read_resources(request: Request) -> Any:
"""Every machine, what is free of it, and which nodes are queued.
A node waiting its turn looks exactly like a node that has hung — the run
sits at `running` and says nothing — so what is waiting, and for what, has
to be readable somewhere.
"""
placer = getattr(request.app.state, "placer", None)
if placer is None:
raise HTTPException(status_code=503, detail="Resources are not accounted here")
return placer.snapshot()
@router.post(
"/tokens",
response_model=TokenIssued,
dependencies=[Depends(get_current_active_superuser)],
)
def issue_token(body: TokenRequest) -> Any:
"""Mint the credential a worker presents when it dials in.
Shown once. It is signed with the same keypair the agent tokens use, so
rotating that key revokes every worker along with them.
"""
token = security.create_worker_token(body.name, timedelta(days=TOKEN_DAYS))
return TokenIssued(name=body.name, token=token)
@router.get(
"/runtime",
response_class=PlainTextResponse,
dependencies=[Depends(get_current_user)],
)
def read_runtime() -> str:
"""The worker's own code, so a fresh host installs by fetching one file.
It is the same module the engine's local workers run — deliberately
standard library only, and with nothing of the engine importable in it.
"""
return worker_main.__file__ and open(worker_main.__file__).read()
@router.websocket("/attach")
async def attach(websocket: WebSocket, token: str = "") -> None:
"""A worker's connection, for as long as it holds.
The token goes in the query string for the same reason the dashboard's
does: a websocket handshake carries no headers of its own.
"""
try:
claims = security.decode_worker_token(token)
except InvalidTokenError:
await websocket.close(code=1008)
return
await websocket.accept()
try:
hello = await asyncio.wait_for(websocket.receive_json(), timeout=30)
except (TimeoutError, ValueError):
await websocket.close(code=1002)
return
if hello.get("op") != "hello" or int(hello.get("protocol", 0)) != PROTOCOL:
await websocket.send_json(
{"op": "refused", "reason": f"this engine speaks protocol {PROTOCOL}"}
)
await websocket.close(code=1002)
return
# The token names the worker; what it calls itself is a suggestion, so two
# hosts cannot fight over one identity by claiming the same name.
name = str(claims.get("sub") or hello.get("name") or "worker")
hub = _hub(websocket.app)
worker = RemoteWorker(
name=name,
labels=[str(label) for label in (hello.get("labels") or [])],
send=websocket.send_json,
loop=asyncio.get_running_loop(),
max_parallel=max(1, int(hello.get("max_parallel") or 1)),
info={
"python": hello.get("python"),
"venv_digest": hello.get("venv_digest"),
},
inventory=WorkerInventory.from_hello(hello.get("inventory")),
)
hub.attach(worker)
await websocket.send_json({"op": "welcome", "protocol": PROTOCOL, "name": name})
try:
while True:
message = await websocket.receive_json()
worker.deliver(message)
except Exception:
# Any way this ends is the same thing: the socket is gone, and whatever
# was waiting on it has to be told rather than left hanging.
logger.info("Worker '%s' disconnected", name)
finally:
hub.detach(name)