The engine answered "where does this node run" twice, in two ways that could not see each other: a device sent it to a worker carrying that label, and resources were counted against the engine's own cores. Declaring both meant the second answer won and nothing was counted at all — which the data-science getting-started page and the worked example both do. One question now, in flow/placement.py: of every machine attached, which could grant what this node asked for, and which of those has it free. The books move onto each machine — one accountant per worker, built from the inventory it reported — and the waiting moves above them, where one condition variable can be woken by a release anywhere or by a worker attaching. Locks go one way: placer, then a machine's books, never back. So a node asking for a card now finds the box that has one, rather than being clamped down to none and run here. When nothing can grant the ask at all it is still cut down and run — a flow written on a cluster has to work on a laptop — but the ceiling is one real machine now, since taking the largest of each dimension separately can describe a machine nobody has. Two things fixed on the way. A device on a connector node held every batch run of its flow forever, waiting for a worker that could never run an entry point. And `prefer` falling back to the engine skipped the books, so the fallback held nothing. The bench flow's node has taken a `params` argument that with_settings has not forwarded for some time, so the benchmark could not run at all: 62 ms median submit-to-result with this, against the 61 ms on record. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01A6HeySA27EkGANZN95QySW
228 lines
6.9 KiB
Python
228 lines
6.9 KiB
Python
"""Remote workers: how one attaches, and what is attached right now.
|
|
|
|
A worker dials in rather than being dialled: the GPU box and the engine are
|
|
usually on different networks, and only one of them can be reached. It presents
|
|
a token minted here, says what it can do, and then answers calls on the socket
|
|
it opened.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import logging
|
|
from datetime import timedelta
|
|
from typing import Any
|
|
|
|
from fastapi import APIRouter, Depends, HTTPException, Request, WebSocket
|
|
from fastapi.responses import PlainTextResponse
|
|
from fluksio_worker import worker_main
|
|
from jwt.exceptions import InvalidTokenError
|
|
from pydantic import BaseModel, Field
|
|
|
|
from fluksio.api.deps import get_current_active_superuser, get_current_user
|
|
from fluksio.core import security
|
|
from fluksio.flow.remote import (
|
|
PROTOCOL,
|
|
RemoteWorker,
|
|
RemoteWorkerHub,
|
|
WorkerInventory,
|
|
)
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
router = APIRouter(prefix="/workers", tags=["workers"])
|
|
|
|
#: Long, because a worker is a machine somebody set up once and left running.
|
|
TOKEN_DAYS = 365
|
|
|
|
|
|
class WorkerInfo(BaseModel):
|
|
name: str
|
|
labels: list[str] = Field(default_factory=list)
|
|
max_parallel: int = 1
|
|
in_flight: int = 0
|
|
attached_at: float = 0.0
|
|
last_seen: float = 0.0
|
|
python: str = ""
|
|
venv_digest: str = ""
|
|
cpus: int = 1
|
|
gpus: int = 0
|
|
ram_mb: int | None = None
|
|
|
|
|
|
class TokenRequest(BaseModel):
|
|
name: str
|
|
|
|
|
|
class TokenIssued(BaseModel):
|
|
name: str
|
|
token: str
|
|
expires_days: int = TOKEN_DAYS
|
|
|
|
|
|
def _hub(app: Any) -> RemoteWorkerHub:
|
|
hub: RemoteWorkerHub | None = getattr(app.state, "worker_hub", None)
|
|
if hub is None:
|
|
raise HTTPException(status_code=503, detail="Remote workers are not available")
|
|
return hub
|
|
|
|
|
|
@router.get(
|
|
"", response_model=list[WorkerInfo], dependencies=[Depends(get_current_user)]
|
|
)
|
|
def read_workers(request: Request) -> Any:
|
|
"""What is attached, and how busy it is."""
|
|
return [
|
|
WorkerInfo(
|
|
name=worker.name,
|
|
labels=sorted(worker.labels),
|
|
max_parallel=worker.max_parallel,
|
|
in_flight=worker.in_flight,
|
|
attached_at=worker.attached_at,
|
|
last_seen=worker.last_seen,
|
|
python=str(worker.info.get("python") or ""),
|
|
venv_digest=str(worker.info.get("venv_digest") or ""),
|
|
cpus=worker.inventory.cpus,
|
|
gpus=worker.inventory.gpus,
|
|
ram_mb=worker.inventory.ram_mb,
|
|
)
|
|
for worker in _hub(request.app).workers()
|
|
]
|
|
|
|
|
|
class ResourceLevel(BaseModel):
|
|
total: int
|
|
free: int
|
|
|
|
|
|
class WaitingNode(BaseModel):
|
|
node: str
|
|
reason: str
|
|
seconds: float
|
|
|
|
|
|
class TargetResources(BaseModel):
|
|
"""One machine: this engine, or a worker attached to it."""
|
|
|
|
target: str
|
|
cpus: ResourceLevel
|
|
gpus: ResourceLevel
|
|
#: Absent where the machine did not say how much memory it has.
|
|
ram_mb: ResourceLevel | None = None
|
|
labels: list[str] = Field(default_factory=list)
|
|
in_flight: int = 0
|
|
|
|
|
|
class ResourcesSnapshot(BaseModel):
|
|
#: This engine's own figures, kept where they have always been.
|
|
cpus: ResourceLevel
|
|
gpus: ResourceLevel
|
|
waiting: list[WaitingNode] = Field(default_factory=list)
|
|
targets: list[TargetResources] = Field(default_factory=list)
|
|
provisioners: list[dict[str, Any]] = Field(default_factory=list)
|
|
|
|
|
|
@router.get(
|
|
"/resources",
|
|
response_model=ResourcesSnapshot,
|
|
dependencies=[Depends(get_current_user)],
|
|
)
|
|
def read_resources(request: Request) -> Any:
|
|
"""Every machine, what is free of it, and which nodes are queued.
|
|
|
|
A node waiting its turn looks exactly like a node that has hung — the run
|
|
sits at `running` and says nothing — so what is waiting, and for what, has
|
|
to be readable somewhere.
|
|
"""
|
|
placer = getattr(request.app.state, "placer", None)
|
|
if placer is None:
|
|
raise HTTPException(status_code=503, detail="Resources are not accounted here")
|
|
return placer.snapshot()
|
|
|
|
|
|
@router.post(
|
|
"/tokens",
|
|
response_model=TokenIssued,
|
|
dependencies=[Depends(get_current_active_superuser)],
|
|
)
|
|
def issue_token(body: TokenRequest) -> Any:
|
|
"""Mint the credential a worker presents when it dials in.
|
|
|
|
Shown once. It is signed with the same keypair the agent tokens use, so
|
|
rotating that key revokes every worker along with them.
|
|
"""
|
|
token = security.create_worker_token(body.name, timedelta(days=TOKEN_DAYS))
|
|
return TokenIssued(name=body.name, token=token)
|
|
|
|
|
|
@router.get(
|
|
"/runtime",
|
|
response_class=PlainTextResponse,
|
|
dependencies=[Depends(get_current_user)],
|
|
)
|
|
def read_runtime() -> str:
|
|
"""The worker's own code, so a fresh host installs by fetching one file.
|
|
|
|
It is the same module the engine's local workers run — deliberately
|
|
standard library only, and with nothing of the engine importable in it.
|
|
"""
|
|
return worker_main.__file__ and open(worker_main.__file__).read()
|
|
|
|
|
|
@router.websocket("/attach")
|
|
async def attach(websocket: WebSocket, token: str = "") -> None:
|
|
"""A worker's connection, for as long as it holds.
|
|
|
|
The token goes in the query string for the same reason the dashboard's
|
|
does: a websocket handshake carries no headers of its own.
|
|
"""
|
|
try:
|
|
claims = security.decode_worker_token(token)
|
|
except InvalidTokenError:
|
|
await websocket.close(code=1008)
|
|
return
|
|
|
|
await websocket.accept()
|
|
try:
|
|
hello = await asyncio.wait_for(websocket.receive_json(), timeout=30)
|
|
except (TimeoutError, ValueError):
|
|
await websocket.close(code=1002)
|
|
return
|
|
|
|
if hello.get("op") != "hello" or int(hello.get("protocol", 0)) != PROTOCOL:
|
|
await websocket.send_json(
|
|
{"op": "refused", "reason": f"this engine speaks protocol {PROTOCOL}"}
|
|
)
|
|
await websocket.close(code=1002)
|
|
return
|
|
|
|
# The token names the worker; what it calls itself is a suggestion, so two
|
|
# hosts cannot fight over one identity by claiming the same name.
|
|
name = str(claims.get("sub") or hello.get("name") or "worker")
|
|
hub = _hub(websocket.app)
|
|
worker = RemoteWorker(
|
|
name=name,
|
|
labels=[str(label) for label in (hello.get("labels") or [])],
|
|
send=websocket.send_json,
|
|
loop=asyncio.get_running_loop(),
|
|
max_parallel=max(1, int(hello.get("max_parallel") or 1)),
|
|
info={
|
|
"python": hello.get("python"),
|
|
"venv_digest": hello.get("venv_digest"),
|
|
},
|
|
inventory=WorkerInventory.from_hello(hello.get("inventory")),
|
|
)
|
|
hub.attach(worker)
|
|
await websocket.send_json({"op": "welcome", "protocol": PROTOCOL, "name": name})
|
|
|
|
try:
|
|
while True:
|
|
message = await websocket.receive_json()
|
|
worker.deliver(message)
|
|
except Exception:
|
|
# Any way this ends is the same thing: the socket is gone, and whatever
|
|
# was waiting on it has to be told rather than left hanging.
|
|
logger.info("Worker '%s' disconnected", name)
|
|
finally:
|
|
hub.detach(name)
|