Each was a loose end recorded under `### SDK` in the notepad. `serve` takes its own pidfile down on SIGTERM. uvicorn restores the handler it found and re-raises the signal it stopped on, so the default handler ended the process without unwinding and the `finally` never ran — which is what a stop sends, and what left `serve.pid` behind. `serve.log` is cut back past 5 MB by the engine rather than by the screen that started it, so an adopted engine is bounded too. Gated on its own stdout being an appended regular file, which is what makes the cut safe: the kernel then puts the next write at the new end. Cards are counted from `/dev/nvidia[0-9]*`, so `FLOW_GPUS`/`--gpus` of 0 means "work it out" the way `FLOW_CPUS` always has. The engine counts, not the accountant — a remote worker builds one of those from its own inventory, and detecting there would hand it the engine host's cards. The worker counts last: what a batch job says it was granted still wins. `GET /runs/metrics/names` is the distinct over a selection that `--list` and the terminal's metric picker were approximating by reading the newest run that had measured anything, which missed a name only an older run ever wrote. `MetricSink` announces each batch it has written (`run_metric`, carrying the names). Not a per-point event: one covers up to 500 points or two seconds of them, and the rows stay the record. The terminal comparison fills in as the first readings land instead of staying blank until reopened, and the browser refetches the run and any comparison rather than the list behind them. `retry --group` pages the list route by `before` instead of stopping at 500. The terminal dashboard takes the terminal's colours (`ansi-dark`), and the web UI can re-pair from Settings without disconnecting first. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01PRQ9bmTvCbqCwXo9mxZzzV
281 lines
12 KiB
Python
281 lines
12 KiB
Python
import os
|
|
import secrets
|
|
import warnings
|
|
from pathlib import Path
|
|
from typing import Annotated, Any, Literal, Self
|
|
|
|
from pydantic import (
|
|
AnyUrl,
|
|
BeforeValidator,
|
|
EmailStr,
|
|
HttpUrl,
|
|
PositiveInt,
|
|
computed_field,
|
|
model_validator,
|
|
)
|
|
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
|
|
#: Everything the engine keeps on disk, relative to :attr:`Settings.DATA_DIR`.
|
|
#: One setting to move the lot; each still overridable on its own, which is
|
|
#: what the container images do.
|
|
DERIVED_PATHS = {
|
|
"FLOWS_DIR": "flows",
|
|
"SECRETS_FILE": "secrets.enc",
|
|
"ALERTS_FILE": "alerts.json",
|
|
"WEBPUSH_FILE": "webpush.json",
|
|
"PANELS_FILE": "panels.json",
|
|
"OAUTH_PRIVATE_KEY_FILE": "oauth-key.pem",
|
|
"CLOUD_CONFIG_FILE": "cloud.json",
|
|
"PROVISIONERS_FILE": "provisioners.json",
|
|
}
|
|
|
|
|
|
def parse_cors(v: Any) -> list[str] | str:
|
|
if isinstance(v, str) and not v.startswith("["):
|
|
return [i.strip() for i in v.split(",") if i.strip()]
|
|
elif isinstance(v, list | str):
|
|
return v
|
|
raise ValueError(v)
|
|
|
|
|
|
class Settings(BaseSettings):
|
|
model_config = SettingsConfigDict(
|
|
# The stack's own file, one level above ./backend/. An installed
|
|
# `fluksio` has no such tree, so its CLI points this at the data
|
|
# directory instead — and at nothing it might find in the cwd.
|
|
env_file=os.environ.get("FLUKSIO_ENV_FILE", "../.env"),
|
|
env_ignore_empty=True,
|
|
extra="ignore",
|
|
)
|
|
API_V1_STR: str = "/api/v1"
|
|
SECRET_KEY: str = secrets.token_urlsafe(32)
|
|
# 60 minutes * 24 hours * 8 days = 8 days
|
|
ACCESS_TOKEN_EXPIRE_MINUTES: int = 60 * 24 * 8
|
|
FRONTEND_HOST: str = "http://localhost:5173"
|
|
ENVIRONMENT: Literal["local", "staging", "production"] = "local"
|
|
|
|
#: Everything this instance keeps: the database, the flow repository,
|
|
#: secrets, artifacts and the user venv. The paths below derive from it
|
|
#: unless they are set explicitly.
|
|
DATA_DIR: Path = Path("flow-data")
|
|
#: Where the database file goes, as a SQLite URL. The default puts it in
|
|
#: the data directory, which is what makes `fluksio serve` need no
|
|
#: infrastructure at all. **SQLite only**: the upserts the metrics
|
|
#: collector and the run recorder are built on are written against that
|
|
#: dialect, so another backend would parse, migrate, serve — and then
|
|
#: silently drop every rollup and fail every run.
|
|
DATABASE_URL: str | None = None
|
|
|
|
# Flows live on disk as a git repository; secrets stay outside it.
|
|
FLOWS_DIR: Path = Path("flow-data/flows")
|
|
SECRETS_FILE: Path = Path("flow-data/secrets.enc")
|
|
# Which failures reach which channel. Beside the flows, not in them:
|
|
# alerting is the deployment's concern, not any one flow's.
|
|
ALERTS_FILE: Path = Path("flow-data/alerts.json")
|
|
# This instance's web push keypair and the browsers subscribed to it.
|
|
# Beside the alerts it serves; deleting it makes every device subscribe
|
|
# again.
|
|
WEBPUSH_FILE: Path = Path("flow-data/webpush.json")
|
|
# Where machines can be started from when a node needs one and nothing that
|
|
# could take it is attached. Operator-authored, like the alerts beside it,
|
|
# and absent on an instance that has nowhere to start one.
|
|
PROVISIONERS_FILE: Path = Path("flow-data/provisioners.json")
|
|
# Which dashboards each device shows. Beside the flows for the same reason
|
|
# alerting is: where a screen hangs is the deployment's concern rather than
|
|
# any one dashboard's.
|
|
PANELS_FILE: Path = Path("flow-data/panels.json")
|
|
# Which interpreter node code runs on. "auto" adopts the venv the engine
|
|
# was installed into, when it was installed into one and there is no venv
|
|
# of its own to lose — which is the `pip install fluksio` beside your own
|
|
# packages case. "managed" always builds a separate one, which is what a
|
|
# container wants. A path names an interpreter outright.
|
|
NODE_VENV: str = "auto"
|
|
# The MCP endpoint, and the OAuth server agents authenticate against. Off
|
|
# until someone asks for it: it opens client registration to the network.
|
|
MCP_ENABLED: bool = False
|
|
# Unauthenticated test-only endpoints (user seeding). Requires an explicit
|
|
# opt-in on top of ENVIRONMENT=local, so a deployment that merely kept the
|
|
# default environment never exposes them.
|
|
PRIVATE_API_ENABLED: bool = False
|
|
DOMAIN: str = "localhost"
|
|
OAUTH_PRIVATE_KEY_FILE: Path = Path("flow-data/oauth-key.pem")
|
|
# Written only when someone enrols this instance with a portal.
|
|
# Its absence is what keeps remote access off.
|
|
CLOUD_CONFIG_FILE: Path = Path("flow-data/cloud.json")
|
|
OAUTH_CODE_EXPIRE_SECONDS: int = 60
|
|
# Short, because an agent's token is a bearer secret held by a program
|
|
# rather than a person, and it can refresh unattended.
|
|
MCP_TOKEN_EXPIRE_MINUTES: int = 60
|
|
MCP_REFRESH_EXPIRE_DAYS: int = 30
|
|
# The three below are all pool sizes, so 0 says neither "none" nor
|
|
# "unlimited" — it is a pool that cannot be built and an engine that would
|
|
# accept no work. Rejected here rather than quietly read as the default,
|
|
# because a limit somebody set and did not get is the worse surprise.
|
|
FLOW_MAX_WORKERS: PositiveInt = 4
|
|
# How many cascades may be in flight at once. Sustained throughput is this
|
|
# over the mean cascade time, so an instance whose nodes wait on the
|
|
# network rather than on a CPU wants it higher than the core count.
|
|
FLOW_MAX_CASCADES: PositiveInt = 4
|
|
# How many batch runs are driven at once. A different limit from the one
|
|
# above: a run drives a whole graph, and its nodes are bounded by the worker
|
|
# pool rather than by cascade slots. A sweep is what this governs.
|
|
FLOW_MAX_RUNS: PositiveInt = 4
|
|
# How long a python node may be silent before its worker is killed, unless
|
|
# the node sets its own. 0, the default, disables it: a dead worker still
|
|
# fails fast, and a slow one is left to finish. Set it where silence means
|
|
# stuck rather than working.
|
|
FLOW_NODE_TIMEOUT: float = 0.0
|
|
# Cores nodes may be given, for the ones that declare `resources`. 0 works
|
|
# it out: every core but two, which are what keeps the engine's own event
|
|
# loop answering while the machine is busy. Nodes that declare nothing are
|
|
# not accounted against it — they only get its fair share as a thread cap.
|
|
FLOW_CPUS: int = 0
|
|
# GPUs on this machine, each held by one node at a time. 0 counts the
|
|
# NVIDIA device nodes, the way 0 cores works the core count out; anything
|
|
# else — another vendor, or keeping a card back — is said outright.
|
|
FLOW_GPUS: int = 0
|
|
# How long the engine's own metrics, events and run records are kept.
|
|
#: The largest body `PUT /artifacts` will take, in bytes. A checkpoint or a
|
|
#: video segment is a legitimate artifact, so this is generous rather than
|
|
#: small; 0 removes the limit. Nothing capped it at all before, so any
|
|
#: account or worker credential could fill the data volume.
|
|
MAX_ARTIFACT_BYTES: int = 2 * 1024 * 1024 * 1024
|
|
|
|
OBS_RETENTION_DAYS: int = 30
|
|
# How often artifact bytes nothing refers to any more are swept away; 0
|
|
# never sweeps. A flow streaming media writes one artifact per frame, so
|
|
# without this the store only grows.
|
|
ARTIFACT_GC_INTERVAL_S: int = 3600
|
|
# How long a freshly written artifact is spared, whatever refers to it.
|
|
# Storing bytes and recording the reference are two steps; this is the
|
|
# window between them.
|
|
ARTIFACT_GC_GRACE_S: int = 3600
|
|
#: Where frames a flow only shows live are held: memory rather than the
|
|
#: data volume, so a camera at ten frames a second is not writing to an SD
|
|
#: card. Empty works it out — a directory under `/dev/shm` named for this
|
|
#: data directory, so two instances on one host do not trim each other —
|
|
#: and falls back to the temporary directory where there is no `/dev/shm`.
|
|
ARTIFACT_VOLATILE_DIR: Path | None = None
|
|
#: How much the volatile ring holds before the oldest frames fall out of
|
|
#: it; 0 turns it off, and a volatile save then lands in the durable store
|
|
#: like any other. Under Docker's default 64 MB `/dev/shm`: raise
|
|
#: `shm_size` with it.
|
|
ARTIFACT_VOLATILE_BYTES: int = 48 * 1024 * 1024
|
|
# Without a Redis host the engine keeps its state in memory.
|
|
REDIS_HOST: str | None = None
|
|
REDIS_PORT: int = 6379
|
|
|
|
BACKEND_CORS_ORIGINS: Annotated[
|
|
list[AnyUrl] | str, BeforeValidator(parse_cors)
|
|
] = []
|
|
|
|
@model_validator(mode="before")
|
|
@classmethod
|
|
def _derive_data_paths(cls, data: Any) -> Any:
|
|
"""Put every stored thing under ``DATA_DIR`` unless it was named.
|
|
|
|
``setdefault``, so the container images keep their explicit ``/data``
|
|
paths and a checkout keeps ``flow-data/``.
|
|
"""
|
|
if not isinstance(data, dict):
|
|
return data
|
|
base = Path(str(data.get("DATA_DIR", "flow-data"))).expanduser()
|
|
data["DATA_DIR"] = base
|
|
for key, name in DERIVED_PATHS.items():
|
|
data.setdefault(key, base / name)
|
|
return data
|
|
|
|
@computed_field # type: ignore[prop-decorator]
|
|
@property
|
|
def oauth_issuer(self) -> str:
|
|
"""Who issues MCP tokens — this app, on its API host.
|
|
|
|
Kept separate from the app's own URL because a hosted deployment can
|
|
later point agents at a different issuer without the resource server
|
|
changing: it validates whatever issuer it is configured to trust.
|
|
"""
|
|
scheme = "http" if self.ENVIRONMENT == "local" else "https"
|
|
return f"{scheme}://api.{self.DOMAIN}"
|
|
|
|
@computed_field # type: ignore[prop-decorator]
|
|
@property
|
|
def mcp_resource(self) -> str:
|
|
"""The resource an MCP token is issued for (RFC 8707)."""
|
|
return f"{self.oauth_issuer}/mcp"
|
|
|
|
@computed_field # type: ignore[prop-decorator]
|
|
@property
|
|
def all_cors_origins(self) -> list[str]:
|
|
return [str(origin).rstrip("/") for origin in self.BACKEND_CORS_ORIGINS] + [
|
|
self.FRONTEND_HOST
|
|
]
|
|
|
|
PROJECT_NAME: str = "Fluksio"
|
|
SENTRY_DSN: HttpUrl | None = None
|
|
|
|
@computed_field # type: ignore[prop-decorator]
|
|
@property
|
|
def SQLALCHEMY_DATABASE_URI(self) -> str:
|
|
"""SQLite in the data directory, unless a URL names another file.
|
|
|
|
One engine process owns this database — the same reason the image runs
|
|
a single uvicorn worker — so a file beside the flows is the honest
|
|
shape for it, and needs nothing running to be one.
|
|
"""
|
|
if self.DATABASE_URL:
|
|
return self.DATABASE_URL
|
|
return f"sqlite:///{(self.DATA_DIR / 'fluksio.db').expanduser().resolve()}"
|
|
|
|
SMTP_TLS: bool = True
|
|
SMTP_SSL: bool = False
|
|
SMTP_PORT: int = 587
|
|
SMTP_HOST: str | None = None
|
|
SMTP_USER: str | None = None
|
|
SMTP_PASSWORD: str | None = None
|
|
EMAILS_FROM_EMAIL: EmailStr | None = None
|
|
EMAILS_FROM_NAME: str | None = None
|
|
|
|
@model_validator(mode="after")
|
|
def _set_default_emails_from(self) -> Self:
|
|
if not self.EMAILS_FROM_NAME:
|
|
self.EMAILS_FROM_NAME = self.PROJECT_NAME
|
|
return self
|
|
|
|
EMAIL_RESET_TOKEN_EXPIRE_HOURS: int = 48
|
|
|
|
@computed_field # type: ignore[prop-decorator]
|
|
@property
|
|
def emails_enabled(self) -> bool:
|
|
return bool(self.SMTP_HOST and self.EMAILS_FROM_EMAIL)
|
|
|
|
EMAIL_TEST_USER: EmailStr = "test@example.com"
|
|
# Absent means "the CLI will make one on first run" — a pip install is not
|
|
# asked for two environment variables before it can start.
|
|
FIRST_SUPERUSER: EmailStr | None = None
|
|
FIRST_SUPERUSER_PASSWORD: str | None = None
|
|
|
|
def _check_default_secret(self, var_name: str, value: str | None) -> None:
|
|
if value == "changethis":
|
|
message = (
|
|
f'The value of {var_name} is "changethis", '
|
|
"for security, please change it, at least for deployments."
|
|
)
|
|
if self.ENVIRONMENT == "local":
|
|
warnings.warn(message, stacklevel=1)
|
|
else:
|
|
raise ValueError(message)
|
|
|
|
@model_validator(mode="after")
|
|
def _enforce_non_default_secrets(self) -> Self:
|
|
self._check_default_secret("SECRET_KEY", self.SECRET_KEY)
|
|
self._check_default_secret(
|
|
"FIRST_SUPERUSER_PASSWORD", self.FIRST_SUPERUSER_PASSWORD
|
|
)
|
|
|
|
return self
|
|
|
|
|
|
# No arguments and no required environment: a fresh install boots on the
|
|
# defaults above, into a data directory of its own.
|
|
settings = Settings()
|