From 11e032386b8740c758234936a54151ae75e9c809 Mon Sep 17 00:00:00 2001 From: stroblme Date: Sat, 22 Aug 2026 05:55:34 +0200 Subject: [PATCH] Publish the documentation site: docs.fluksio.com MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A zensical site under docs/, served by a new `docs` compose service behind Traefik, built with --strict in CI. Same pattern the sibling n3xd workspace uses. Getting started splits the way the landing page does — one path is `pip install fluksio` and a training script, the other is a Docker stack and an afternoon in the browser — because the two audiences will not spend the same amount of time. Everything after that is shared: the concepts, the web interface (app and portal), the CLI and the API, and a reference for node types, payload types and configuration. The three flow guides move here from the docs submodule rather than being copied, so there is one version of them. Styling mirrors DESIGN-GUIDELINES.md: the app's token palette remapped onto Material's variables in both schemes, Inter, the 16px panel radius, and the one terracotta accent spent on the facility lane of the audience split. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_01M7Xv3cJEW5c8AXxn2hoojV --- .gitea/workflows/docs.yml | 27 ++ .gitignore | 3 + Makefile | 18 +- README.md | 15 + docker/Dockerfile.docs | 19 ++ docker/compose.dev.yml | 5 + docker/compose.yml | 30 ++ docker/nginx.docs.conf | 27 ++ docs/assets/favicon-dark.svg | 183 +++++++++++ docs/assets/favicon.svg | 186 +++++++++++ docs/code/agents.md | 111 +++++++ docs/code/api.md | 223 ++++++++++++++ docs/code/cli.md | 153 ++++++++++ docs/code/nodes.md | 202 ++++++++++++ docs/code/workers.md | 137 +++++++++ docs/concepts/flows.md | 183 +++++++++++ docs/concepts/runs.md | 271 ++++++++++++++++ docs/concepts/state.md | 75 +++++ docs/concepts/values.md | 67 ++++ docs/getting-started/data-science.md | 323 ++++++++++++++++++++ docs/getting-started/facility-automation.md | 272 +++++++++++++++++ docs/getting-started/index.md | 72 +++++ docs/index.md | 57 ++++ docs/interface/dashboards.md | 137 +++++++++ docs/interface/flow-editor.md | 156 ++++++++++ docs/interface/index.md | 88 ++++++ docs/interface/operations.md | 131 ++++++++ docs/interface/portal.md | 117 +++++++ docs/llms.txt | 38 +++ docs/reference/configuration.md | 127 ++++++++ docs/reference/node-types.md | 286 +++++++++++++++++ docs/reference/payload-types.md | 129 ++++++++ docs/stylesheets/extra.css | 177 +++++++++++ zensical.toml | 124 ++++++++ 34 files changed, 4167 insertions(+), 2 deletions(-) create mode 100644 .gitea/workflows/docs.yml create mode 100644 docker/Dockerfile.docs create mode 100644 docker/nginx.docs.conf create mode 100644 docs/assets/favicon-dark.svg create mode 100644 docs/assets/favicon.svg create mode 100644 docs/code/agents.md create mode 100644 docs/code/api.md create mode 100644 docs/code/cli.md create mode 100644 docs/code/nodes.md create mode 100644 docs/code/workers.md create mode 100644 docs/concepts/flows.md create mode 100644 docs/concepts/runs.md create mode 100644 docs/concepts/state.md create mode 100644 docs/concepts/values.md create mode 100644 docs/getting-started/data-science.md create mode 100644 docs/getting-started/facility-automation.md create mode 100644 docs/getting-started/index.md create mode 100644 docs/index.md create mode 100644 docs/interface/dashboards.md create mode 100644 docs/interface/flow-editor.md create mode 100644 docs/interface/index.md create mode 100644 docs/interface/operations.md create mode 100644 docs/interface/portal.md create mode 100644 docs/llms.txt create mode 100644 docs/reference/configuration.md create mode 100644 docs/reference/node-types.md create mode 100644 docs/reference/payload-types.md create mode 100644 docs/stylesheets/extra.css create mode 100644 zensical.toml diff --git a/.gitea/workflows/docs.yml b/.gitea/workflows/docs.yml new file mode 100644 index 0000000..cd7560e --- /dev/null +++ b/.gitea/workflows/docs.yml @@ -0,0 +1,27 @@ +name: Docs + +on: + push: + branches: + - main + pull_request: + types: + - opened + - synchronize + +jobs: + docs: + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@v4 + with: + persist-credentials: false + - name: Install uv + uses: astral-sh/setup-uv@v7 + # Pinned to the version docker/Dockerfile.docs ships, so the gate matches + # what docs.fluksio.com actually builds with. --strict fails on warnings + # (dead internal links, unknown config keys), which is the point of + # running it here at all — the build itself takes about a second. + - name: Build the docs site + run: uvx zensical==0.0.46 build --strict --clean diff --git a/.gitignore b/.gitignore index d93df3f..3d98533 100644 --- a/.gitignore +++ b/.gitignore @@ -19,3 +19,6 @@ __pycache__/ # Output of `make verify` (root Makefile). /frontend/screenshots/ *.tsbuildinfo + +# Output of `make docs` (the zensical build of docs/). +/site/ diff --git a/Makefile b/Makefile index d122565..2e7cd8b 100644 --- a/Makefile +++ b/Makefile @@ -4,7 +4,7 @@ .PHONY: dev-utils dev dev-local up down update install dev-backend dev-frontend \ generate-client seed-example seed-demo seed-house seed-aircon seed-hosted-demo test test-backend test-frontend soak bench-startup lint lint-backend \ - lint-frontend format-frontend umami build clean help + lint-frontend format-frontend umami build docs docs-serve clean help COMPOSE_ROOT := $(CURDIR) # Explicit project name keeps this stack isolated from the sibling website @@ -178,9 +178,23 @@ lint-frontend: ## Lint frontend with biome format-frontend: ## Apply biome's fixes to the frontend (what `lint-frontend` only reports) cd frontend && bun run format +# ── Documentation ───────────────────────────────────────────────── +# The public site on docs.${DOMAIN}, served by the `docs` service in +# docker/compose.yml. These targets are for local authoring: zensical runs on +# demand via uvx, so it never touches the backend environment. Pinned to the +# version docker/Dockerfile.docs and .gitea/workflows/docs.yml ship, so local +# authoring builds with what docs.fluksio.com actually gets. +ZENSICAL := zensical==0.0.46 + +docs: ## Build the documentation site into ./site + uvx $(ZENSICAL) build --clean + +docs-serve: ## Serve the documentation site locally with live reload + uvx $(ZENSICAL) serve + # ── Cleanup ─────────────────────────────────────────────────────── clean: ## Remove build artifacts and caches rm -rf frontend/dist frontend/blob-report frontend/test-results - rm -rf backend/.pytest_cache backend/htmlcov + rm -rf backend/.pytest_cache backend/htmlcov site find backend -type d -name __pycache__ -exec rm -rf {} + 2>/dev/null || true diff --git a/README.md b/README.md index c85e4f7..9d518de 100644 --- a/README.md +++ b/README.md @@ -15,6 +15,7 @@ backend/ FastAPI + SQLModel + Alembic + SQLite — the `fluksio` distribu fluksio/cli.py `fluksio serve` / `enroll` / `worker` worker/ the `fluksio-worker` distribution: the agent and the node runner frontend/ React 19 + TanStack Router + Tailwind 4 + shadcn/ui +docs/ the public documentation site (zensical), served on docs.${DOMAIN} docker/ compose.yml → compose.dev.yml → compose.local.yml (+ compose.traefik.yml) scripts/ generate-client.sh, test.sh ``` @@ -62,6 +63,20 @@ make generate-client # regenerate the frontend SDK from the OpenAPI schema ## Documentation +The public site lives in [`docs/`](docs/) and is built with zensical: + +```sh +make docs-serve # live preview on :8000 +make docs # static build into ./site +``` + +It is served at `docs.${DOMAIN}` by the `docs` service in `docker/compose.yml`, +and `.gitea/workflows/docs.yml` builds it with `--strict` on every push. Style +follows the root `DESIGN-GUIDELINES.md`; the tokens are mirrored in +`docs/stylesheets/extra.css`. + +Repo-only material: + - [`ROADMAP.md`](ROADMAP.md) — strategy and feature record - [`NOTEPAD.md`](NOTEPAD.md) — deferred work and findings - [`DESIGN.md`](DESIGN.md) — points at the workspace root's `DESIGN-GUIDELINES.md` diff --git a/docker/Dockerfile.docs b/docker/Dockerfile.docs new file mode 100644 index 0000000..3e59873 --- /dev/null +++ b/docker/Dockerfile.docs @@ -0,0 +1,19 @@ +# ─── Fluksio docs site (zensical) ─── +# Multi-stage: build the static documentation with zensical, serve it with Nginx. +# Routed to docs.${DOMAIN} by the `docs` service in compose.yml. The runtime +# image is just Nginx + static files (no Python), mirroring the SPA approach. + +# ── Stage 1: Build the static site ─────────────────────────────── +FROM python:3.13-slim AS docs-build +# Pin zensical so a rebuild is reproducible (it is only a build-time tool). +RUN pip install --no-cache-dir zensical==0.0.46 +WORKDIR /docs +COPY zensical.toml ./ +COPY docs/ ./docs/ +RUN zensical build --clean + +# ── Stage 2: Serve static files ────────────────────────────────── +FROM nginx:alpine +COPY docker/nginx.docs.conf /etc/nginx/conf.d/default.conf +COPY --from=docs-build /docs/site /usr/share/nginx/html +EXPOSE 80 diff --git a/docker/compose.dev.yml b/docker/compose.dev.yml index b493208..69a4fc3 100644 --- a/docker/compose.dev.yml +++ b/docker/compose.dev.yml @@ -137,6 +137,11 @@ services: - VITE_API_URL=http://localhost:8000 - NODE_ENV=development + # Reachable at http://docs.${DOMAIN}, but a container rebuild per edit is a + # poor authoring loop — use `make docs-serve` for that. + docs: + restart: "no" + playwright: build: context: .. diff --git a/docker/compose.yml b/docker/compose.yml index 9aeaf35..fb78247 100644 --- a/docker/compose.yml +++ b/docker/compose.yml @@ -188,6 +188,36 @@ services: - traefik.http.routers.${STACK_NAME?Variable not set}-frontend-http.middlewares=https-redirect + # Static documentation site (zensical), served at docs.${DOMAIN}. Content and + # config live in ../docs/ and ../zensical.toml; `make docs-serve` previews it. + docs: + container_name: fluksio-docs + restart: always + security_opt: + - no-new-privileges:true + build: + context: .. + dockerfile: docker/Dockerfile.docs + expose: + - "80" + networks: + - proxy + labels: + - traefik.enable=true + - traefik.docker.network=proxy + - traefik.constraint-label=proxy + + - traefik.http.services.${STACK_NAME?Variable not set}-docs.loadbalancer.server.port=80 + + - traefik.http.routers.${STACK_NAME?Variable not set}-docs-http.rule=Host(`docs.${DOMAIN?Variable not set}`) + - traefik.http.routers.${STACK_NAME?Variable not set}-docs-http.entrypoints=http + - traefik.http.routers.${STACK_NAME?Variable not set}-docs-http.middlewares=https-redirect + + - traefik.http.routers.${STACK_NAME?Variable not set}-docs-https.rule=Host(`docs.${DOMAIN?Variable not set}`) + - traefik.http.routers.${STACK_NAME?Variable not set}-docs-https.entrypoints=https + - traefik.http.routers.${STACK_NAME?Variable not set}-docs-https.tls=true + - traefik.http.routers.${STACK_NAME?Variable not set}-docs-https.tls.certresolver=le + # Docker never restarts a merely *unhealthy* container on its own; autoheal # closes that gap for the services labeled autoheal=true. # diff --git a/docker/nginx.docs.conf b/docker/nginx.docs.conf new file mode 100644 index 0000000..03ba066 --- /dev/null +++ b/docker/nginx.docs.conf @@ -0,0 +1,27 @@ +# Static documentation site (zensical output). Clean-URL directories served from +# per-directory index.html, with a styled 404, gzip, and long-cached assets. +server { + listen 80; + server_name _; + root /usr/share/nginx/html; + index index.html; + + gzip on; + gzip_types text/css application/javascript application/json image/svg+xml text/plain; + gzip_min_length 256; + + location / { + try_files $uri $uri/ =404; + } + + # Fingerprinted CSS/JS/images emitted under assets/ can be cached hard. + location /assets/ { + expires 30d; + add_header Cache-Control "public, immutable"; + } + + error_page 404 /404.html; + location = /404.html { + internal; + } +} diff --git a/docs/assets/favicon-dark.svg b/docs/assets/favicon-dark.svg new file mode 100644 index 0000000..cccdc8a --- /dev/null +++ b/docs/assets/favicon-dark.svg @@ -0,0 +1,183 @@ + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/docs/assets/favicon.svg b/docs/assets/favicon.svg new file mode 100644 index 0000000..55fa8ac --- /dev/null +++ b/docs/assets/favicon.svg @@ -0,0 +1,186 @@ + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/docs/code/agents.md b/docs/code/agents.md new file mode 100644 index 0000000..29527b3 --- /dev/null +++ b/docs/code/agents.md @@ -0,0 +1,111 @@ +# Agents over MCP + +The engine exposes the flow API to agents over the +[Model Context Protocol](https://modelcontextprotocol.io). Point Claude Code, +Claude Desktop, Codex or anything else that speaks MCP at it, and it can read, +build, publish and run flows. + +## Switching it on + +```sh +MCP_ENABLED=true +``` + +It is off by default, because switching it on opens client registration to +whoever can reach the host. Turn it on deliberately. + +The endpoint is `https://api.${DOMAIN}/mcp` — or `http://api.localhost/mcp` on +a local stack. + +## Connecting a client + +```sh +claude mcp add --transport http fluksio https://api.example.com/mcp +``` + +The OAuth flow does the rest: the client registers itself, you approve it in +the browser, and it exchanges the grant for a token. Nothing to paste. + +## What an agent may do + +An agent acts as **the person who approved it**, and only through the same REST +API the dashboard uses. It cannot do anything you could not do in the browser, +and it cannot do anything *you* cannot do — an agent approved by a non-superuser +is not one either. + +Its tokens are signed with a keypair of their own, kept beside the flow store. +Deleting that file revokes every agent's access without logging anyone out of +the dashboard. + +Registered clients are managed under **Admin**. + +## The tools + +Read-only: + +| Tool | Returns | +|---|---| +| `list_flows` | every flow with its node count and state | +| `get_flow` | one flow's definition, node status and issues | +| `get_node_source` | a node's Python | +| `list_node_types` | every type and its parameter schema | +| `get_flow_state` | the current value of each of a flow's messages | +| `get_message_history` | one message's recent numeric values | +| `get_graph` | every flow as one graph | +| `list_shared_nodes` | the shared library and its usages | +| `list_secrets` | secret **names** — never values | +| `get_modules` | the packages node code can import | +| `get_health` | flows, nodes, queue and loop lag | +| `get_metrics` | executions, errors and timings per minute | +| `list_failures` | what went wrong recently | +| `list_runs` | recent cascades | + +Editing: + +| Tool | Does | +|---|---| +| `save_flow` | write a draft | +| `save_node_source` | write a node's code into the draft | +| `validate_flow` | the issues, without saving | +| `publish_flow` | put the draft live (takes the version it saw) | +| `discard_draft` | throw the draft away | +| `delete_flow` | delete it | +| `apply_modules` | replace the package manifest | + +Operating: + +| Tool | Does | +|---|---| +| `run_flow` | run every node once from the values given | +| `trigger_node` | feed one node and run what is downstream | +| `start_flow` / `stop_flow` | activate or tear down | +| `pause_flow` / `resume_flow` | hold messages, or let them through | +| `cancel_node` | kill a node's code while it runs | + +## Why it is shaped this way + +The tools do not reach into the engine. Each one calls the same REST endpoint +the dashboard calls, over an in-process transport. + +That keeps exactly one description of what a flow is and how it may be changed +— the validation, the draft/publish split, the version check that stops two +clients overwriting each other. An agent that saved a flow through a private +back door would be an agent that could write one the canvas cannot open. + +The caller's token rides along on every hop, so the API sees the agent's own +identity rather than a service account. The audit trail on Home names it. + +## A note on working this way + +The pattern that works is the same one that works with a person: let the agent +read the flow and its issues, let it write a **draft**, and publish yourself +once you have looked. `validate_flow` before `publish_flow` is cheap and tells +you what the canvas would have told you. + +## Limits + +- MCP is not currently reachable through a [portal](../interface/portal.md) + tunnel — the proxy forwards `/api/v1/` only. Connect an agent on the same + network as the installation. +- Secrets are never readable, by an agent or by anyone else. `list_secrets` + returns names. diff --git a/docs/code/api.md b/docs/code/api.md new file mode 100644 index 0000000..254befe --- /dev/null +++ b/docs/code/api.md @@ -0,0 +1,223 @@ +# The HTTP API + +Everything the browser does, the API does first. The dashboard is a generated +client of this schema, not a privileged path into the engine — so anything you +can click, you can script. + +Base URL: `https://api.${DOMAIN}/api/v1`, or `http://127.0.0.1:8000/api/v1` for +a `fluksio serve` installation. + +```sh +export FLUKSIO=http://127.0.0.1:8000/api/v1 +``` + +## Authenticating + +```sh +export TOKEN=$(curl -s -X POST $FLUKSIO/login/access-token \ + -d "username=admin@example.com&password=..." | jq -r .access_token) + +curl -s $FLUKSIO/users/me -H "Authorization: Bearer $TOKEN" +``` + +A bearer token, valid for eight days. `POST /login/test-token` checks one. +Password recovery and reset are `POST /password-recovery/{email}` and +`POST /reset-password/`. + +Agents authenticate differently — see [Agents over MCP](agents.md). + +!!! tip "The interactive schema" + + When `ENVIRONMENT` is not `production`, the full OpenAPI schema is at + `/docs` (Swagger) and `/redoc`, and the raw document at + `/api/v1/openapi.json`. That is the authoritative reference; this page is + the tour. + + It is closed in production on purpose: the schema enumerates every endpoint + the installation serves, including the paths webhook nodes mounted at + runtime. + +## Flows + +| Method | Path | What | +|---|---|---| +| `GET` | `/flows/` | every flow, with node counts, draft and running state | +| `GET` | `/flows/{name}` | one flow — the draft if there is one — with node status and issues | +| `PUT` | `/flows/{name}` | save a draft (`version` must match, or 409) | +| `POST` | `/flows/{name}/publish` | put the draft live | +| `POST` | `/flows/{name}/discard-draft` | throw the draft away | +| `POST` | `/flows/{name}/rename` | rename it | +| `DELETE` | `/flows/{name}` | delete it | +| `GET` | `/flows/node-types` | every node type and its parameter schema | +| `GET` | `/flows/graph` | every flow as one graph — what Home draws | + +Reading a flow returns the *draft* when one exists, along with a preview of how +it would run: node statuses and validation issues for the version you are +editing, not the one running underneath it. + +### Node source + +```sh +curl -s $FLUKSIO/flows/house/nodes/decide/source -H "Authorization: Bearer $TOKEN" +curl -X PUT $FLUKSIO/flows/house/nodes/decide/source -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' -d '{"code": "def process(x):\n return {\"y\": x * 2}\n"}' +``` + +`POST .../share` promotes a node's source to the shared library; +`POST .../unshare` gives it a private copy back. `GET /flows/library` lists the +shared sources and which flows use each. + +### Running and controlling + +| Method | Path | What | +|---|---|---| +| `POST` | `/flows/{name}/start` · `/stop` | activate or tear down its subscriptions and schedules | +| `POST` | `/flows/{name}/pause` · `/resume` | hold messages instead of running them | +| `POST` | `/flows/{name}/step` | release exactly one held message | +| `POST` | `/flows/{name}/validate` | the issues, without saving | +| `POST` | `/flows/{name}/run` | run every node once, from the values you pass | +| `POST` | `/flows/{name}/nodes/{id}/trigger` | feed values into one node | +| `POST` | `/flows/{name}/nodes/{id}/cancel` | kill the worker running it right now | +| `POST` | `/flows/{name}/nodes/{id}/acknowledge` | clear the failure the canvas is marking | + +`POST /flows/{name}/run` on a **batch** flow submits a run instead, because +that is what running one means — the parameters, the series and the result are +the point, and a call that quietly did something else would be a trap. + +## Messages + +```sh +curl -s $FLUKSIO/messages/ -H "Authorization: Bearer $TOKEN" +curl -X POST $FLUKSIO/messages/house.setpoint -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' -d '{"value": 22.5}' +curl -s $FLUKSIO/messages/house.temperature/history -H "Authorization: Bearer $TOKEN" +``` + +`GET /messages/` is the catalogue: every message any published flow declares, +with its last value. Publishing puts a value into the graph exactly as a +dashboard control does — which means only a message some flow *declares* can be +published to. Flows own the namespace; everything else is a client of it. + +## Runs + +| Method | Path | What | +|---|---|---| +| `POST` | `/runs/flows/{name}` | queue one run — `{"params": {...}, "seed": 7, "draft": false}` | +| `POST` | `/runs/flows/{name}/sweep` | queue many, sharing a `group_id` | +| `GET` | `/runs` | the queryable history: `?flow=`, `?status=`, `?group=`, `?digest=`, `?limit=` | +| `GET` | `/runs/{id}` | one run in full: params, result, per-node record, artifacts | +| `POST` | `/runs/{id}/cancel` | stop it | +| `GET` | `/runs/{id}/metrics?name=&stride=` | one metric's series, in step order | +| `GET` | `/runs/series/compare?ids=a,b,c&metric=` | that metric across several runs | + +Submitting answers immediately with a `queued` run. Wrong parameters — an +undeclared name, a value of the wrong type — come back as a 422 naming the +problem, before anything executes. + +`?digest=` filters by the hash of the parameters, which is how you find "every +run that used exactly this configuration". + +`compare` answers in the same shape a chart widget draws, so three training +curves side by side is a widget binding rather than a screen of its own. + +See [Runs: pipelines that finish](../concepts/runs.md). + +## Artifacts + +```sh +curl -X PUT $FLUKSIO/artifacts -H "Authorization: Bearer $TOKEN" \ + --data-binary @model.pt +curl -s $FLUKSIO/artifacts/sha256:abc... -H "Authorization: Bearer $TOKEN" -o model.pt +``` + +Content-addressed, so uploading the same bytes twice stores them once. Node +code normally reaches these through `fluksio.save_artifact` / +`fluksio.load_artifact` rather than here. + +## Dashboards and panels + +| Method | Path | What | +|---|---|---| +| `GET` | `/dashboards/` | every dashboard, without its contents | +| `GET` | `/dashboards/{name}?draft=true` | the published document, or the editor's copy | +| `POST` | `/dashboards/{name}` | create one | +| `PUT` | `/dashboards/{name}` | save a draft | +| `POST` | `/dashboards/{name}/publish` · `/discard` · `/rename` | | +| `GET` `PUT` | `/panels/` | which device shows which dashboards | +| `POST` | `/panels/pair` | start a pairing | +| `GET` | `/panels/pair/{code}` | what is holding a code | + +## Secrets, modules, alerts + +```sh +curl -s $FLUKSIO/secrets/ -H "Authorization: Bearer $TOKEN" # names only +curl -X PUT $FLUKSIO/secrets/influx-token -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' -d '{"value": "..."}' + +curl -s $FLUKSIO/modules/ -H "Authorization: Bearer $TOKEN" +curl -X POST $FLUKSIO/modules/apply -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' -d '{"requirements": "numpy>=2\n"}' + +curl -s $FLUKSIO/alerts/config -H "Authorization: Bearer $TOKEN" +curl -X POST $FLUKSIO/alerts/test/my-phone -H "Authorization: Bearer $TOKEN" +``` + +Secrets are write-only over the API: you can list names and set values, never +read one back. + +## Observability + +| Path | What | +|---|---| +| `/observability/summary` | engine health — always 200, degraded or not | +| `/observability/timeseries` | executions and failures over a window | +| `/observability/flows` | per-flow rollups with a 60-slice trend | +| `/observability/runs` | recent cascades, with `?flow=`, `?since=`, `?until=` | +| `/observability/events?kind=failure\|audit` | what went wrong, or who changed what | +| `/observability/dead-letter` | work the engine gave up on | + +`GET /utils/health/` is the deep health check the container probe uses: it +fails when the event loop is wedged or the state backend is gone, not just when +the process is dead. + +## Workers + +| Method | Path | What | +|---|---|---| +| `GET` | `/workers` | what is attached, its labels and how busy it is | +| `POST` | `/workers/tokens` | mint a worker credential (superuser; shown once) | +| `GET` | `/workers/runtime` | the node runner's source, for a host without pip | +| `WS` | `/workers/attach` | where a worker dials in | + +See [Remote workers](workers.md). + +## Live events + +```text +ws://127.0.0.1:8000/api/v1/flows/ws?token= +``` + +The websocket authenticates from its query string. It sends a snapshot on +connect and then every engine event: node executions, values published, health +changes, run started and finished. This is what the canvas and the dashboards +draw from. + +## Errors + +| Code | Means | +|---|---| +| `400` | the request was malformed, or the node refused it with a message | +| `401` / `403` | not signed in, or not allowed | +| `404` | no such flow, dashboard, run or message | +| `409` | someone else saved first — the body carries `current_version` | +| `422` | a parameter, port or binding did not typecheck | +| `503` | that subsystem is not available on this installation | + +A 409 on a save or a publish is not an error to retry blindly: it means the +stored version moved past the one you were editing. Re-read, merge, save again. + +## Generating a client + +The frontend's TypeScript client is generated from the OpenAPI schema +(`make generate-client`). Any OpenAPI generator will do the same for your +language — point it at `/api/v1/openapi.json` on a non-production installation. diff --git a/docs/code/cli.md b/docs/code/cli.md new file mode 100644 index 0000000..9926781 --- /dev/null +++ b/docs/code/cli.md @@ -0,0 +1,153 @@ +# The `fluksio` command + +```sh +pip install fluksio +``` + +Installs the engine and the `fluksio` command. Python 3.10 or newer, Linux or +macOS. + +There is a second, smaller distribution — `fluksio-worker` — for a machine that +should only *run nodes* for an engine elsewhere. It has none of the engine in +it. See [Remote workers](workers.md). + +## `fluksio serve` + +Runs the engine. + +```sh +fluksio serve +``` + +On the first start it creates an admin account and prints its password **once**. +Nothing else has to be running: no database server, no message broker, no +Docker. + +| Option | Default | What it does | +|---|---|---| +| `--data-dir PATH` | `~/.fluksio` (or `$FLUKSIO_HOME`) | where this installation keeps everything | +| `--host HOST` | `127.0.0.1` | what to bind | +| `--port PORT` | `8000` | what to listen on | +| `--log-level LEVEL` | `info` | uvicorn's log level | +| `--admin-email ADDR` | `admin@example.com` | the account created on first run | +| `--admin-password PW` | generated | set it instead of having one generated | +| `--enroll CODE` | — | pair with a portal as part of coming up | +| `--portal URL` | — | the portal `--enroll` redeems at | + +`--enroll` with `--portal` is the one-command setup: it pairs before the engine +starts, so the connection is dialled as part of coming up rather than needing a +restart. It is skipped if the installation is already enrolled. + +!!! warning "One process" + + `fluksio serve` holds the flow engine. A second one is a *second engine* — + duplicated subscriptions, duplicated cron ticks, two webhooks answering the + same path. Run one, and distribute work with + [workers](workers.md) instead. + +!!! note "`$HOME` on a cluster" + + A login node's home directory is often NFS, where SQLite's write-ahead log + does not work — the database would be locked or corrupt. `fluksio serve` + warns when it notices; point `--data-dir` at local disk. + +### What it prints + +```text +Created the admin account admin@example.com + password: k3Qm-8vTpLdX + Shown once. Change it from the dashboard. +Fluksio 0.1.0 — data in /home/you/.fluksio + API http://127.0.0.1:8000/api/v1 + No portal. Pair this installation with: + fluksio enroll --portal https://hub.example.com +``` + +An enrolled installation says which portal it is on instead, and notes that the +dashboard is served from there rather than here. + +## `fluksio enroll` + +Pairs an existing installation with a portal. + +```sh +fluksio enroll ABCD-1234 --portal https://hub.fluksio.com +``` + +| Option | What it does | +|---|---| +| `--portal URL` | **required** — the portal the code was minted on | +| `--as EMAIL` | the local account a portal session arrives as | +| `--data-dir PATH` | which installation, if not the default | + +Get the code from the portal under **Installations → Add installation**. It is +single-use and expires in fifteen minutes. `--as` matters when the installation +has several superusers — without it, enrolment refuses rather than guessing. + +Afterwards, `fluksio serve` dials the portal as it comes up. See +[Accounts and the portal](../interface/portal.md). + +## `fluksio worker` + +Runs nodes for an engine elsewhere. Everything after `worker` belongs to the +agent's own parser — it is the same program `fluksio-worker` installs, so the +two are interchangeable: + +```sh +fluksio worker --url wss://api.example.com/api/v1/workers/attach \ + --token "$FLUKSIO_WORKER_TOKEN" --labels gpu +``` + +See [Remote workers](workers.md). + +## What lives in the data directory + +```text +~/.fluksio/ +├── fluksio.db SQLite: users, runs, metrics, observability, agents +├── flows/ a git repository — one directory per flow +│ ├── house/ +│ │ ├── flow.json the published structure +│ │ ├── nodes/*.py the published node code +│ │ ├── flow.draft.json unpublished edits, if any +│ │ └── nodes.draft/*.py +│ ├── _lib/ shared node sources +│ ├── _dashboards/ dashboards, drafts and all +│ └── requirements.txt what the Modules screen installs +├── artifacts/ content-addressed bytes, two levels deep +├── user-venv/ the interpreter your node code runs on +├── secrets.enc encrypted credentials, deliberately outside flows/ +├── alerts.json alert channels and rules +├── panels.json wall-panel pairings +├── oauth-key.pem signs agent tokens +├── cloud.json the portal enrolment, if there is one +├── secret_key signs sessions and derives the secrets key +└── env optional settings file +``` + +Two things follow from this layout and are worth internalising: + +**`flows/` is a real git repository.** `git log` is the history of every change +anyone made to any flow. A run records the commit it ran at, so `git show` on +that hash is literally the code that produced the number. + +**Backing up the data directory backs up the installation.** Everything else is +rebuildable. Copy it while the engine is stopped, or use SQLite's online backup +for the database if it is not. + +## Settings + +Settings come from the environment, or from an `env` file in the data +directory. The ones you are most likely to touch: + +| Variable | Default | What it does | +|---|---|---| +| `DATA_DIR` | `~/.fluksio` via the CLI | everything below it derives from this | +| `DATABASE_URL` | SQLite in the data dir | any SQLAlchemy URL | +| `REDIS_HOST` | unset | flow state in Redis instead of memory; survives a restart | +| `FRONTEND_HOST` | — | the address used in mails, OAuth metadata and panel links | +| `ENVIRONMENT` | `local` | `production` closes the interactive API schema | +| `MCP_ENABLED` | `false` | opens the [agent endpoint](agents.md) | +| `SECRET_KEY` | generated once | signs sessions, derives the secrets key | + +The full list is in [Configuration](../reference/configuration.md). diff --git a/docs/code/nodes.md b/docs/code/nodes.md new file mode 100644 index 0000000..6f6ad33 --- /dev/null +++ b/docs/code/nodes.md @@ -0,0 +1,202 @@ +# Writing node code + +A Function node is a Python file. That is all it is — no base class, no +decorator, no framework import unless you want one. + +```python +def process(temperature, setpoint=21.0): + """Ask for heat when the room is below the comfort point.""" + return {"heat": temperature < setpoint} +``` + +## The rules + +**One function called `process`.** If the file defines exactly one public +function under another name, that one is used instead. Two, and the node +refuses to load rather than guessing. + +**Arguments come from ports and settings, by name.** `temperature` above is an +input port; `setpoint` is a setting typed into the node's panel. Both arrive as +keyword arguments, which is why a setting may not share a name with a port. +See [Where a node's values come from](../concepts/values.md). + +**The return value is a dict keyed by output ports.** Every value is checked +against the port's declared type before it is published. A key that is not a +declared port is an error, not a silent drop — nothing leaves a node except +through a port it declared. + +**Nothing else is importable from the engine.** Node code runs in a separate +process, on a separate interpreter, with none of Fluksio's own modules on its +path. What it can import is what the [Modules](../interface/operations.md) +screen installed. + +**A node is a pure function of its inputs.** No context object, no global +store, no handle to reach for. A running total or a debounce timer has a +specific shape — see [Keeping state in a flow](../concepts/state.md). + +## Producing values over time + +A node that produces values *during* its execution is a generator. Every +`yield` is a dict keyed by output port, published the instant it happens: + +```python +def process(lr, steps): + loss = 1.0 + for _ in range(steps): + loss = train_one_step(lr) + yield {"loss": loss} # published now + return {"final_loss": loss} +``` + +Whatever the generator `return`s at the end is the node's result — what +downstream nodes read. If you never `return`, the last thing you yield is the +result instead. + +Mark the port so the flow says what it does: + +```json +{"name": "loss", "dtype": "float", "stream": true} +``` + +Two consequences. In a [run](../concepts/runs.md), the whole series is kept as +that run's metrics — this is why there is no `log_metric()` anywhere in the +API. And **the node's timeout starts measuring silence rather than duration**: +each emission resets the deadline, so a node yielding every few seconds can run +for hours under a timeout of 300. + +### `fluksio.emit` + +Where a `yield` cannot reach — the value comes from inside somebody else's +callback, and they call you rather than the other way round: + +```python +import fluksio + + +def process(): + model.fit(callbacks=[LambdaCallback( + on_epoch_end=lambda epoch, logs: fluksio.emit(loss=logs["loss"]) + )]) + return {"weights": ...} +``` + +Same ports, same type checking, same publication. Prefer `yield` where you can +reach it; `emit` where you cannot. + +## Bytes: artifacts + +Messages are JSON, which is what lets the same value pass through Redis, the +work queue and the worker protocol unchanged. A checkpoint is not that. + +```python +import fluksio + + +def process(dataset): + path = fluksio.load_artifact(dataset) # → a local path to read + ... + return { + "weights": fluksio.save_artifact("model.pt", media_type="application/octet-stream"), + "score": 0.94, + } +``` + +`save_artifact` takes bytes or a path, stores them by their SHA-256 digest, and +returns a small reference — digest, size, media type, name — which is what an +`artifact`-typed port carries. + +Because the address is the content's hash, a sweep whose fifty configs share +one preprocessed input stores it once, and a reference stays valid wherever the +store is reachable from — including on another machine. + +## Printing + +`print` works and is captured. The first 16 KB per call is kept and shown in +the flow editor's log panel and on the run's per-node record; the rest is +dropped, so a node printing in a loop cannot fill anything up. + +Use it to debug. Do not use it to record results — a number worth keeping is an +output port, not a line of text. + +## Errors + +An exception fails that node's execution, not the flow. The message you see is +one line from the frame in *your* code, not a stack through the engine — that +is a deliberate choice about what is actionable. + +The node keeps its last error visible after it recovers, so a failure that +fired an alert at 03:00 still says what it was at 09:00. It can also be +acknowledged from the canvas. + +## Timeouts + +`timeout` on a node is how many seconds its code may run before it is stopped. +The default is 30, and it covers the *first* call's imports, which can be much +slower than the body — a node importing torch is not being slow, it is loading. + +Above 60 seconds, a live flow may deliver the same work again while the node is +still running. In a batch run, which never redelivers, it is an idle timeout +instead: silence this long is a kill. + +## Running a node somewhere else + +A node declares the label of the machine it needs: + +```json +{"id": "train", "device": "gpu", "device_policy": "require", "timeout": 7200} +``` + +`require` (the default) waits for a worker carrying that label; `prefer` runs +locally when none is attached. A node bound to a device is compiled *on that +machine* — a node importing `torch` is correct on the GPU box and a missing +module on the engine, so checking it here would fail something that is fine. + +See [Remote workers](workers.md). + +## Sharing code between flows + +A node's source can be promoted to the shared library from its panel, and other +flows can then use it by reference. One copy, one place to edit — and every +flow using it runs the edit, which is the point and also the caution. + +Shared sources live in `_lib/` in the flow repository, so they are versioned +with everything else. + +## Packages + +Node code runs in a virtual environment of its own, on the installation's data +volume — deliberately separate from the one Fluksio itself runs on. + +Declare what you import in [Modules](../interface/operations.md), or over the +API: + +```sh +curl -X POST $FLUKSIO/modules/apply -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' \ + -d '{"requirements": "numpy>=2\npandas\n"}' +``` + +It is a pip manifest installed with `uv pip sync`, versioned alongside your +flows. An install takes effect immediately; nothing restarts. + +## A worked example + +The repository ships a small supervised fit as a seedable demo — three nodes, +a batch flow, streaming metrics, artifacts between stages, and a GPU-labelled +node that falls back to the engine when no worker is attached. It is the +shortest complete thing to read: + +```text +prepare ──dataset(artifact)──▶ train ──weights(artifact)──▶ evaluate + │ + └── loss (streaming float) ──▶ chart +``` + +`make seed-demo` builds it against a running stack. + +## See also + +- [Where a node's values come from](../concepts/values.md) +- [Keeping state in a flow](../concepts/state.md) +- [Runs: pipelines that finish](../concepts/runs.md) +- [Node types](../reference/node-types.md) — the ones you do not have to write diff --git a/docs/code/workers.md b/docs/code/workers.md new file mode 100644 index 0000000..3e6f133 --- /dev/null +++ b/docs/code/workers.md @@ -0,0 +1,137 @@ +# Remote workers + +The engine runs where the automations are. The GPU is somewhere else, the +Raspberry Pi with the relays is in a shed, and neither of them is on the same +network as the other. + +A **worker** is a process that runs the code of nodes marked for it. It dials +*out* to the engine over one authenticated websocket, so nothing on that +machine has to be reachable — and nothing has to expose the engine's state +backend across hosts, which it never should. + +## Install and attach + +```sh +pip install fluksio-worker + +fluksio-worker \ + --url wss://api.example.com/api/v1/workers/attach \ + --token "$FLUKSIO_WORKER_TOKEN" \ + --labels gpu,cuda12 \ + --python /opt/torch-venv/bin/python +``` + +`fluksio-worker` is its own distribution — the agent, the node runner, and +`websockets`. Nothing of the engine, so a GPU box does not install a database +driver in order to run a training step. An engine host already has it, and +`fluksio worker …` is the same program. + +| Option | Default | What it does | +|---|---|---| +| `--url` | **required** | `wss://…/api/v1/workers/attach` | +| `--token` | `$FLUKSIO_WORKER_TOKEN` | the credential, minted on the engine | +| `--name` | this host's name | how it shows up in the worker list | +| `--labels` | none | comma-separated; what a node's `device` matches | +| `--python` | this interpreter | the interpreter node code runs on | +| `--parallel` | `1` | how many node calls it will take at once | +| `--artifact-url` | derived from `--url` | where the artifact store is, if not beside the socket | + +`--python` is the important one. It is how this machine keeps its own wheels — +the CUDA build, the vendor SDK, the thing that will not install anywhere else — +without the engine ever installing them or knowing about them. + +## Mint the token + +On the engine, as a superuser: + +```sh +curl -X POST $FLUKSIO/workers/tokens -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' -d '{"name": "gpu-dev"}' +``` + +Shown once, valid for a year — a worker is a machine somebody sets up and +leaves running. It is signed with the same keypair agent tokens use, so +rotating that key revokes every worker along with them. + +??? note "A host where pip is not an option" + + The two files work copied into one directory and run with `python agent.py + …`. The engine serves the runner itself at `GET /api/v1/workers/runtime` — + it is the same module its own local workers run, deliberately standard + library only. + +## Send a node to it + +A node declares the label of the machine it needs: + +```json +{ + "id": "train", + "device": "gpu", + "device_policy": "require", + "timeout": 7200 +} +``` + +| `device_policy` | Behaviour when nothing carrying the label is attached | +|---|---| +| `require` (default) | the run stays `queued` and says what it is waiting for | +| `prefer` | it runs on the engine instead | + +`prefer` is what makes a flow work before the GPU box exists. `require` is what +you want once it does. + +!!! note "Set from the API" + + `device` and `device_policy` are not yet fields in the node panel. Set them + with `PUT /flows/{name}`. + +## What follows from this + +- **The node's source travels with every call.** Nothing has to be deployed to + the worker, and changing a node's code takes effect on the next execution. +- **A node bound to a device is compiled on that machine.** A node importing + `torch` is correct on the GPU box and a missing module on the engine, so + checking it here would fail something that is fine. +- **`import fluksio` inside a node is the worker's own reporter.** `emit`, + `save_artifact`, `load_artifact` — installed before your code runs, so an + installed `fluksio` package on that box never shadows it. +- **Cancelling a run kills what it is executing**, there or here, and leaves + other runs of the same node alone. +- **If the worker disappears mid-call**, the run fails in seconds with + `worker went away mid-call` rather than waiting out its timeout. +- **A worker sends a heartbeat while it executes**, so a long node is + distinguishable from a dead socket. Ninety seconds of silence is gone. + +## Artifacts across machines + +An artifact reference names content by its hash, not a location, so it stays +valid wherever the store is reachable from. A worker that shares the engine's +filesystem writes to it directly; one that does not fetches and uploads over +HTTP, using the artifact endpoint beside the socket it already has. Either way +your node code is the same two calls. + +## Seeing what is attached + +```sh +curl -s $FLUKSIO/workers -H "Authorization: Bearer $TOKEN" | jq +``` + +Name, labels, how many calls it will take at once, how many are in flight, when +it attached, when it was last seen, its Python version, and a digest of its +environment. + +## What a worker is not + +It is not a second engine. Subscriptions, schedules, webhooks, the dashboards +and the run queue all stay in one process — that is what keeps a message having +one definition and a cron tick happening once. A worker executes node bodies. + +Running two engines against one data directory is not supported. Distribute +work with workers. + +## See also + +- [Runs: pipelines that finish](../concepts/runs.md#running-a-node-somewhere-else) +- [Writing node code](nodes.md) +- [Getting started: data science](../getting-started/data-science.md) diff --git a/docs/concepts/flows.md b/docs/concepts/flows.md new file mode 100644 index 0000000..518654d --- /dev/null +++ b/docs/concepts/flows.md @@ -0,0 +1,183 @@ +# Flows, nodes and messages + +Three ideas hold the whole system up. They are worth twenty minutes, because +almost everything else follows from them. + +## A flow is a graph you did not draw + +A **flow** is a set of **nodes**. Each node declares the messages it needs +(`requires`) and the messages it produces (`provides`). The graph is whatever +those declarations imply: + +```python +# node "read" +def process(): + return {"temperature": read_sensor()} + +# node "decide" +def process(temperature, setpoint=21.0): + return {"heat": temperature < setpoint} +``` + +`decide` is downstream of `read` because it needs `temperature` and `read` +produces it. Nobody drew a wire. + +This is the one structural decision everything else rests on, so it is worth +being explicit about the consequences: + +- **Fan-in is free.** Two nodes providing `temperature` are two producers of + one message. The consumer does not change. +- **A wire cannot be wrong.** There is no wire. There is a name that either + matches or does not, and the canvas tells you at edit time which it is. +- **Layout is not a document.** The canvas computes the arrangement, so a flow + has no stored positions to maintain, merge or fight over. +- **Flows stay small.** A graph nobody can hand-arrange is one worth keeping + small — which is the intent. Several atomic flows that name each other beat + one flow with sixty nodes in it. + +### Message names are namespaced + +Inside flow `house`, a message named `temperature` is really +`house.temperature`. A bare name is qualified with its own flow; a dotted name +is used as written. That is how two flows share a value: + +```python +# in flow `dashboard`, reading a message that flow `house` produces +def process(house_temperature): # port bound to "house.temperature" + ... +``` + +The canvas draws messages arriving from another flow as labelled endpoints, so +you can see where they come from without opening the other flow. + +## A node is a function with declared ports + +Most nodes are **Function** nodes: a Python file defining `process(...)`. Its +arguments are its input ports by name; its return value is a dict keyed by +output ports. + +```python +def process(reading, unit="C"): + return {"shown": reading if unit == "C" else reading * 1.8 + 32} +``` + +`reading` is a port. `unit` is a **setting** — a constant of this node's code, +typed into its panel and stored with the flow. Both arrive as arguments, which +is why a setting may not share a name with a port. See +[Where a node's values come from](values.md). + +The rest of the node types are the ones that would be tedious or unsafe to +write yourself: MQTT, HTTP, InfluxDB, schedules, switches, notifications. Each +one is configured by filling in a form the editor generates from its parameter +schema, so they all behave the same way. The full list is in +[Node types](../reference/node-types.md). + +### Ports are typed + +A port declares a `dtype`: `float`, `int`, `str`, `bool`, `json`, `record`, +`list`, `series` or `artifact`. Every value that passes through is checked +against it. + +Types are not decoration. They are what lets the dashboard editor offer you +only the messages a gauge can actually draw, and what lets the canvas refuse a +binding before anything runs. See [Payload types](../reference/payload-types.md). + +Everything on the wire is JSON. Bytes — a checkpoint, an image, a model — +travel as an `artifact`: the bytes go to a content-addressed store and the +message carries a small reference to them. + +### Nodes are pure + +A node is called with the values of the messages it declares and returns the +values of the messages it provides. There is no context object, no global +store, no handle to reach for. + +That is deliberate: a node with hidden state cannot run twice in parallel, +cannot be replayed, and cannot be moved to another machine. Plenty of real +automations do need to remember something, and there is a specific way to say +so — see [Keeping state in a flow](state.md). + +## Two shapes of flow + +Set `mode` on the flow: + +| | `live` (default) | `batch` | +|---|---|---| +| Runs | continuously | once per run, on request | +| Started by | subscriptions, schedules, webhooks | `POST /runs/flows/{name}` | +| Ends | never | when the graph drains | +| Keeps | the last value of each message | a run record: params, result, metrics, artifacts | +| Is | a thermostat, an ETL job on a cron | an experiment, a CI-style job | + +A batch flow is built and validated like any other, appears on the same canvas +and is type-checked the same way. It is simply never *activated*: no +subscriptions, no schedules, no webhooks. See +[Runs: pipelines that finish](runs.md). + +## Editing is separate from running + +Every flow has a published version and, while you are working, a draft. + +- **Saving** writes the draft. The engine keeps running the published version. +- **Publishing** promotes the draft. The engine reloads and picks it up. +- **Discarding** throws the draft away. + +The store is a git repository — `flow.json` for the structure, `nodes/*.py` for +the code — and each save is a commit. So a flow's history is readable with +ordinary git tooling, and copying a flow between installations is copying a +directory. + +Saving carries the version you last saw. If someone else saved in between, you +get a 409 instead of quietly overwriting their work. + +## What can be wrong, and when you find out + +The canvas validates continuously and names problems on the nodes they belong +to: + +| Issue | What it means | +|---|---| +| `unconnected_input` | a port needs a message nothing in reach provides | +| `missing_initial_value` | the message exists but has never held a value, and nothing will give it one | +| `cycle` | A waits for B and B waits for A — nothing could ever start | +| `self_loop_needs_initial` | a node reads a message it also writes, with no starting value | +| `node_error` | the node's code did not load: a syntax error, a missing import | +| `unauthenticated_hook` | advisory — a webhook with no shared secret is open to anyone | + +A flow with any of these except the advisory one does not run. The health +summary on Home counts them, so "why is nothing happening?" has an answer that +does not involve reading logs. + +## What happens at runtime + +- **A flow can be started and stopped.** Stopped means its subscriptions and + schedules are torn down. +- **A flow can be paused and stepped.** Paused holds messages instead of + running them; step releases exactly one. This is how you test something + before it moves a relay. +- **A failing node does not take the flow down.** It reports an error, keeps + its last error visible after it recovers, and can fire an alert. +- **A flow whose background tasks keep crashing is quarantined.** The engine + stops restarting them and says so, rather than spinning. Publishing a change + gives it another chance. + +## Values that arrive from outside + +Some messages are not computed by any node: a dashboard control writes them, +the API publishes them, a batch run passes them in. Declare those as the flow's +**inputs**, with the value they start from: + +```json +{"inputs": [{"spec": {"name": "setpoint", "dtype": "float"}, "initial": 21.0}]} +``` + +Without that, the node reading `setpoint` waits for something nothing provides, +and the canvas says so. With it, the flow starts at 21.0 and whatever writes +the message afterwards takes over. + +## Where to next + +- [Where a node's values come from](values.md) — ports, settings, flow inputs +- [Keeping state in a flow](state.md) — the sanctioned way to remember something +- [Runs: pipelines that finish](runs.md) — the batch half of the engine +- [Writing node code](../code/nodes.md) — the practical side of `process()` diff --git a/docs/concepts/runs.md b/docs/concepts/runs.md new file mode 100644 index 0000000..d740048 --- /dev/null +++ b/docs/concepts/runs.md @@ -0,0 +1,271 @@ +# Runs: pipelines that finish + +A flow that automates a house never ends — a value arrives, nodes fire, and it +waits for the next one. A research pipeline is the other shape: parameters go +in, stages execute in order, and at some point it is *done* and has produced +something worth keeping. Fluksio calls the second one a **run**, and it is the +same engine either way. + +This is what makes Fluksio usable where Kedro, MLflow or ClearML would be: a +run has parameters that identify it, a result, per-step metrics, artifacts and +a place in a queryable history — without a second server, and without paying a +project bootstrap on every execution. + +## A batch flow + +Set `mode: "batch"` on the flow and name the messages its result should hold: + +```json +{ + "name": "train_polymer_gnn", + "mode": "batch", + "outputs": ["final_loss", "report"], + "inputs": [ + {"spec": {"name": "lr", "dtype": "float"}, "initial": 0.1}, + {"spec": {"name": "steps", "dtype": "int"}, "initial": 3000} + ], + "nodes": [{"id": "train", "timeout": 7200, "device": "gpu", "...": "..."}] +} +``` + +A batch flow is built and validated like any other — it appears on the canvas, +its ports are type-checked — but it is never *activated*: no subscriptions, no +schedules, no webhooks. It runs when a run asks it to, and not otherwise. + +Its `inputs` are its parameters. A run supplies values for them; anything it +does not supply keeps the declared initial value. + +One thing a batch flow may not do is rate-limit a port (`interval`). A rate +limit holds a value back for a timer to release, and a run has no timer — the +value would be dropped rather than delayed, so submitting is refused instead. + +## Submitting + +```bash +curl -X POST $FLUKSIO/runs/flows/train_polymer_gnn \ + -H "Authorization: Bearer $TOKEN" -H 'Content-Type: application/json' \ + -d '{"params": {"lr": 0.3, "steps": 4000}, "seed": 7}' +``` + +The answer is immediate and the run is `queued`; a training run is measured in +hours, so nothing waits for it. Poll `GET /api/v1/runs/{id}` for its status, +result, per-node record and artifacts. + +Wrong parameters are refused before anything executes — an undeclared name, or +a value of the wrong type, comes back as a 422 naming the problem. + +### Sweeps + +An ensemble is the same parameters at different seeds; a grid search is the +parameters spread out. Both are one call, and the caller builds the list: + +```bash +curl -X POST $FLUKSIO/runs/flows/train_polymer_gnn/sweep \ + -H "Authorization: Bearer $TOKEN" -H 'Content-Type: application/json' -d '{ + "runs": [{"params": {"lr": 0.1}, "seed": 1}, {"params": {"lr": 0.3}, "seed": 1}] +}' +``` + +They share a `group_id`, so `GET /api/v1/runs?group=…` is the sweep, and they +execute in parallel. That is safe because **each run has a state backend of its +own**: message names are global keys, so two runs of one flow would otherwise +overwrite each other's values. They do not. + +## Producing values before you are finished + +A training loop has numbers worth keeping long before it has a result. Those +numbers are **outputs**, not logs: a node declares a port for them and produces +them over time, which in Python is a generator. + +```python +def process(lr, steps): + loss = 1.0 + for _ in range(steps): + loss = train_one_step(lr) + yield {"loss": loss} # published now, on the `loss` port + return { + "weights": fluksio.save_artifact(dump(model), "weights.npz"), + "final_loss": loss, + } +``` + +Mark the port it streams on, so the flow says what it does: + +```json +{"name": "loss", "dtype": "float", "stream": true} +``` + +Every yield is published the instant it happens — same port, same type check, +same place on the canvas as any other value. Whatever the generator `return`s +is the node's result, and is what downstream nodes read. If you never +`return`, the last thing you yield is the result instead. + +This is the whole reason the framework does not have a logging API. A metric +that escapes through `log_metric()` is undeclared: invisible to validation, +absent from the canvas, and stored somewhere the graph knows nothing about. +A metric that leaves through a port is a message — so a chart binds to it +directly, a downstream node can consume it, and the run keeps its series +without anyone asking. + +**Where a yield cannot reach** — the value comes from inside somebody else's +callback, and they call you rather than the other way round — `fluksio.emit` +writes the same ports the same way: + +```python +import fluksio + + +def process(): + model.fit(callbacks=[LambdaCallback( + on_epoch_end=lambda epoch, logs: fluksio.emit(loss=logs["loss"]) + )]) + return {"weights": ...} +``` + +### What a run does with them + +Every number a node emits is kept as the run's series, stepped by the count of +emissions on that message. Read one back with +`GET /api/v1/runs/{id}/metrics?name=.loss`, or compare runs: + +``` +GET /api/v1/runs/series/compare?ids=,,&metric=.loss +``` + +That answers in the `series` shape a chart widget already draws, so three +training curves side by side is a widget binding. During a run the values also +arrive live on the flow socket, so a chart bound to the port fills in as the +training goes. + +A streaming port may set `interval` to thin out what reaches the canvas — the +run's history still keeps every value, because the interval is asking for the +display not to be flooded, not for the curve to have holes in it. + +Emitting has a second effect: **a node's timeout measures silence, not +duration**. A node that yields every few seconds can run for hours under a +`timeout` of 300; one that says nothing for longer than its timeout is killed. +Set `timeout` on a long node to how long it may plausibly go quiet. + +In a *live* flow, an emission also wakes whatever is downstream of it, exactly +as a subscriber publishing does. In a run it does not: a run's graph is +scheduled once, and three thousand mid-node cascades would leave "the run has +finished" with nothing to mean. + +## Artifacts + +Bytes never travel as a message. `save_artifact` writes them to a +content-addressed store and returns a small reference — digest, size, media +type, name — which is what an `artifact`-typed port carries: + +```python +def process(weights): # requires: weights, dtype "artifact" + path = fluksio.load_artifact(weights) + ... +``` + +Because the address is the content's hash, a sweep whose fifty configs share +one preprocessed input stores it once, and a reference stays valid wherever +the store is reachable from. Artifacts a run produced are listed on it and +downloadable at `GET /api/v1/artifacts/{digest}`. + +## Objects that cannot be serialized + +A live model, a `DataLoader`, a JAX-compiled function — these do not cross a +node boundary, and no framework flag will make them. There are exactly two +patterns, and they are both deliberate: + +* **Keep them in one node.** Stages that must share live memory are one node. + Building the model and training it is one stage; the fact that Kedro would + make them two nodes is Kedro's problem, not a structure worth reproducing. +* **Cross at a checkpoint.** Save what matters as an artifact and rebuild from + it on the other side. That is the boundary that also survives the next node + running on a different machine. + +## Running a node somewhere else + +A node that needs a GPU declares the label of the machine that has one: + +```json +{"id": "train", "device": "gpu", "device_policy": "require", "timeout": 7200} +``` + +A worker on that machine dials **out** to the engine, because the engine +generally cannot reach it — different network, no inbound route — and because +nothing should expose Redis across hosts. Install it on the box, mint it a +token, and start it: + +```bash +curl -X POST $FLUKSIO/workers/tokens -d '{"name": "gpu-dev"}' # once, as an admin + +pip install fluksio-worker + +fluksio-worker \ + --url wss://api.example.com/api/v1/workers/attach \ + --token "$FLUKSIO_WORKER_TOKEN" \ + --labels gpu,cuda12 \ + --python /opt/torch-venv/bin/python +``` + +`fluksio-worker` is its own distribution — the agent, the node runner, and +`websockets`. Nothing of the engine, so a GPU box does not install a database +driver to run a training step. Where pip is not an option, the two files still +work copied into one directory and run with `python agent.py …`; the engine +serves the runner at `GET /api/v1/workers/runtime`. + +`--python` is the interpreter node code runs on, which is how the GPU box keeps +its CUDA wheels without the engine ever installing them. The node's source +travels with every call, so nothing has to be deployed there. + +A few consequences worth knowing: + +* `import fluksio` inside a node is the worker's own reporter — `emit`, + `save_artifact`, `load_artifact` — installed before the node's code runs, so + the installed `fluksio` package (if the box has one) never shadows it. +* A node bound to a device is **compiled on that machine**. A node importing + `torch` is correct on the GPU box and a missing module on the engine, so + checking it on the engine would fail a node that is fine. +* If nothing carrying the label is attached, the run stays `queued` and says + what it is waiting for. Submit first, switch the GPU box on later. +* Cancelling a run kills what it is executing, there or here, and leaves other + runs of the same node alone. +* If the worker disappears mid-call, the run fails in seconds with + `worker went away mid-call` rather than waiting out its timeout. +* `device_policy: "prefer"` runs locally when no such worker is attached; + `"require"` (the default) waits for one. + +## Durability + +Submitting journals the run to a Redis stream of its own, separate from the +one the automations use — a burst of five hundred sweep runs must not stand +between a house and its heating. An engine that is down when a run is +submitted picks it up when it starts. + +From the moment a run is claimed, its database row is the record and the queue +is finished with it. Redelivering two hours of training because an +acknowledgement was late is not recovery; instead a running run refreshes a +lease, and one whose lease goes stale is marked `abandoned` — which is what a +run whose engine was killed mid-training becomes. + +## What this costs, compared + +The repository ships a benchmark that measures submitting a run against a +Kedro project doing the same nothing: + +``` +fluksio — submit accepted median 15.3 ms + submit -> result median 60.8 ms +kedro — kedro run median 1109.5 ms +``` + +The difference is not the orchestration; it is that Fluksio does not boot a +project per run. The engine is already up, and the workers already have the +node's code compiled. On a 510-run sweep, that gap is about nine minutes of +pure startup that never happens. + +## See also + +- [Getting started: data science](../getting-started/data-science.md) — the same + material as a first setup +- [Writing node code](../code/nodes.md) — generators, `fluksio.emit`, artifacts +- [Remote workers](../code/workers.md) — attaching the machine with the GPU +- [The HTTP API](../code/api.md) — every endpoint used above diff --git a/docs/concepts/state.md b/docs/concepts/state.md new file mode 100644 index 0000000..c41df0d --- /dev/null +++ b/docs/concepts/state.md @@ -0,0 +1,75 @@ +# Keeping state in a flow + +Logic nodes are pure functions of their inputs. There is no `context`, no +`global`, no handle to a store — a node is called with the values of the +messages it declares and returns the values of the messages it provides. + +That is deliberate: a node with hidden state cannot be run twice in parallel, +cannot be replayed, and cannot be moved to another worker. But plenty of real +automations need to remember something. This is how. + +## State is a message the node both reads and writes + +A running total, a debounce timer, a last-seen reading — each is a value that +survives between runs. Give it a message name, declare it as both an input and +an output, and it is state: + +```python +def process(reading, total=0.0): + return {"total": total + reading} +``` + +with `reading` and `total` as inputs and `total` as the output. Each run reads +the total the previous run wrote. + +Two things make this work rather than loop forever: + +- **A node never depends on itself.** The graph is built from message names, but + a node providing a message it also requires is not placed downstream of + itself. Publishing `total` does not re-run the node that wrote it. +- **The value must start somewhere.** The first run has nothing to read. Declare + the message as a flow input with a starting value; otherwise the flow reports + `self_loop_needs_initial` at edit time and refuses to publish. + +The port needs a default in the function signature (`total=0.0` above) so the +first call is well-formed even if the value is missing. + +## Feeding a value back between two nodes + +Sometimes the value comes from a different node — a controller reading back what +an estimator computed from its own last output. Written plainly that is a cycle, +and the validator rejects it, because a graph where A waits for B and B waits +for A can never start. + +What is actually meant is that the fed-back value is *state*, not a trigger. +Say so, by marking the input non-triggering: + +```json +{"name": "estimate", "port": "estimate", "dtype": "float", "trigger": false} +``` + +A non-triggering input: + +- creates no dependency, so it cannot form a cycle; +- never makes the node wait — if the message has no value yet, the port is + simply left out of the call, and the function's default applies; +- is read fresh from state whenever the node does run, for whatever reason. + +Use it for the back edge only. An input that should wake the node is an ordinary +input, and marking it non-triggering would mean the node never runs at all. + +## What still holds engine-side state + +Built-in nodes that are *about* time or change — rate limiting, filter-on-change, +delay, cron — keep what they need in the engine's own state backend, under keys +that never appear as messages. That is the engine's business, not a flow's: the +line is that node *code* you write never reaches for state, while node *types* +the engine ships may. + +Storage and direct-I/O nodes are the other sanctioned exception, since talking +to a device or a database is stateful by nature. + +## See also + +- [Where a node's values come from](values.md) — ports, settings and flow inputs +- [Node types](../reference/node-types.md) — the built-ins that do keep engine-side state diff --git a/docs/concepts/values.md b/docs/concepts/values.md new file mode 100644 index 0000000..69f5325 --- /dev/null +++ b/docs/concepts/values.md @@ -0,0 +1,67 @@ +# Where a node's values come from + +A node function is called with one argument per thing it needs, and every one +of them is declared somewhere you can see. There are two kinds. + +## Ports: what the graph carries + +A port binds to a message name. Whatever last published that message is the +value the node is called with, and the wiring follows from the name rather than +from a line somebody drew. + +```python +def process(reading, setpoint): + return {"heat": reading < setpoint} +``` + +with `reading` and `setpoint` as inputs and `heat` as an output. + +## Settings: constants of this node + +A setting is a value that belongs to this node's code rather than to the graph: +how many retries, which unit, how long to wait. It is typed into the node's +**Settings** section, stored with the flow, and arrives as an argument like a +port: + +```python +def process(reading, unit="C"): + return {"shown": reading if unit == "C" else reading * 1.8 + 32} +``` + +Because both arrive by name, a setting cannot share a name with a port — the +node reports it rather than picking one. + +Settings are part of the flow document, so changing one is an edit that gets +published, not something that happens at runtime. **A value that should change +while the flow runs is a message, not a setting.** + +There is no `global` and no `flow` context. A constant several nodes need is a +node that provides it: one place it is set, an ordinary message out, and every +consumer visibly downstream of it. + +## Flow inputs: what arrives from outside + +Some messages are not computed by any node in the flow — a dashboard control +writes them, the API publishes them, a batch run passes them in. Declare those +as the flow's **inputs**, with the value they start from: + +```json +{"inputs": [{"spec": {"name": "setpoint", "dtype": "float"}, "initial": 21.0}]} +``` + +Without that, the node reading `setpoint` waits for something nothing provides, +and the canvas says so. With it, the flow starts from 21.0 and whatever writes +the message afterwards takes over. + +The canvas draws each one as a labelled endpoint feeding the nodes that read +it, the same way it draws a dashboard tile or another flow — so a value never +appears from nowhere. They are edited in the flow's own panel. + +A batch flow's inputs are also its run parameters, and its **outputs** name what +a run reports as its result; see [Runs](runs.md). + +## See also + +- [Flows, nodes and messages](flows.md) — how the graph is built from the names +- [Keeping state in a flow](state.md) — a value a node reads *and* writes +- [Writing node code](../code/nodes.md) — the practical side diff --git a/docs/getting-started/data-science.md b/docs/getting-started/data-science.md new file mode 100644 index 0000000..155ee0e --- /dev/null +++ b/docs/getting-started/data-science.md @@ -0,0 +1,323 @@ +# Getting started: data science + +You have a training script. It works. What you do not have is any reliable +answer to "what was the learning rate on the run that got 94%?", and the +`results_final_v3_ACTUAL.csv` in your home directory is not helping. + +This page adds Fluksio to what you already have. It takes about five minutes, +installs one Python package, and does not ask you to restructure anything. + +## Install + +```sh +pip install fluksio +fluksio serve +``` + +That is the whole installation. No Docker, no database server, no ports to +open. The first run prints something like: + +```text +Created the admin account admin@example.com + password: k3Qm-8vTpLdX + Shown once. Change it from the dashboard. +Fluksio 0.1.0 — data in /home/you/.fluksio + API http://127.0.0.1:8000/api/v1 + No portal. Pair this installation with: + fluksio enroll --portal https://hub.example.com +``` + +**Write that password down.** It is shown once and it is how you authenticate +from here on. + +Everything the installation owns lives in `~/.fluksio`: a SQLite database, a +git repository holding your flows, the artifact store, and a virtual +environment your node code runs in. Move it with `--data-dir`, which is worth +doing on a cluster where `$HOME` is a network filesystem — SQLite's +write-ahead log does not work on NFS, and `fluksio serve` warns you when it +notices. + +!!! tip "Keep it running" + + The engine is meant to be resident. That is the whole reason submitting a + run costs about 15 ms instead of the second a project-bootstrapping + orchestrator spends before it does anything. Leave it in a `tmux` window, + or write a small `systemd --user` unit for it. + +## Get a token + +Everything below is the HTTP API. Grab a token once: + +```sh +export FLUKSIO=http://127.0.0.1:8000/api/v1 +export TOKEN=$(curl -s -X POST $FLUKSIO/login/access-token \ + -d "username=admin@example.com&password=k3Qm-8vTpLdX" | jq -r .access_token) +``` + +While you are experimenting, the interactive schema at + is the fastest way to see what is available. + +## Tell it about your packages + +Node code runs in `~/.fluksio/user-venv`, deliberately separate from the +environment Fluksio itself is installed in — so a pin of yours can never +collide with one of ours. That venv starts empty, so the first thing to do is +say what your script imports: + +```sh +curl -X POST $FLUKSIO/modules/apply -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' \ + -d "{\"requirements\": $(jq -Rs . < requirements.txt)}" +``` + +It is a pip manifest, installed with `uv pip sync`, and it is versioned +alongside your flows — so what a run imported is recorded with what it ran. +Adding a package takes effect immediately; nothing restarts. + +??? note "Already have a venv you would rather not duplicate?" + + Attach it as a worker instead of reinstalling into it. Mint a token, then + point the agent at your existing interpreter: + + ```sh + curl -X POST $FLUKSIO/workers/tokens -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' -d '{"name": "laptop"}' + + fluksio worker --url ws://127.0.0.1:8000/api/v1/workers/attach \ + --token "$WORKER_TOKEN" --labels local --python "$(which python)" + ``` + + Then mark the node `"device": "local"` and it runs on that interpreter. It + is the same mechanism that sends a node to a GPU box, and it is worth + knowing about early — see [Remote workers](../code/workers.md). + +## Wrap your training script + +A **flow** is a graph of nodes. A **batch flow** is one that runs on demand +from parameters to a result, which is what an experiment is. Your existing +script becomes the body of a node. + +Say your script looks roughly like this: + +```python +def train(lr, epochs): + model = build_model() + for epoch in range(epochs): + loss = step(model, lr) + print(f"epoch {epoch}: {loss}") # ← this is what you are losing + torch.save(model.state_dict(), "weights.pt") + return loss +``` + +Two changes turn it into a node: + +```python +"""Fit the model. A generator, so numbers escape while it is still running.""" + +import fluksio + + +def process(lr, epochs): + model = build_model() + for epoch in range(int(epochs)): + loss = step(model, lr) + yield {"loss": loss} # ← published now, on the loss port + torch.save(model.state_dict(), "weights.pt") + return { + "weights": fluksio.save_artifact("weights.pt"), + "final_loss": loss, + } +``` + +`yield` replaces `print`. Each one publishes on the node's `loss` port the +instant it happens, and the run keeps every value as a series — which is why +Fluksio has no `log_metric()` call. A metric that leaves through a port is a +message like any other: a chart can bind to it, a downstream node can consume +it, and it shows up on the canvas. A metric that escapes through a logging +function is invisible to all three. + +`fluksio.save_artifact` handles the things too big to be messages — a +checkpoint, a dataset, a plot. It stores the bytes by their hash and returns a +small reference. Nothing changes about how you write the file. + +!!! note "Where a `yield` cannot reach" + + If the number comes from inside somebody else's callback — Keras, Lightning, + HuggingFace `Trainer` — `fluksio.emit(loss=...)` writes the same port the + same way: + + ```python + model.fit(callbacks=[LambdaCallback( + on_epoch_end=lambda epoch, logs: fluksio.emit(loss=logs["loss"]) + )]) + ``` + +## Create the flow + +There is no scaffolding command yet, so a flow is created by PUTting its +definition. That is a fifteen-line script you run once: + +```python +"""Create the `train` flow. Run once; edit it in the canvas afterwards.""" + +import httpx + +API = "http://127.0.0.1:8000/api/v1" +api = httpx.Client(base_url=API, timeout=60) +token = api.post( + "/login/access-token", + data={"username": "admin@example.com", "password": "k3Qm-8vTpLdX"}, +).json()["access_token"] +api.headers["Authorization"] = f"Bearer {token}" + +api.put("/flows/train", json={ + "name": "train", + "title": "Model training", + # Batch: nothing is activated, nothing fires until a run asks. + "mode": "batch", + # Its inputs are the run's parameters, with the values a run gets when it + # names none. + "inputs": [ + {"spec": {"name": "lr", "dtype": "float"}, "initial": 0.01}, + {"spec": {"name": "epochs", "dtype": "int"}, "initial": 50}, + ], + # What a run reports as its result. + "outputs": ["final_loss", "weights"], + "nodes": [{ + "id": "train", + "type": "python", + "title": "Fit the model", + # An *idle* timeout once the node streams: this is how long it may go + # quiet, not how long it may run. + "timeout": 600, + "requires": [ + {"name": "lr", "dtype": "float"}, + {"name": "epochs", "dtype": "int"}, + ], + "provides": [ + # `stream` says this port publishes repeatedly during one execution. + {"name": "loss", "dtype": "float", "stream": True}, + {"name": "final_loss", "dtype": "float"}, + {"name": "weights", "dtype": "artifact"}, + ], + }], +}).raise_for_status() + +api.put("/flows/train/nodes/train/source", + json={"code": open("train_node.py").read()}).raise_for_status() + +version = api.get("/flows/train", params={"draft": True}).json()["definition"]["version"] +api.post("/flows/train/publish", json={"version": version}).raise_for_status() +print("published") +``` + +Two things worth noticing. Ports are declared, not inferred — `process(lr, +epochs)` gets its arguments from the ports of the same name, and the types are +checked on every value. And saving writes a *draft*; `publish` is what the +engine picks up. That separation is what lets you edit a flow that is running. + +## Run it + +```sh +curl -X POST $FLUKSIO/runs/flows/train -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' \ + -d '{"params": {"lr": 0.003, "epochs": 200}, "seed": 7}' +``` + +It answers immediately with a queued run — training is measured in hours, so +nothing waits for it. A parameter you did not declare, or one of the wrong +type, is refused with a 422 before anything executes. + +Then, whenever you like: + +```sh +curl -s $FLUKSIO/runs/ -H "Authorization: Bearer $TOKEN" | jq +curl -s "$FLUKSIO/runs//metrics?name=train.loss" -H "Authorization: Bearer $TOKEN" | jq +``` + +The run carries its parameters, a digest of them, the seed, its result, how +long each node took, what it logged, and every artifact it produced. That is +the answer to "what was the learning rate on the run that got 94%?". + +## Sweep it + +A grid search and an ensemble are the same call — you build the list, Fluksio +runs them in parallel: + +```sh +curl -X POST $FLUKSIO/runs/flows/train/sweep -H "Authorization: Bearer $TOKEN" \ + -H 'Content-Type: application/json' -d '{"runs": [ + {"params": {"lr": 0.001}, "seed": 1}, + {"params": {"lr": 0.003}, "seed": 1}, + {"params": {"lr": 0.010}, "seed": 1} + ]}' +``` + +They share a `group_id`, so the sweep is `GET /runs?group=…`. Running them +concurrently is safe because each run gets a state backend of its own — two +runs of one flow cannot overwrite each other's values. + +Compare the curves in one call: + +```sh +curl -s "$FLUKSIO/runs/series/compare?ids=$A,$B,$C&metric=train.loss" \ + -H "Authorization: Bearer $TOKEN" +``` + +which answers in exactly the shape a chart widget draws. + +## Small scripts you are just playing with + +The same machinery, minus the ceremony. If what you want is "keep a record of +every version of this thing I keep tweaking", note that **your flows are +already a git repository**: + +```sh +cd ~/.fluksio/flows +git log --oneline +``` + +Every save is a commit — the node source, the parameters, the graph. A run +records the commit it ran at, so `git show` on that hash is literally the code +that produced the number. You get the versioning without adopting anything. + +For quick iteration, keep the flow small (one node is fine), keep the engine +running, and submit from wherever you are working: + +```python +import httpx +run = httpx.post(f"{API}/runs/flows/train", json={"params": {"lr": lr}}, + headers=auth).json() +``` + +A submit is around 15 ms, so calling that in a loop is a reasonable thing to do. + +## Get a browser onto it + +The pip install gives you the engine and the API, not a web interface — a +machine with no inbound route cannot serve one usefully anyway. To see the +canvas, the run history and live loss curves, pair the installation with a +portal, which serves the dashboard from its side: + +```sh +fluksio enroll --portal https://hub.fluksio.com +``` + +Get the claim code from the portal under **Installations → Add installation**. +Nothing needs to be exposed: your machine dials out and holds the connection +open. See [Accounts and the portal](../interface/portal.md). + +If you would rather stay entirely offline, run the dashboard SPA yourself from +the app's Docker image — see the [facility path](facility-automation.md), which +is the same stack. + +## Where to go next + +- [Runs: pipelines that finish](../concepts/runs.md) — the full picture: + artifacts, sweeps, durability, what happens when your engine dies mid-training +- [Writing node code](../code/nodes.md) — generators, settings, what a node may + and may not do +- [Remote workers](../code/workers.md) — send the training node to the GPU box + and keep the rest on your laptop +- [The flow editor](../interface/flow-editor.md) — once you have a portal, this + is where the graph gets easier to change than the script did diff --git a/docs/getting-started/facility-automation.md b/docs/getting-started/facility-automation.md new file mode 100644 index 0000000..7801fef --- /dev/null +++ b/docs/getting-started/facility-automation.md @@ -0,0 +1,272 @@ +# Getting started: facility automation + +You have a box in a cupboard, a handful of sensors that already publish +somewhere, and an ambition to make the house do something about them. This +page brings up a Fluksio instance you can leave running for years, then wires +the first sensor through to a dashboard. + +Budget an afternoon. Most of it is the browser, which is the point: you will be +in this interface a lot, so it is worth learning it properly. + +## What you are standing up + +```text + sensors ──MQTT──▶ ┌──────────────┐ ──▶ InfluxDB (history) + │ Fluksio │ + HTTP / webhooks ──▶│ flow engine │ ──▶ dashboards (what you look at) + │ │ + schedules ────────▶└──────────────┘ ──▶ MQTT / HTTP (what you control) + │ + └── workers on other boxes (optional) +``` + +One engine holds every flow. Nothing here is a plugin you install separately — +the broker client, the time-series writer, the dashboards and the alerting are +all part of the same process, editing the same graph. + +## Prerequisites + +- Docker and Compose v2 on the host +- A hostname you can point at it. `fluksio.local`, a subdomain, or just + `localhost` if you only ever reach it from that machine +- Optionally: an MQTT broker and an InfluxDB you already run. If not, the stack + can start both for you + +## Bring up the stack + +Clone the app repository and start it: + +```sh +git clone https://git.stroblme.de/Fluksio/app.git ~/fluksio +cd ~/fluksio +cp .env.example .env +$EDITOR .env # DOMAIN, FIRST_SUPERUSER, ENVIRONMENT=production +make up +``` + +`.env` is the whole configuration. The four settings that matter on day one: + +| Setting | What it does | +|---|---| +| `DOMAIN` | the hostname everything is served under; the SPA lands on `app.${DOMAIN}` and the API on `api.${DOMAIN}` | +| `FIRST_SUPERUSER` | the account you sign in with | +| `FIRST_SUPERUSER_PASSWORD` | leave it as `changethis` and one is generated for you | +| `ENVIRONMENT` | `production` closes the interactive API schema; `local` leaves it open | + +Everything the installation owns — the database, your flows, secrets, +artifacts, the packages your node code imports — is on one Docker volume. +Backing that volume up is backing up the installation. + +!!! tip "Reverse proxy" + + The stack emits Traefik labels and ships a Traefik you can bring up + alongside it (`docker/compose.traefik.yml`). If you already run Nginx + Proxy Manager or Caddy, attach it to the `proxy` network instead and + forward `app.${DOMAIN}` → `fluksio-app:80` and `api.${DOMAIN}` → + `fluksio-api:8000`. + +??? note "Even smaller: no Docker at all" + + `pip install fluksio && fluksio serve` gives you the same engine with no + containers, keeping its data in `~/.fluksio`. What it does not give you is + the web interface, which the SPA container serves — so you would drive it + from the API, or [pair it with a portal](../interface/portal.md) that + serves the dashboard for you. Good for a Raspberry Pi that only runs flows; + less good as your main instance. + +Open `http://app.${DOMAIN}` and sign in. You should be looking at Home: an +empty brain graph, a health summary, and a flow list with nothing in it. + +## Your first flow + +Go to **Flows → New flow** and call it `house`. You land on the canvas. + +A flow is a set of nodes that talk to each other through **named messages**. +You do not draw wires: a node says which messages it needs and which it +produces, and the canvas draws the graph that follows from those names. That +sounds like a small difference and turns out to be a large one — renaming is +safe, fan-in is free, and two flows can share a value by naming it. + +### Read a sensor + +Press **Add node** (or ⌘K / Ctrl-K, which opens the command palette) and pick +**MQTT**. In its panel on the right: + +- **Broker host** — your broker's hostname, `mosquitto` if you are using the + one the stack can start +- **Topic** — map each output to a topic: + `{"living_temperature": "zigbee2mqtt/living/temperature"}` +- **Provides** — add one output port named `living_temperature`, type `float` + +That is a working node. Press **Publish** (⌘S) and the engine picks it up. + +The canvas now draws your node with a live value on its output as soon as the +broker sends one. Click the wire to see the last payload and its history. + +### Do something with it + +Add a **Function** node. This is a Python node — the code editor opens in its +panel: + +```python +def process(living_temperature, comfortable=21.0): + """Ask for heat when the room is below the comfort point.""" + return {"heat_wanted": living_temperature < comfortable} +``` + +Declare `living_temperature` as an input (type `float`) and `heat_wanted` as an +output (type `bool`). `comfortable` is not a port — it is a **setting**, +because it is a constant of this node rather than something the graph carries. +It shows up as a field in the node's Settings section. + +The canvas now draws MQTT → your function, because the message names line up. +Nothing else was needed. + +### Act on it + +Add a second **MQTT** node, this time with `heat_wanted` as an *input*, and a +topic mapping to whatever your relay listens on. A node with inputs publishes; +a node with outputs subscribes. + +Publish the flow. You have a thermostat. + +!!! warning "Test before it touches a relay" + + A flow can be **paused** (it holds messages instead of running them) and + **stepped** (release exactly one). Together with the run button — which + injects a value by hand — that is how you convince yourself the logic is + right before the contactor finds out. Both live on the dock at the bottom + of the canvas. + +## Store the history + +A live value is enough to control something and useless for answering "was last +February colder?". That is what the **InfluxDB** node is for. + +Add one, and configure it to write the message you already have: + +```json +{ + "url": "http://influxdb:8086", + "token": {"$secret": "influx-token"}, + "org": "home", + "bucket": "sensors", + "writes": { + "living_temperature": { + "measurement": "environment", + "field": "temp_c", + "tags": {"room": "living"} + } + } +} +``` + +Give it `living_temperature` as an input and every value that passes gets a +point. + +Note the token. Credentials never sit in a flow: `{"$secret": "influx-token"}` +is a reference into an encrypted store, and the editor renders those fields as +a secret picker. Add the actual value once under **Secrets**. Flows are a git +repository you might well push somewhere — this is what keeps a password out of +it. + +Reading back is the same node with `queries` instead of `writes`, or — for +anything a chart asks for — a pair of small Python nodes on either side that +build a Flux query and shape its rows. That indirection is deliberate: the +database node holds the connection and nothing else, so a dashboard widget +never learns which database answered it. + +## Put it on a screen + +**Dashboards → New dashboard**, then drag widgets onto the grid and bind each +one to a message. A gauge on `house.living_temperature`, a switch on +`house.heat_wanted`, a chart on the history. + +Widgets are typed the same way ports are: a switch binds to a `bool`, a gauge +to a number, an agenda to a `list`. Bind it wrong and the editor says so rather +than drawing nothing. + +Controls work in the other direction — a switch on a dashboard publishes the +message it is bound to, exactly as a node would. The canvas draws it as a +labelled endpoint feeding the nodes that read it, so a value never appears from +nowhere. + +For a tablet on the wall, see [Dashboards and panels](../interface/dashboards.md): +a panel is a named device, it pairs with a six-character code instead of a +login, and it can only reach the dashboards you gave it. + +## Spread it across machines + +You now have one box doing everything. Two reasons to change that: something +lives on a different network, or something needs hardware the engine's host +does not have. + +The unit of distribution is the **worker**. It runs the code of nodes you mark +for it, and it dials *out* to the engine — so the Pi in the shed does not need +an inbound route, and the engine does not need to reach it. + +On the engine, mint a token: + +```sh +curl -X POST https://api.${DOMAIN}/api/v1/workers/tokens \ + -H "Authorization: Bearer $TOKEN" -d '{"name": "shed-pi"}' +``` + +On the other machine: + +```sh +pip install fluksio-worker +fluksio-worker \ + --url wss://api.${DOMAIN}/api/v1/workers/attach \ + --token "$WORKER_TOKEN" \ + --labels shed,gpio \ + --parallel 2 +``` + +Then mark the node that talks to the shed's GPIO with `device: shed`, and it +runs there. Everything else stays where it is. A node bound to a label no +attached worker carries simply waits rather than failing, so you can write the +flow before the hardware arrives. + +!!! note "What a worker is not" + + It is not a second engine. Subscriptions, schedules, webhooks and the + dashboards all stay in one process — that is what keeps a value having one + definition. A worker executes node bodies, nothing else. Scaling *the + engine* to several processes is not supported: run one. + +## Make it tell you when something breaks + +Under **Alerts**, add a channel and a rule. Channels are ntfy (a push +notification on your phone), SMTP, a webhook, or a message a dashboard +notification widget reads. + +What you probably want on day one is everything, to ntfy: + +- add an ntfy channel with your server and topic +- add a rule with no events ticked, which means *all of them* + +The engine deduplicates aggressively — the same node failing every second is +one alert, not thirty-six thousand — and caps the total at ten an hour however +bad it gets. See [Secrets, modules and alerts](../interface/operations.md). + +## Keep it alive + +- **Back up the data volume.** That is the database, the flows, the secrets and + the artifacts. Everything else is rebuildable. +- **Watch Home.** The health summary names what is wrong — a quarantined flow, a + node that will not load, a stalled queue, a flow that cannot run because its + graph does not validate. +- **Flows are git.** `git log` inside the flow store is the history of every + change anyone made, and reverting one is a revert. + +## Where to go next + +- [Flows, nodes and messages](../concepts/flows.md) — the model, properly +- [Node types](../reference/node-types.md) — everything you can put on a canvas + without writing Python +- [The flow editor](../interface/flow-editor.md) — the canvas, in detail +- [Keeping state in a flow](../concepts/state.md) — running totals, debounces, + and the one rule that makes them safe +- [Accounts and the portal](../interface/portal.md) — reach the installation + from outside the house without opening a port diff --git a/docs/getting-started/index.md b/docs/getting-started/index.md new file mode 100644 index 0000000..118a1d3 --- /dev/null +++ b/docs/getting-started/index.md @@ -0,0 +1,72 @@ +# Pick your starting point + +People arrive at Fluksio from two directions, and the honest answer to "how do +I set this up?" is different for each — not just in the commands, but in how +much of an afternoon it is reasonable to spend. + +Pick the one that sounds like you. Everything past this section is the same for +both. + +
+ +
+ +### Data science + +*"I have a training script. I want to stop losing track of what I ran."* + +One `pip install`, one command, and you are writing Python again. No Docker, no +database, no ports to open. Flows are files, runs are rows, and the metrics are +just the numbers your loop already produces. + +[Set up for experiments →](data-science.md) + +
+ +
+ +### Facility automation + +*"I have a homelab and a pile of sensors. I want them to do something."* + +A stack you bring up once and leave running: the engine, a broker, a +time-series database, dashboards, alerting. Most of the work happens in the +browser, and it is worth doing properly because you will live in it. + +[Set up a homelab instance →](facility-automation.md) + +
+ +
+ +## Not sure? + +Some rough tells: + +| | Data science | Facility automation | +|---|---|---| +| **The flow** | starts, finishes, has a result | never ends | +| **You mostly** | write Python | wire nodes in the browser | +| **Time to first result** | a few minutes | an afternoon | +| **Runs on** | your laptop, or a login node | a box in a cupboard | +| **Data lives in** | SQLite beside the flows | InfluxDB, usually | +| **The thing you look at** | run history and loss curves | a dashboard, maybe on a wall | + +If both describe you — a lab with instruments to drive *and* models to +fit — start with the data-science path. It is the smaller installation, and it +grows into the other one without being reinstalled: the same engine, the same +flows, just more of them running all the time. + +## What is the same either way + +Whichever door you came in: + +- **Flows are files in a git repository.** Every save is a commit. You can read + the history with ordinary git, and you can copy a flow between installations + by copying a directory. +- **Editing is separate from running.** You edit a draft; the engine keeps + running what was published until you publish. +- **Nodes are typed.** A port declares what it carries, and a mismatch is + caught at edit time rather than at three in the morning. +- **Everything the browser does is an API call.** The dashboard is a client of + the same REST API you can script against. diff --git a/docs/index.md b/docs/index.md new file mode 100644 index 0000000..063ba9d --- /dev/null +++ b/docs/index.md @@ -0,0 +1,57 @@ +# Fluksio + +Fluksio is a node-based automation engine. You describe what should happen as a +graph of small pieces of logic, and it keeps that graph running — reacting to +what arrives, or executing once from parameters to a result. + +Two very different jobs turn out to be the same shape, which is why the same +engine does both: + +- **A house, a lab or a plant** produces values forever. A sensor publishes, a + rule fires, a relay closes, a dashboard on the wall shows what happened. The + flow never ends. +- **An experiment** produces a value once. Parameters go in, stages execute, + and at some point it is *done* and has left behind metrics, artifacts and a + record you can compare against last month's. + +The first is a **live flow**, the second is a **run**. Both are the same nodes, +the same type checking, the same editor and the same API. + +## What you actually get + +- A **flow engine** that owns your graph, checks the types on every edge, and + keeps running when a node fails rather than taking the rest down with it. +- A **canvas** that lays flows out for you and an editor for the Python inside + each node — with the running values drawn on the wires while you work. +- **Dashboards** built next to the logic that feeds them, including ones you + can hang on a wall tablet that has no keyboard. +- **Runs**: parameters, metrics, artifacts, sweeps, and a queryable history — + without a second server and without paying a project bootstrap per execution. +- **Distributed workers**: a node marked `device: gpu` runs on the machine that + has one, which dials out to the engine rather than needing to be reachable. +- An **HTTP API** that came first — everything the browser does, you can do from + a script — plus an MCP endpoint for agents. + +## Start here + +The setup is genuinely different depending on what you are here for, so +[Getting started](getting-started/index.md) splits in two: one path installs a +Python package and gets out of your way, the other stands up a server you will +be running for years. Everything after that is shared. + +If you would rather look before installing, the hosted demo at +[fluksio.com](https://fluksio.com) runs a real instance with a small-house panel +and a training pipeline on it. + +## Where things are + +| If you want to… | Read | +|---|---| +| Understand what a flow, a node and a message are | [Concepts](concepts/flows.md) | +| Drive Fluksio from the browser | [The interface](interface/index.md) | +| Drive it from Python, a shell or CI | [Code and the CLI](code/cli.md) | +| Look up a node type or a payload type | [Reference](reference/node-types.md) | + +Fluksio is self-hosted by default. An installation runs offline, keeps its data +on its own disk, and never contacts anything unless you +[connect it to a portal](interface/portal.md) yourself. diff --git a/docs/interface/dashboards.md b/docs/interface/dashboards.md new file mode 100644 index 0000000..945edba --- /dev/null +++ b/docs/interface/dashboards.md @@ -0,0 +1,137 @@ +# Dashboards and panels + +A dashboard is a grid of widgets bound to message names — the same names that +wire the graph. It is its own document rather than a set of nodes, so it reads +across flows without belonging to any of them, and a flow stays the logic it +was. + +There is no second tool and no separate deployment. You build the panel next to +the thing feeding it. + +## Building one + +**Dashboards → New dashboard**, then drag widgets onto the grid from the widget +picker and bind each to a message. + +Like flows, dashboards are drafts until you publish. A wall panel reads only +the published version, so a half-arranged page never reaches the wall. + +On a phone the dashboard is read, not arranged: the grid stacks to one column +and dragging is off. Picking a widget and editing its settings still works. + +## The widgets + +### Display + +| Widget | Binds to | Notes | +|---|---|---| +| **Value** | anything | a formatted reading with a unit and a precision | +| **Gauge** | `float`, `int` | min, max, unit | +| **Chart** | `float`, `int` | up to five series; see *Two kinds of chart* below | +| **Bar** | `float`, `int` | a reading, optionally nesting up to three inside it | +| **Icon** | numbers, booleans, weather strings | maps a value onto a glyph | +| **Text** | — | markdown you write; a label, a note, an instruction | +| **Agenda** | `list` | upcoming items, e.g. from a calendar connector | +| **Forecast** | `list` | a short outlook strip | +| **Notification** | `record` | title, body and severity — what an alert channel writes | +| **Clock** | — | the time, in a size a wall can read | + +### Input + +| Widget | Publishes | Notes | +|---|---|---| +| **Button** | a fixed value | one-shot: run it, open it, reset it | +| **Switch** | `bool` | on/off | +| **Slider** | `float`, `int` | min, max, step | +| **Input** | text or a number | free entry | +| **Dropdown** | one of a list | a mode, a scene, a preset | + +A control publishes the message it is bound to, exactly as a node would. On the +flow canvas it is drawn as a labelled endpoint feeding the nodes that read it, +so nobody has to guess where the value came from. + +### Typed bindings + +Widgets are type-checked against the message the same way ports are: a switch +takes a `bool`, a gauge takes a number, an agenda takes a `list`, a +notification takes a `record`. Bind one wrong and the editor says so rather +than drawing nothing. The same table is enforced on the server. + +Charts are the exception worth knowing: because identity across five series is +carried by lightness alone, a chart with more than one series always draws a +legend. + +## Two kinds of chart + +**Live.** Bind up to five messages and the chart draws the engine's own ring +buffer of recent values, extended over the socket as new ones land. Nothing +else is involved. This is what you want for "the last few hours of the living +room". + +**Querying.** The chart publishes a *request* — the window and resolution it +wants — exactly as a slider publishes a value, and draws the `series` some flow +answers with. What serves that request is the flow's business: typically a +small Python node that builds a Flux query, an InfluxDB node that runs it, and +another Python node that shapes the rows. + +That indirection is the point. The widget never learns which database answered +it, so swapping the store is a change to one flow and nothing else. The answer +also states what it was computed for, so an answer to a different question is +ignored rather than two charts overwriting each other's picture. + +## Showing one + +- `/view/{name}` — a browser tab pointed at one dashboard. Needs an ordinary + session. +- **Panels** — a named device that pairs instead of logging in. Below. + +## Panels: hanging a screen on a wall + +A wall tablet has no keyboard, so it pairs. + +1. **Dashboards → Panels**, add a panel named after where it hangs, and tick + the dashboards it shows. More than one and the screen draws a rail to switch + between them. +2. Point the device's browser at the link the dialog shows. The device then + displays a six-character code. +3. Type that code into the same panel's **Pair device** field. The line under + it names what is holding the code — check it is the screen you just hung, + because approving adopts whatever answered. The screen picks the credential + up within a few seconds and never asks again. + +What the screen holds is not a login. It reaches that panel's published +dashboards and the message endpoints its widgets speak, and nothing else. + +It cannot be made strictly read-only, and that is honest rather than an +oversight: a querying chart publishes its request, and a control on a panel is +the reason you put one there. Deleting the panel revokes the credential, which +is also how you retire a device — the screen falls back to asking for a new +code. + +!!! note "If the link is wrong" + + The pairing link is built from the installation's `FRONTEND_HOST`. If that + is not the address devices on your network actually reach, fix the setting + rather than the link: it is the same one password-reset mails and the OAuth + metadata are built from. + +### A screen somewhere you cannot reach + +Another building, someone else's network, no route in. An installation +[enrolled with a portal](portal.md) shows a second link, +`https://hub.${DOMAIN}/i/{installation-id}/panel`, and the same three steps +work through it: the portal serves that one page without a session, forwards +the pairing calls down the tunnel, and mints the credential when you approve +the code. The pairing line then reads *via portal*. + +The portal names the panel and nothing else. What the panel may read is decided +on the installation, on every call, by the same check a locally paired screen +passes. Two differences: it acts as the account the installation was enrolled +with rather than as whoever approved it, and deleting the panel stops it here +immediately while the portal's copy of the token expires on its own. + +## See also + +- [Payload types](../reference/payload-types.md) — what a widget can bind to +- [Flows, nodes and messages](../concepts/flows.md) — where the names come from +- [Accounts and the portal](portal.md) — reaching all of this from outside diff --git a/docs/interface/flow-editor.md b/docs/interface/flow-editor.md new file mode 100644 index 0000000..dc52dfb --- /dev/null +++ b/docs/interface/flow-editor.md @@ -0,0 +1,156 @@ +# The flow editor + +The canvas is where most of the work happens. It floats its chrome over a +full-bleed graph, so what you are looking at is always the flow. + +## The canvas + +Nodes are laid out for you — left to right on a desktop, top to bottom on a +phone — and the arrangement is recomputed as the graph changes. There is +nothing to drag and nothing to tidy. + +That is on purpose. The edges come from message names, so a stored layout would +be a second, weaker description of the same thing, and something to merge when +two people edit the same flow. It also keeps flows honest: a graph nobody can +hand-arrange is one worth splitting. + +What is drawn: + +- **Your nodes**, with a status dot — running, succeeded, failed, or nothing at + all when idle. Status is always named in words too, never colour alone. +- **Edges**, which pulse when a value goes down them. Click one to inspect it. +- **Endpoints** — anything wired into this flow that is not a node in it: a + dashboard control writing a message, a tile reading one, a node in another + flow on the far side of a dotted name. Drawn as labelled boxes at the + boundary, so a value never appears to come from nowhere. + +Values are live. The editor holds a websocket to the engine, and the connection +banner tells you when it is not. + +## Adding a node + +**Add node** on the dock, or ⌘K / Ctrl-K for the command palette, which also +jumps between flows and offers your shared nodes. Pick a type and it appears on +the canvas with its panel open. + +## The node panel + +Floats on the right; a full-screen sheet on a phone. + +**Consumes** and **Provides** are the ports. Each row is a message name, a +payload type, and — on an output — a *streaming* toggle for ports that publish +repeatedly during one execution. The name field suggests messages already in +the flow, which is usually how you wire something: type the name that already +exists. + +Renaming an output renames the message everywhere it is read. Inputs are not +renamed, because an input is more often re-pointed at a different message than +renamed. + +**Settings** is generated from the node type's parameter schema — so an MQTT +node shows broker, topic and QoS, and an ntfy node shows server, topic and +priority. A field marked as a credential renders as a picker over your stored +[secrets](operations.md) instead of a text box. + +A Function node gets a second settings section where you add your own: a +setting is a constant of this node's code and arrives as an argument beside its +ports. See [Where a node's values come from](../concepts/values.md). + +**Timeout** (Function nodes) is how many seconds the code may run before it is +stopped. Once a node streams, this becomes an *idle* timeout: it measures +silence, not duration, so a node that yields every few seconds can run for +hours under a timeout of 300. + +**Code** is the editor. It saves as you type and applies on ⌘S; the node +reloads without the flow stopping. A node file defines `process(...)`; if it +defines exactly one public function under another name, that one is used +instead. + +**Shared** turns a node's code into a library entry other flows can reuse. +Editing a shared node edits the copy every flow using it runs, which is the +point and also the caution. + +!!! note "Not in the panel (yet)" + + `device` and `device_policy` — which machine a node's code runs on — are set + through the API rather than the panel. See + [Remote workers](../code/workers.md). + +## The flow panel + +Click the canvas title to open the flow's own settings. + +- **Running** — the switch that starts and stops the flow. Stopped means none + of its subscriptions, schedules or webhooks exist. +- **Mode** — `live` or `batch`. Live reacts to what arrives; batch only runs + when a run asks it to. See [Runs](../concepts/runs.md). +- **Inputs** — messages the flow takes from outside, with the value each starts + from. For a batch flow these are also the run's parameters. +- **Outputs** (batch only) — which messages a run reports as its result. +- **Contents** — a reminder that this flow's name namespaces every message + inside it. + +## Inspecting an edge + +Click a wire. You get the last payload that went down it, when, and a sparkline +of its recent numeric history. You can also republish that value from the node +that produced it, which re-runs everything downstream — the fastest way to test +a change without waiting for the real sensor. + +## The dock + +Along the bottom, floating over the canvas. + +| Control | What it does | +|---|---| +| **Add node** | opens the palette | +| **Fit** | returns to the view the flow opened with | +| **Run** | injects a value by hand; on a batch flow, opens the parameters dialog | +| **Pause / Resume** | holds messages instead of running them | +| **Step** | releases exactly one held message | +| **Publish** | puts the draft live | +| **Discard** | throws the draft away | +| **Issues** | what is stopping this flow from running, per node | +| **Logs** | what the nodes printed, filterable | + +Pause and step together are how you test a flow before it moves anything +physical: pause it, inject a value, step it through, watch the values appear on +the edges, then publish. + +## Shortcuts + +| Chord | Action | +|---|---| +| ⌘K / Ctrl-K | command palette | +| ⌘S / Ctrl-S | publish the flow — or, with focus in the code editor, apply the code | +| ⌘Z / ⌘⇧Z | undo / redo (the flow; the code editor has its own) | +| ⌘C / ⌘V | copy and paste nodes, including between flows | + +## Drafts and publishing + +Saving writes a draft; the engine keeps running what was published. Publishing +promotes it and reloads. + +If someone else saved the same flow while you were working, publishing answers +with a conflict rather than overwriting them. The Flows list offers **Publish +all changes** when several flows are sitting on drafts. + +## Issues + +The canvas validates as you edit and marks the node each issue belongs to: + +- a port that needs a message nothing in reach provides +- a message that exists but has never held a value +- a dependency cycle +- a node reading a message it also writes, with nothing to start it from +- code that did not load +- a webhook with no shared secret (advisory — it does not stop the flow) + +A flow with any of these except the last does not run, and the health summary +on Home counts it. + +## See also + +- [Flows, nodes and messages](../concepts/flows.md) — the model behind the canvas +- [Node types](../reference/node-types.md) — everything the palette offers +- [Writing node code](../code/nodes.md) — what goes in the editor diff --git a/docs/interface/index.md b/docs/interface/index.md new file mode 100644 index 0000000..df99f09 --- /dev/null +++ b/docs/interface/index.md @@ -0,0 +1,88 @@ +# The dashboard app + +The web interface is a single-page app served at `app.${DOMAIN}` — or, for an +installation reached through a portal, at `${DOMAIN}/i/{installation-id}`. +Either way it is the same application, and it is a client of the same REST API +you can script against. + +Sign in with the account the installation was created with. On a fresh +installation that account was printed once, on the first start. + +## The shell + +A floating sidebar on the left, the selected screen filling the rest. On a +phone the sidebar collapses to a sheet. + +| Entry | What lives there | +|---|---| +| **Home** | the brain graph, health, and everything that recently happened | +| **Flows** | the list of flows, and the canvas for each | +| **Dashboards** | the widget canvases, and the panels that display them | +| **Secrets** | credentials your nodes reference without holding | +| **Modules** | the Python packages your node code may import | +| **Alerts** | where failures get sent | +| **Admin** | users (superusers only) | +| **Settings** | your account, appearance, and remote access | + +## Home + +The one screen you leave open. Three things share it. + +### The brain graph + +Every flow drawn as a neuron, wired to the flows it exchanges messages with. +This is the brand mark made live, and it is also the fastest read on the +installation: a neuron pulses when its flow is running work, and its ring turns +terracotta when the flow cannot run as written. A neuron with a problem keeps +its label showing so you can see which one it is without hovering. + +Each flow also has a switch beside it in the list, which starts and stops it. + +### Health + +Always answers, degraded or not. The tiles cover: + +- **Flows** — total, running, paused, quarantined, and how many cannot run + because their graph does not validate +- **Nodes** — how many failed to load +- **Queue** — depth, and how old the oldest pending item is +- **Loop lag** — whether the engine's event loop is keeping up + +`status: degraded` comes with a list of named problems, in words. "3 flow(s) +cannot run: house, pv, hallway" is more useful than a red dot, so that is what +it says. + +### Activity + +Charts of executions and failures over the selected range (1h / 6h / 24h / 7d), +with the recent runs, recent failures, dead-lettered work and the audit trail +underneath. + +The charts are scrubbable: hover a moment and the lists below filter to it, +click to hold it while you read. That turns "something went wrong around two +o'clock" into the actual rows. + +## Flows + +The list shows each flow's title, node count, whether it has unpublished +changes, and whether it is enabled, paused or quarantined. The toolbar searches, +creates, and offers **Publish all changes** when several flows have drafts. + +Opening one takes you to [the flow editor](flow-editor.md). + +## Everything else + +- [The flow editor](flow-editor.md) — the canvas, the code editor, running and + testing +- [Dashboards and panels](dashboards.md) — widgets, bindings, and hanging a + screen on a wall +- [Secrets, modules and alerts](operations.md) — the three screens that keep an + installation running +- [Accounts and the portal](portal.md) — reaching an installation from outside + its network + +## Appearance + +Light and dark follow your system by default; **Settings → Appearance** +overrides it. Both themes are first-class — the wall-panel view in particular is +designed to be legible in dark from across a room. diff --git a/docs/interface/operations.md b/docs/interface/operations.md new file mode 100644 index 0000000..3dc6a6f --- /dev/null +++ b/docs/interface/operations.md @@ -0,0 +1,131 @@ +# Secrets, modules and alerts + +Three screens that have nothing to do with each other except that an +installation you actually depend on needs all of them. + +## Secrets + +A node never holds a credential. A parameter written as + +```json +{"$secret": "influx-token"} +``` + +is replaced with the stored value when the node is built, and the editor +renders any field a node type marks as a credential as a picker over what you +have stored. + +**Secrets** is where the values live. Add a name and a value; the value is +never shown again, and the list only ever shows names. + +They are encrypted at rest with a key derived from the installation's +`SECRET_KEY`, and kept **outside** the flow repository. That matters because +flows are a git repository you may well push somewhere: what gets committed and +shared never contains a password. + +A name that no longer resolves is a clear error on the node that wanted it — +"No secret named 'x' — add it under Secrets" — rather than a mysterious +authentication failure. + +!!! warning "Rotating `SECRET_KEY`" + + The secrets store is encrypted with a key derived from it. Change + `SECRET_KEY` and the store no longer decrypts. Re-enter the secrets, or + plan the rotation properly. + +## Modules + +Node code runs in a virtual environment of its own, on the installation's data +volume — separate from the packages Fluksio itself runs on. A pin of yours can +never shadow one of ours, and vice versa. + +**Modules** is a pip manifest and an **Apply** button. It is installed with +`uv pip sync`, so a line you delete is uninstalled rather than left behind. +The manifest is versioned alongside your flows, so what a deployment installed +is recorded with what uses it. + +The page shows what is currently installed, the interpreter's Python version, +and whether the manifest and the environment agree. A failed resolve comes back +as `uv`'s own output, in the pane, because that is the only thing anyone can +act on. + +An install takes effect immediately — the worker processes pick up the new +interpreter state without the engine restarting. + +!!! note "Big scientific stacks" + + Installing torch into this venv works and is often the simplest thing. The + alternative, when you already have an environment you would rather not + duplicate, is to attach it as a worker: `fluksio-worker --python + /path/to/venv/bin/python`. See [Remote workers](../code/workers.md). + +## Alerts + +Everything that goes wrong already travels the engine's event bus. Alerts is +where you say who hears about it. + +### Channels + +| Kind | Settings | +|---|---| +| **ntfy** | server, topic, and a token for a protected topic | +| **SMTP** | an address to send to (the installation's mail settings do the rest) | +| **Webhook** | a URL to POST to | +| **Dashboard** | a message name a notification widget reads | + +Any channel setting may hold a `{"$secret": "name"}` reference instead of a +literal, so a webhook URL with a token in it does not sit in the config. + +### Rules + +A rule is a set of events and the channels they go to. **A rule with no events +ticked covers all of them**, including ones added in later versions — which is +usually what you want for the first rule you write. + +| Event | Fires when | +|---|---| +| `node_error` | a node failed | +| `node_health` | a connection dropped | +| `flow_quarantined` | a flow crashed often enough that the engine stopped restarting it | +| `task_crashed` | a background task crashed | +| `engine_degraded` | the engine is struggling | +| `cascade_dropped` | work was given up on | +| `queue_unavailable` | the work queue is unreachable | + +**Test** sends one message through a channel so you find out it is misconfigured +now rather than at three in the morning. + +### What it mostly does is not send + +This is the part worth knowing before you decide it is broken: + +- the same fault repeating is **one** alert, with a cooldown (15 minutes by + default per rule) +- a connection flapping is one story, not one alert per transition +- however bad it gets, **ten alerts an hour** is the ceiling + +An engine that is genuinely on fire produces a handful of messages, not +thirty-six thousand. + +## Admin: users + +Superusers get an **Admin** screen: add, edit and remove accounts, and see who +is a superuser. + +Accounts that arrived through a portal are badged *Portal*. They have no +password — they sign in at the portal, not here — and they are never +superusers. Deleting one cuts their access immediately and independently of the +portal, so it works even if the portal cannot be reached. See +[Accounts and the portal](portal.md). + +The same screen manages OAuth clients, which is how [agents](../code/agents.md) +are registered. + +## Settings + +Your own account: name, email, password, and appearance (light, dark, or +follow the system). + +**Remote access** is where an installation is paired with a portal, remote +users are admitted, and the link is cut again. That has [its own +page](portal.md). diff --git a/docs/interface/portal.md b/docs/interface/portal.md new file mode 100644 index 0000000..1703869 --- /dev/null +++ b/docs/interface/portal.md @@ -0,0 +1,117 @@ +# Accounts and the portal + +Fluksio is self-hosted by default. An installation runs offline, keeps its data +on its own disk, and never contacts anything unless you tell it to. + +The **portal** is optional, and it exists to solve two specific problems: + +1. **Your machine has no inbound route.** A homelab behind CGNAT, a cluster + node with no open ports, a laptop. Opening one is work, and often not + allowed. +2. **You want a browser on it anyway.** A `pip install fluksio` installation + has no web server for the dashboard at all. + +An enrolled installation dials *out* to the portal and holds one websocket +open. The portal serves the dashboard from its own side, and only the API calls +travel down the tunnel — so the interface loads at portal speed and your +machine stays unreachable from the internet. + +## Enrolling + +Two halves, deliberately: whoever performs the second step decides what the +installation's owner gets. + +**On the portal** (`hub.${DOMAIN}`, or [fluksio.com](https://fluksio.com) for +the hosted one): **Installations → Add installation**, give it a name, and copy +the code. + +**On the installation**, either from the dashboard: + +> **Settings → Remote access**, enter the portal URL and the code, press +> **Connect**. + +or from the command line, which is the path for an installation with no web +interface of its own: + +```sh +fluksio enroll ABCD-1234 --portal https://hub.fluksio.com +``` + +The code is single-use and expires in fifteen minutes. + +A portal session then arrives as *that local account* — the settings screen +states this plainly, because it is the whole security model in one sentence. +Use `--as someone@example.com` to enrol as a specific local account when the +installation has several superusers. + +Once enrolled, the installation appears under **Installations** with its +status, when it was last seen and its version. **Open** takes you to its +dashboard at `${DOMAIN}/i/{installation-id}`. + +## What the portal can and cannot do + +The portal holds one credential for your installation and proxies requests down +the tunnel. What those requests may do is decided **on the installation**, by +the same checks a local session passes. + +The trust anchor is a signing keypair on the portal. Every installation pins +its public half at enrolment and rejects anything else — which is what stops a +hijacked DNS entry or a mis-issued certificate from impersonating the portal. + +## Letting someone else in + +Anyone else on the portal reaches your installation only if a superuser there +admits them, and they arrive as a local user of their own rather than as you. + +1. **They**: **Installations → Join an installation**, and copy the code. It is + bound to their portal account and expires in fifteen minutes. +2. **You**, on the installation: **Settings → Remote access → Add remote + user**, and enter the code. +3. They now see the installation under **Installations**, marked *Shared by*, + with **Open** and nothing else. Renaming, re-keying and removing stay with + you. + +The installation redeems that code against the portal using its own credential. +A portal session cannot do this — which is what stops somebody you let in from +letting others in. + +On the installation they appear under **Admin → Users**, badged *Portal*, never +a superuser and with no password. + +## Cutting it off + +| From | Action | Effect | +|---|---|---| +| The portal | **New code** | rotates the credential and drops the current link | +| The portal | **Remove** | deletes the registration and cuts the connection | +| The installation | **Disconnect** | unilateral and immediate — the portal's tokens stop verifying here whatever the portal still has on file | +| The installation | delete a user under **Admin → Users** | that one person, immediately, independent of the portal | + +**New code** also cuts every credential the portal minted for this installation, +including [wall panels paired through it](dashboards.md#a-screen-somewhere-you-cannot-reach). + +The installation's own **Disconnect** is the one to reach for if you are ever +unsure: it does not need the portal's cooperation. + +## Running your own portal + +The portal is the `index` stack's `hub` service — accounts, the registry of +connected installations, and the websocket each one dials in on. Two things +about it are load-bearing: + +- **It runs a single process.** It keeps its attached installations in the + memory of the process holding their sockets, so a second worker would answer + for links it does not hold. Scaling out needs a routing layer first. +- **Back up the signing keypair with the database.** Replacing it forces every + installation to be enrolled again. + +Websocket upgrades must be enabled on the `hub` hostname in whatever proxy +fronts it. Without them every installation sits in a reconnect loop and the +portal shows them all offline. + +## See also + +- [The `fluksio` command](../code/cli.md) — `fluksio enroll`, and what it writes +- [Dashboards and panels](dashboards.md) — pairing a wall screen through the portal +- [Getting started: data science](../getting-started/data-science.md) — the + shortest path from a pip install to a browser diff --git a/docs/llms.txt b/docs/llms.txt new file mode 100644 index 0000000..e36af6a --- /dev/null +++ b/docs/llms.txt @@ -0,0 +1,38 @@ +# Fluksio + +> Fluksio is a node-based automation engine. A flow is a graph of small +> functions wired by message name, kept running by one resident engine. The +> same engine serves two shapes of work: live flows that never end (buildings, +> labs, homelabs) and batch runs that finish and leave a record (experiments, +> ML pipelines, CI-style jobs). Self-hosted by default; an optional portal +> exists only to reach an installation that has no inbound route. + +## Getting started +- [Pick your starting point](/getting-started/): the two setup paths and how to choose +- [Data science](/getting-started/data-science/): pip install, wrap a training script, runs and sweeps +- [Facility automation](/getting-started/facility-automation/): Docker stack, MQTT to dashboard, InfluxDB, workers + +## Concepts +- [Flows, nodes and messages](/concepts/flows/): the graph comes from message names, not drawn wires +- [Where a node's values come from](/concepts/values/): ports, settings, flow inputs +- [Keeping state in a flow](/concepts/state/): a message a node both reads and writes +- [Runs: pipelines that finish](/concepts/runs/): batch mode, metrics, artifacts, sweeps, durability + +## The interface +- [The dashboard app](/interface/): the shell, Home, health and activity +- [The flow editor](/interface/flow-editor/): canvas, node panel, dock, shortcuts, drafts +- [Dashboards and panels](/interface/dashboards/): widgets, typed bindings, wall screens +- [Secrets, modules and alerts](/interface/operations/): credentials, packages, failure routing +- [Accounts and the portal](/interface/portal/): enrolment, remote users, revoking + +## Code and the CLI +- [The fluksio command](/code/cli/): serve, enroll, worker, and the data directory +- [Writing node code](/code/nodes/): process(), generators, fluksio.emit, artifacts +- [The HTTP API](/code/api/): auth, flows, messages, runs, observability, websocket +- [Remote workers](/code/workers/): fluksio-worker, device labels, artifacts across hosts +- [Agents over MCP](/code/agents/): the tool list and the security model + +## Reference +- [Node types](/reference/node-types/): every built-in type and its settings +- [Payload types](/reference/payload-types/): float, int, str, bool, json, record, list, series, artifact +- [Configuration](/reference/configuration/): every environment variable diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md new file mode 100644 index 0000000..63ec324 --- /dev/null +++ b/docs/reference/configuration.md @@ -0,0 +1,127 @@ +# Configuration + +Every setting comes from the environment, or from an env file. Which file +depends on how the installation was started: + +| Started with | Reads | +|---|---| +| `fluksio serve` | `env` inside the data directory (`$FLUKSIO_ENV_FILE`) | +| the Docker stack | `.env` beside `docker/` | + +Anything already exported wins over the file. + +## Storage + +| Variable | Default | Notes | +|---|---|---| +| `DATA_DIR` | `flow-data` (`~/.fluksio` via the CLI) | everything below derives from this | +| `DATABASE_URL` | SQLite in `DATA_DIR` | any SQLAlchemy URL | +| `FLOWS_DIR` | `$DATA_DIR/flows` | the git repository holding flows | +| `SECRETS_FILE` | `$DATA_DIR/secrets.enc` | encrypted credentials, deliberately outside the repo | +| `ALERTS_FILE` | `$DATA_DIR/alerts.json` | channels and rules | +| `PANELS_FILE` | `$DATA_DIR/panels.json` | wall-panel pairings | +| `OAUTH_PRIVATE_KEY_FILE` | `$DATA_DIR/oauth-key.pem` | signs agent and worker tokens | +| `CLOUD_CONFIG_FILE` | `$DATA_DIR/cloud.json` | the portal enrolment, if any | + +Set `DATA_DIR` and the rest follow. Set one explicitly and it wins — which is +what the container images do to pin everything onto `/data`. + +!!! warning "The four files that must be on persistent storage" + + `secrets.enc`, `alerts.json`, `panels.json` and `oauth-key.pem` are written + at runtime. In a container, anything not on a volume lands in the writable + layer and is lost on the next rebuild — un-pairing every screen and + revoking every agent. + +## State + +| Variable | Default | Notes | +|---|---|---| +| `REDIS_HOST` | unset | without it, flow state lives in memory and does not survive a restart | +| `REDIS_PORT` | `6379` | | + +Flow state is the last value of every message, node memory, and the run queue. +Redis here is persistence, not a cache — run it with append-only persistence +on. + +## Identity and access + +| Variable | Default | Notes | +|---|---|---| +| `SECRET_KEY` | generated | signs sessions and derives the secrets-store key | +| `ACCESS_TOKEN_EXPIRE_MINUTES` | `11520` (8 days) | | +| `FIRST_SUPERUSER` | — | absent means the CLI creates one on first run | +| `FIRST_SUPERUSER_PASSWORD` | — | absent means one is generated and printed once | +| `DOMAIN` | `localhost` | what the API and OAuth issuer are built from | +| `FRONTEND_HOST` | `http://localhost:5173` | used in mails, OAuth metadata and panel pairing links | +| `BACKEND_CORS_ORIGINS` | `[]` | comma-separated; `FRONTEND_HOST` is always allowed | + +!!! danger "Rotating `SECRET_KEY`" + + The secrets store is encrypted with a key derived from it. Change it and + the store stops decrypting, and every session is signed out. Re-enter your + secrets, or plan the rotation properly. + +## Environment + +| Variable | Default | Notes | +|---|---|---| +| `ENVIRONMENT` | `local` | `local`, `staging` or `production` | +| `PRIVATE_API_ENABLED` | `false` | unauthenticated test-only endpoints; needs `ENVIRONMENT=local` too | + +`production` closes `/docs`, `/redoc` and the OpenAPI document, because the +schema enumerates every endpoint the installation serves — including the paths +webhook nodes mounted at runtime. It also turns a `changethis` secret from a +warning into a refusal to start. + +## The engine + +| Variable | Default | Notes | +|---|---|---| +| `FLOW_MAX_WORKERS` | `4` | node-code subprocesses run in parallel | +| `FLOW_NODE_TIMEOUT` | `30.0` | seconds a node may run, unless it sets its own | +| `OBS_RETENTION_DAYS` | `30` | how long metrics, events and run records are kept | + +## Agents + +| Variable | Default | Notes | +|---|---|---| +| `MCP_ENABLED` | `false` | opens the `/mcp` endpoint **and** OAuth client registration | +| `MCP_TOKEN_EXPIRE_MINUTES` | `60` | an agent's token is a bearer secret held by a program | +| `MCP_REFRESH_EXPIRE_DAYS` | `30` | | +| `OAUTH_CODE_EXPIRE_SECONDS` | `60` | | + +See [Agents over MCP](../code/agents.md). + +## Mail + +Needed for password-reset mails. Without `SMTP_HOST` and `EMAILS_FROM_EMAIL`, +mail is simply off. + +| Variable | Default | +|---|---| +| `SMTP_HOST` | — | +| `SMTP_PORT` | `587` | +| `SMTP_USER` / `SMTP_PASSWORD` | — | +| `SMTP_TLS` / `SMTP_SSL` | `true` / `false` | +| `EMAILS_FROM_EMAIL` | — | +| `EMAILS_FROM_NAME` | `Fluksio` | +| `EMAIL_RESET_TOKEN_EXPIRE_HOURS` | `48` | + +## Monitoring + +| Variable | Default | Notes | +|---|---|---| +| `SENTRY_DSN` | — | error reporting, if you want it | + +## Health check + +`GET /api/v1/utils/health/` is a *deep* check: it fails when the event loop is +wedged or the state backend is gone, not just when the process is up. That is +what the container healthcheck probes, and what an autoheal sidecar restarts +on. + +## See also + +- [The `fluksio` command](../code/cli.md) — what the data directory holds +- [Getting started: facility automation](../getting-started/facility-automation.md) diff --git a/docs/reference/node-types.md b/docs/reference/node-types.md new file mode 100644 index 0000000..d0ce49c --- /dev/null +++ b/docs/reference/node-types.md @@ -0,0 +1,286 @@ +# Node types + +Every type the canvas can place. Each is configured by filling in a form the +editor generates from its parameter schema, so they all behave the same way. + +Anything here could be written as a **Function** node — that is what the +function node is for. These exist because the same handful of shapes account +for most of a real installation, and a rule you fill in is easier to read on a +canvas, and to change, than five lines of code repeated eighty times. + +`GET /flows/node-types` returns this list with each type's full parameter +schema, including any [connectors](#connectors) installed on your instance. + +## Function + +**`python`** — your own code, run on every incoming message. + +The one type with a source file. Its arguments are its input ports and its own +settings; its return value is a dict keyed by output ports. See +[Writing node code](../code/nodes.md). + +Settings on a Function node are free-form: you add them by name, and they +arrive as keyword arguments. + +## Integrations + +### MQTT + +**`mqtt`** — subscribe to topics, or publish what arrives on its inputs. + +A node with *outputs only* subscribes; a node with *inputs* publishes. + +| Setting | Default | Notes | +|---|---|---| +| `topic` | `*` | one topic for every port, or `{"message": "some/topic"}` per port | +| `broker_host` | `localhost` | | +| `broker_port` | `1883` | | +| `username` / `password` | — | `password` takes a secret reference | +| `client_id` | — | | +| `qos` | `0` | 0, 1 or 2 | +| `retain` | `false` | on published messages | +| `keepalive` | `60` | seconds | + +Nodes sharing a broker share one connection. + +### HTTP + +**`http`** — receive data on a webhook, or send it to a URL. + +Outputs only makes it a **webhook**: the engine mounts a route at +`/hooks/{flow}/{url}/{secret}` while the flow runs. Inputs make it a **sender**. + +| Setting | Default | Notes | +|---|---|---| +| `url` | — | the route path in webhook mode, the full URL in sender mode | +| `method` | `POST` | `GET` or `POST` | +| `timeout` | `30` | seconds, sender mode | +| `headers` | `{}` | | +| `secret` | — | shared secret appended to the webhook URL; takes a secret reference | + +!!! warning "A webhook with no secret is open to anyone who can reach the host." + + The canvas flags this as an advisory issue rather than stopping the flow, + because a webhook on a private network is a legitimate thing to want. + +### InfluxDB + +**`influxdb`** — write measurements to a bucket, or read them back. + +| Setting | Notes | +|---|---| +| `url`, `token`, `org`, `bucket` | connection; `token` takes a secret reference | +| `write_precision` | `ns`, `us`, `ms` (default) or `s` | +| `query_range` | default window for queries, e.g. `-1h` | +| `writes` | per-input: `measurement`, `field`, `tags` | +| `queries` | per-output: `measurement`, `field`, `tags`, `range`, `aggregation` | + +```json +{ + "writes": { + "living_temperature": { + "measurement": "environment", + "field": "temp_c", + "tags": {"room": "living"} + } + } +} +``` + +**Query passthrough.** An incoming message holding a `flux` key is run as +written, and the rows come back on the first output port as +`{"rows": [{ts, value, field, measurement, tags}], ...}`. + +That is what keeps a database node a database node: it holds the credentials +and the connection and nothing else. Building a query and shaping its rows are +ordinary Function nodes on either side, so a dashboard widget never learns which +database answered it. + +### Notification + +**`ntfy`** — push an incoming value to a phone through [ntfy](https://ntfy.sh). + +| Setting | Default | Notes | +|---|---|---| +| `server` | `https://ntfy.sh` | | +| `topic` | — | required | +| `title` | — | | +| `priority` | `default` | `min`, `low`, `default`, `high`, `urgent` | +| `tags` | — | comma-separated ntfy tags | +| `token` | — | for a protected topic; takes a secret reference | + +This is a *flow* deciding something is worth saying. The engine reporting that +it broke is [Alerts](../interface/operations.md#alerts), which is a different +thing configured elsewhere. + +## Timing + +### Inject + +**`inject`** — emit a value on request, on a timer, or when the flow starts. + +| Setting | Default | Notes | +|---|---|---| +| `payload` | current time | what to emit | +| `payloads` | `{}` | per-output-port payloads, keyed by port | +| `interval` | `0` | emit every n seconds; 0 means never on its own | +| `cron` | — | a five-field cron expression | +| `at_start` | `false` | emit once when the flow starts | +| `start_delay` | `1.0` | how long to wait before that first emission | + +The most-placed trigger in a real installation — mostly as a button someone +presses. + +### Delay & schedule + +**`delay`** — hold messages back, limit their rate, or emit on a schedule. + +| Setting | Default | Notes | +|---|---|---| +| `delay` | `0` | seconds to hold each message | +| `interval` | `0` | minimum seconds between forwards | +| `mapping` | `{}` | input port → output port; paired in order when empty | +| `cron` | — | five-field expression | + +Order of operations: rate check → delay → forward. With a `cron` and no inputs +it emits the current time on each tick; with inputs it emits the last value it +received. + +!!! note "Not in a batch flow" + + A rate limit holds a value back for a timer to release, and a run has no + timer — the value would be dropped rather than delayed. Submitting a batch + flow with a rate-limited port is refused instead. + +### Trigger + +**`trigger`** — send one value now and another once things go quiet. + +| Setting | Default | Notes | +|---|---|---| +| `first` | `true` | sent as soon as a value arrives | +| `then` | `false` | sent when the wait expires; empty sends nothing | +| `wait` | `60` | seconds of quiet before the second value | +| `extend` | `true` | a value arriving during the wait starts it over | + +The shape this exists for: *the door opened — turn the light on, and off again +in two minutes unless it opens again.* + +## Logic + +### Switch + +**`switch`** — send a value down one branch or another, by rule. + +Each rule names an output port; a matching value leaves through that port. +Comparisons: `eq`, `ne`, `gt`, `gte`, `lt`, `lte`, `contains`, `between`. + +| Setting | Default | Notes | +|---|---|---| +| `rules` | `[]` | checked in order; each names the output it routes to | +| `stop_at_first` | `true` | leave through the first matching rule only | +| `otherwise` | — | output for a value that matched nothing | + +Comparing a string to a number is a mistake in the rule, not a reason to take +the flow down: the rule simply does not match. + +### Change + +**`change`** — scale, offset, round or map a value on its way past. + +| Setting | Default | Notes | +|---|---|---| +| `scale` | `1.0` | multiply numbers by this | +| `offset` | `0.0` | then add this | +| `round_to` | — | decimal places | +| `mapping` | `{}` | replace a value with another, looked up as text | +| `default` | — | value when the lookup misses; empty passes it through | + +### Filter unchanged + +**`rbe`** — pass a value on only when it has actually changed. + +| Setting | Default | Notes | +|---|---|---| +| `deadband` | `0.0` | ignore numeric changes smaller than this | +| `deadband_percent` | `false` | read the deadband as a percentage | + +The cheapest fix for a sensor that reports every second and changes every hour. + +### Join + +**`join`** — gather several inputs into one object or list. + +| Setting | Default | Notes | +|---|---|---| +| `mode` | `object` | `object` or `array` | + +## System + +### Command + +**`exec`** — run a command in the engine's container and read its output. + +| Setting | Default | Notes | +|---|---|---| +| `command` | — | the command to run | +| `append_payload` | `false` | add the incoming value as one argument | +| `timeout` | `30` | seconds | +| `fail_on_error` | `false` | treat a non-zero exit as a node failure rather than output | + +Outputs the command's stdout, stderr and exit code. + +!!! warning "Inside the container, not on the host" + + A flow ported from something that read the host's journal, or poked a host + script, needs either a mount or a small listener on the host side. This + node cannot see the host. + +### File + +**`file`** — read a file into the flow, or write one out of it. + +| Setting | Default | Notes | +|---|---|---| +| `path` | — | relative to the engine's files directory | +| `mode` | `read` | `read`, `write` or `append` | +| `format` | `text` | `text` or `json` | +| `newline` | `true` | end each written record with a newline | + +Confined to a directory the engine owns. A flow that could name any path would +be a way to read the secrets store or overwrite a node's source. + +## Numeric + +### Perceptron + +**`mlp`** — a small neural layer over its numeric inputs. + +`output = weights @ inputs + biases`, with weights drawn from `seed` so a node +reloads identically. + +| Setting | Default | Notes | +|---|---|---| +| `seed` | `0` | | + +Kept as a worked example of numeric logic rather than as a modelling tool. If +you are training something, that is a [batch flow](../concepts/runs.md) and a +Function node. + +## Connectors + +Anything else in the palette came from an installed **connector** package — a +node type written against a published contract and discovered through the +`fluksio.node_types` entry point group. The editor shows which package supplied +it. + +A connector declares its contract version, and one written for a version this +engine does not speak is ignored rather than half-loaded. Installing or +upgrading one takes effect on the next engine restart, because Python does not +re-import a changed module and a rescan would promise more than it delivers. + +## See also + +- [Payload types](payload-types.md) — what a port may carry +- [Writing node code](../code/nodes.md) — the Function node +- [The flow editor](../interface/flow-editor.md) — placing and configuring them diff --git a/docs/reference/payload-types.md b/docs/reference/payload-types.md new file mode 100644 index 0000000..65f37c2 --- /dev/null +++ b/docs/reference/payload-types.md @@ -0,0 +1,129 @@ +# Payload types + +Every port declares a `dtype`, and every value that passes through it is +checked against that declaration. + +This is not decoration. It is what lets the dashboard editor offer you only the +messages a gauge can actually draw, what lets the canvas refuse a binding before +anything runs, and what lets a downstream node know the shape of what it is +getting before the flow starts. + +Everything on the wire is JSON. That is what lets the same value pass through +the state backend, the work queue and the worker protocol unchanged. + +## The scalars + +| `dtype` | Accepts | +|---|---| +| `float` | any number — `int` or `float`, but not `bool` | +| `int` | a whole number, not `bool` | +| `str` | a string | +| `bool` | exactly `true` or `false` | + +`bool` is an `int` subclass in Python and deliberately not a number here: a flag +is not a measurement, and a switch bound to a temperature is a mistake worth +catching. + +## The structured ones + +These are *declared shapes* rather than "some JSON", which is what makes a +widget binding checkable. + +### `record` + +Flat named scalars. + +```json +{"title": "Boiler", "body": "Pressure low", "severity": "warning"} +``` + +Nesting is deliberately out: a record that can contain a record is a schema +language, and the shape stops being readable from the declaration alone. + +Read by the **Notification** widget. It is also what an alert channel of kind +*dashboard* writes. + +### `list` + +Ordered items of one declared shape. The port also declares `item`: + +| `item` | Meaning | +|---|---| +| unset | `record` — what the agenda and forecast widgets read | +| `float`, `int`, `str`, `bool` | a list of scalars | +| `json` | anything | + +A list of lists, or a list of series, is refused. One declared level is the +point. + +### `series` + +Labelled lines of `(timestamp, value)` pairs — what a chart draws. + +```json +{ + "lines": [ + {"label": "living", "points": [[1717000000, 21.4], [1717000060, 21.5]]} + ], + "range": "-24h" +} +``` + +Keys beside `lines` are carried through untouched, which is how a querying +chart puts the window and resolution it asked for on the request and reads them +back off the answer. That is what stops an answer to a *different* question +from overwriting the picture. + +`GET /runs/series/compare` answers in this shape, which is why comparing three +training curves is a widget binding rather than a screen of its own. + +### `artifact` + +A reference to stored bytes. + +```json +{"digest": "sha256:…", "size": 4194304, "media_type": "application/octet-stream", "name": "weights.pt"} +``` + +Binary payloads — tensors, checkpoints, images — never travel as a message. The +bytes go to a content-addressed store and the message carries this. A +thirty-megabyte checkpoint never sits in the state backend, and the reference +stays valid wherever the store is reachable from, including on another machine. + +Node code produces one with `fluksio.save_artifact` and opens one with +`fluksio.load_artifact`. See [Writing node code](../code/nodes.md#bytes-artifacts). + +### `json` + +Anything JSON-serializable. The escape hatch, and the right answer when a +payload genuinely has no fixed shape. + +Reach for it last. A `json` port tells the canvas, the widget picker and the +next author nothing. + +## What a widget will bind to + +| Widget | Accepts | +|---|---| +| Gauge, Chart, Slider, Bar | `float`, `int` | +| Switch | `bool` | +| Agenda, Forecast | `list` | +| Notification | `record` | +| Value | anything | +| Icon | weather strings, booleans and numbers alike | +| Clock, Text | nothing — they bind to no message | + +Enforced on the server as well as in the editor. + +## Type failures + +A value that does not match its port's declaration raises on the node that +published it, naming the port and what arrived. It does not get published, and +it does not reach anything downstream — a wrong value stopping at its source is +much easier to diagnose than one propagating. + +## See also + +- [Flows, nodes and messages](../concepts/flows.md) +- [Node types](node-types.md) +- [Dashboards and panels](../interface/dashboards.md) diff --git a/docs/stylesheets/extra.css b/docs/stylesheets/extra.css new file mode 100644 index 0000000..9e53374 --- /dev/null +++ b/docs/stylesheets/extra.css @@ -0,0 +1,177 @@ +/* Fluksio brand theming for the documentation site. + * + * Follows the zensical custom color scheme approach + * (https://zensical.org/docs/setup/colors/#custom-color-schemes): override the + * Material `--md-*` CSS variables per [data-md-color-scheme]. Colors mirror the + * design tokens in ../DESIGN.md → ../../DESIGN-GUIDELINES.md: neutral white / + * near-black surfaces carrying one slate-blue accent, with terracotta reserved + * for the single secondary affordance per view. The built-in `default` (light) + * and `slate` (dark) schemes stay as the base so all of Material's derived + * styling keeps working; only the brand-relevant variables are remapped. + * + * NOTE: the modern theme paints a frosted, translucent header via + * `.md-header { backdrop-filter: blur(...); background: var(--md-default-bg-color--light) }`, + * which is the same `bg-card/80 backdrop-blur` shell both frontends float their + * chrome on. `--md-default-bg-color--light` therefore has to stay a + * *translucent* alpha of the background — an opaque value stops the header + * being see-through. + */ + +/* Slate-blue brand primary, shared by both schemes. + * + * `#59849b` is the identity blue in the wordmark, and deliberately not used + * here: it measures 4.04:1 against white, below WCAG AA for normal text. The + * values below are the same ones the apps ship as `--primary`. */ +[data-md-color-scheme="default"], +[data-md-color-scheme="slate"] { + --md-primary-fg-color: #4a7189; + --md-primary-fg-color--light: #7ba3b8; + --md-primary-fg-color--dark: #395767; + --md-primary-bg-color: #ffffff; + --md-primary-bg-color--light: rgba(255, 255, 255, 0.7); + + /* The one secondary accent. Used for the facility lane on Getting started + * and nowhere else, mirroring "one --brand-secondary affordance per view". */ + --fluksio-brand-secondary: #de8f6e; + + /* Footer stays dark in both schemes (Material convention), neutral tokens. */ + --md-footer-bg-color: #1a1a1a; + --md-footer-bg-color--dark: #0a0a0a; +} + +/* Light: white page, near-black text (--foreground is #333232, not pure black). */ +[data-md-color-scheme="default"] { + --md-default-bg-color: #ffffff; + --md-default-bg-color--light: rgba(255, 255, 255, 0.7); + --md-default-bg-color--lighter: rgba(255, 255, 255, 0.3); + --md-default-bg-color--lightest: rgba(255, 255, 255, 0.12); + + --md-default-fg-color: #333232; + --md-default-fg-color--light: #6b6b6b; + --md-default-fg-color--lighter: #9e9e9e; + --md-default-fg-color--lightest: #e4e4e4; + + --md-accent-fg-color: #395767; + --md-typeset-a-color: #4a7189; + + --md-code-bg-color: #f2f2f2; + --md-code-fg-color: #333232; +} + +/* Dark: near-black page, off-white text. Surfaces lift to #1a1a1a (--card) so + * code and quoted blocks read against the #0a0a0a backdrop. */ +[data-md-color-scheme="slate"] { + --md-default-bg-color: #0a0a0a; + --md-default-bg-color--light: rgba(10, 10, 10, 0.7); + --md-default-bg-color--lighter: rgba(10, 10, 10, 0.3); + --md-default-bg-color--lightest: rgba(10, 10, 10, 0.12); + + --md-default-fg-color: #f5f5f5; + --md-default-fg-color--light: #a3a3a3; + --md-default-fg-color--lighter: #6b6b6b; + --md-default-fg-color--lightest: #2a2a2a; + + --md-accent-fg-color: #7ba3b8; + --md-typeset-a-color: #7ba3b8; + + --md-code-bg-color: #1a1a1a; + --md-code-fg-color: #f5f5f5; + + --fluksio-brand-secondary: #e5a184; +} + +/* The theme highlights the active sidebar entry with a + * `--md-accent-fg-color--transparent` fill, which is still the built-in indigo + * because only the opaque accent is remapped above. Drop the fill in both + * schemes; the brand accent colour and the heavier weight the theme also + * applies keep the selection visible. */ +.md-nav--primary .md-nav__item .md-nav__link--active { + background: none; +} + +/* Radius follows the app's map: 16px for panels and cards, 12px for the + * smaller inline surfaces. */ +.md-typeset .admonition, +.md-typeset details, +.md-typeset pre > code, +.md-typeset .tabbed-content pre > code { + border-radius: 16px; +} + +.md-typeset code { + border-radius: 12px; +} + +/* Tables are the one place these docs go wide. Nothing may scroll the page + * sideways, so the box scrolls instead — DESIGN-GUIDELINES.md → Responsive. */ +.md-typeset .md-typeset__table { + overflow-x: auto; +} + +/* ── Audience cards ──────────────────────────────────────────────── + * + * Used on Getting started, where the split mirrors the landing page's audience + * switch: data science takes --primary, facility automation takes the one + * terracotta accent. Built out of Markdown lists via attr_list + md_in_html, + * so there is no HTML soup in the page source. + */ +.fluksio-lanes { + display: grid; + gap: 1rem; + margin: 1.5rem 0; +} + +@media screen and (min-width: 768px) { + .fluksio-lanes { + grid-template-columns: 1fr 1fr; + } +} + +.fluksio-lane { + border: 1px solid var(--md-default-fg-color--lightest); + border-radius: 16px; + padding: 1.25rem 1.5rem; + /* min-width:0 keeps a long line from stretching the grid track and, with it, + * the page. */ + min-width: 0; +} + +.fluksio-lane > :first-child { + margin-top: 0; +} + +.fluksio-lane > :last-child { + margin-bottom: 0; +} + +.fluksio-lane--science { + border-color: var(--md-primary-fg-color); +} + +.fluksio-lane--science h3 { + color: var(--md-primary-fg-color); +} + +.fluksio-lane--facility { + border-color: var(--fluksio-brand-secondary); +} + +.fluksio-lane--facility h3 { + color: var(--fluksio-brand-secondary); +} + +/* ── Brand mark ──────────────────────────────────────────────────── + * + * The mark's disc is painted `--foreground`, so the light file is a dark disc + * and disappears on the dark header. The theme renders the logo as an , + * whose src CSS cannot swap — so the image is hidden under `slate` and the + * anchor paints the dark mark instead. Both files are the same 24×24 viewBox, + * so the two look identical apart from that inversion. + */ +[data-md-color-scheme="slate"] .md-logo img { + visibility: hidden; +} + +[data-md-color-scheme="slate"] .md-logo { + background: url("../assets/favicon-dark.svg") center / 1.2rem 1.2rem no-repeat; +} diff --git a/zensical.toml b/zensical.toml new file mode 100644 index 0000000..650bdea --- /dev/null +++ b/zensical.toml @@ -0,0 +1,124 @@ +# Fluksio documentation site, built with zensical (https://zensical.org). +# Content lives in docs/, the static build goes to site/ (gitignored). +# Build: `make docs` | Live preview: `make docs-serve` +# This config is zensical-native (TOML). It is portable to Material for MkDocs +# by translating to mkdocs.yml if the engine ever needs to change. + +[project] +site_name = "Fluksio" +site_description = "Documentation for Fluksio, node-based automation for experiments and for buildings." +site_url = "https://docs.fluksio.com/" +# Footer line. HTML renders as-is, so the legal links (mirroring the index +# landing footer) and a link back to the main site live here. +copyright = """Copyright © 2026 Fluksio ·
fluksio.com · Impressum · Privacy Policy""" + +# Fluksio brand color schemes (slate blue on neutral white/near-black, with +# terracotta as the one secondary accent), layered over the built-in +# default/slate schemes. See docs/stylesheets/extra.css. +extra_css = ["stylesheets/extra.css"] + +# Site navigation. Must stay above the first [project.*] table header, or TOML +# would nest it inside that table and zensical would silently fall back to +# auto-generated (alphabetical) navigation. +nav = [ + { "Home" = "index.md" }, + { "Getting started" = [ + { "Pick your starting point" = "getting-started/index.md" }, + { "Data science" = "getting-started/data-science.md" }, + { "Facility automation" = "getting-started/facility-automation.md" }, + ] }, + { "Concepts" = [ + { "Flows, nodes and messages" = "concepts/flows.md" }, + { "Where a node's values come from" = "concepts/values.md" }, + { "Keeping state in a flow" = "concepts/state.md" }, + { "Runs: pipelines that finish" = "concepts/runs.md" }, + ] }, + { "The interface" = [ + { "The dashboard app" = "interface/index.md" }, + { "The flow editor" = "interface/flow-editor.md" }, + { "Dashboards and panels" = "interface/dashboards.md" }, + { "Secrets, modules and alerts" = "interface/operations.md" }, + { "Accounts and the portal" = "interface/portal.md" }, + ] }, + { "Code and the CLI" = [ + { "The fluksio command" = "code/cli.md" }, + { "Writing node code" = "code/nodes.md" }, + { "The HTTP API" = "code/api.md" }, + { "Remote workers" = "code/workers.md" }, + { "Agents over MCP" = "code/agents.md" }, + ] }, + { "Reference" = [ + { "Node types" = "reference/node-types.md" }, + { "Payload types" = "reference/payload-types.md" }, + { "Configuration" = "reference/configuration.md" }, + ] }, + # The roadmap, the design guidelines and the architecture material stay in + # the repositories — repo-only, intentionally not published. +] + +# Drop the "Made with Zensical" generator notice from the footer. +[project.extra] +generator = false + +[project.theme] +language = "en" +# The brand mark the dashboard SPA uses (frontend/public/favicon.svg), as both +# the browser favicon and the header logo, so the docs match the app. +favicon = "assets/favicon.svg" +logo = "assets/favicon.svg" +features = [ + "content.code.copy", + "navigation.footer", + "navigation.indexes", + "navigation.instant", + "navigation.sections", + "navigation.top", + "navigation.tracking", + "search.highlight", + "toc.follow", +] + +# Body font mirrors both frontends (Inter); code font stays the theme default, +# which matches the app's system monospace stack. +[project.theme.font] +text = "Inter" + +# Three states, the same as both frontends: follow the system, then an explicit +# light and an explicit dark the toggle cycles through. +[[project.theme.palette]] +media = "(prefers-color-scheme)" +toggle.icon = "lucide/monitor-cog" +toggle.name = "Follow the system theme" + +[[project.theme.palette]] +media = "(prefers-color-scheme: light)" +scheme = "default" +toggle.icon = "lucide/sun" +toggle.name = "Switch to dark mode" + +[[project.theme.palette]] +media = "(prefers-color-scheme: dark)" +scheme = "slate" +toggle.icon = "lucide/moon" +toggle.name = "Switch to light mode" + +# Markdown extensions: admonitions, grids (attr_list + md_in_html), code blocks +# with highlighting and copy, content tabs, collapsible details, mermaid. +[project.markdown_extensions.abbr] +[project.markdown_extensions.admonition] +[project.markdown_extensions.attr_list] +[project.markdown_extensions.def_list] +[project.markdown_extensions.footnotes] +[project.markdown_extensions.md_in_html] +[project.markdown_extensions.toc] +permalink = true +[project.markdown_extensions.pymdownx.details] +[project.markdown_extensions.pymdownx.highlight] +anchor_linenums = true +[project.markdown_extensions.pymdownx.inlinehilite] +[project.markdown_extensions.pymdownx.superfences] +custom_fences = [ + { name = "mermaid", class = "mermaid", format = "pymdownx.superfences.fence_code_format" } +] +[project.markdown_extensions.pymdownx.tabbed] +alternate_style = true