Follow a record into its fields, name the metrics, name the version
Docs / docs (push) Successful in 38s
Playwright Tests / test-playwright (1, 2) (push) Successful in 2m56s
Playwright Tests / test-playwright (2, 2) (push) Successful in 2m7s
pre-commit / pre-commit (push) Failing after 2m17s
Test Backend / test-backend (push) Successful in 2m54s
Compose Smoke Test / test-compose (push) Successful in 44s
Playwright Tests / merge-reports (push) Successful in 1m17s
Docs / docs (push) Successful in 38s
Playwright Tests / test-playwright (1, 2) (push) Successful in 2m56s
Playwright Tests / test-playwright (2, 2) (push) Successful in 2m7s
pre-commit / pre-commit (push) Failing after 2m17s
Test Backend / test-backend (push) Successful in 2m54s
Compose Smoke Test / test-compose (push) Successful in 44s
Playwright Tests / merge-reports (push) Successful in 1m17s
Three things the first export pass got wrong for a real study. **Dotted paths.** A node returns a record, not a scalar — the numbers arrive inside `final_metrics` — so `--metrics final_metrics.train_loss` yielded an empty column and `--metrics final_metrics` yielded the whole record in one cell. Both sides of the wide table now take dotted paths, and the defaults reach the same depth: every number a result carries is a column named by its path, and inputs are compared leaf by leaf, so two configurations differing in one field give that field as the axis rather than two blobs that are merely not equal. Lists stay whole — a curve belongs in the long table. **`--list`.** Metric names are flow-qualified, so `--name train_loss` matched nothing and said only that. `fluksio export metrics --list` prints the names the selection carries, and an empty export made with `--name` points at it. **A version to compare.** The CLI ships ahead of the engine and a stale one answered a flat 404 with nothing anywhere in the API to tell how old it was. The engine reports `version` on `/observability/summary`, `fluksio status` prints it, and a 404 from export now names both versions — or says "older" when the field itself predates the engine. Bumped to 0.1.5, which is what makes the number worth reading. Also formats `flow/metrics.py`, which had been committed unformatted and was the last `ruff format --check` failure. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01A9Hdrmf2cwNABCnE5x9UJa
This commit is contained in:
@@ -17,6 +17,7 @@ from sqlalchemy import ColumnElement, Integer, cast, func
|
||||
from sqlalchemy import select as sa_select
|
||||
from sqlmodel import col, select
|
||||
|
||||
from fluksio import __version__
|
||||
from fluksio.api.deps import FlowControllerDep, SessionDep, get_current_user
|
||||
from fluksio.api.routes.runs import elapsed_ms
|
||||
from fluksio.core.config import settings
|
||||
@@ -51,6 +52,11 @@ class HealthSummary(BaseModel):
|
||||
nodes: dict[str, int]
|
||||
queue: dict[str, Any]
|
||||
loop_lag: dict[str, float]
|
||||
#: What this engine is running. A client ships ahead of the engine it
|
||||
#: talks to — a `pip install -U` upgrades one and not the other — and a
|
||||
#: route the client knows and the engine does not answers a flat 404. This
|
||||
#: is what turns that into a sentence. Absent means older than this field.
|
||||
version: str = ""
|
||||
|
||||
|
||||
class SeriesPoint(BaseModel):
|
||||
@@ -220,6 +226,7 @@ async def read_summary(request: Request, controller: FlowControllerDep) -> Any:
|
||||
if watchdog is not None
|
||||
else {"ewma": 0.0, "max_60s": 0.0}
|
||||
),
|
||||
version=__version__,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -391,30 +391,62 @@ def _cell(value: Any) -> Any:
|
||||
return json.dumps(value)
|
||||
|
||||
|
||||
def _leaves(value: Any, prefix: str = "") -> Iterator[tuple[str, Any]]:
|
||||
"""Everything a record holds, by its dotted path.
|
||||
|
||||
A node returning a record rather than a scalar is the ordinary shape —
|
||||
the numbers arrive inside `final_metrics` — and a record in one cell is
|
||||
not a column anybody can compare. Lists are left whole: a curve belongs in
|
||||
the long table, not in a cell of this one.
|
||||
"""
|
||||
if isinstance(value, dict):
|
||||
for key, inner in value.items():
|
||||
yield from _leaves(inner, f"{prefix}.{key}" if prefix else str(key))
|
||||
else:
|
||||
yield prefix, value
|
||||
|
||||
|
||||
def _dig(record: dict[str, Any], path: str) -> Any:
|
||||
"""A dotted path into a record: `final_metrics.train_loss`.
|
||||
|
||||
A key with a dot in its own name is not reachable this way, which is the
|
||||
price of the spelling.
|
||||
"""
|
||||
value: Any = record
|
||||
for part in path.split("."):
|
||||
if not isinstance(value, dict) or part not in value:
|
||||
return None
|
||||
value = value[part]
|
||||
return value
|
||||
|
||||
|
||||
def _varying(runs: list[Run]) -> list[str]:
|
||||
"""The inputs that differ across these runs — the axis of a sweep.
|
||||
|
||||
What a reader comparing arms wants as columns. Under two runs nothing can
|
||||
differ, and a table of one run with none of its inputs in it is not worth
|
||||
reading, so all of them are kept.
|
||||
What a reader comparing arms wants as columns, compared leaf by leaf: two
|
||||
configurations differing in one field give that field as a column rather
|
||||
than two blobs that are not the same. Under two runs nothing can differ,
|
||||
and a table of one run with none of its inputs in it is not worth reading,
|
||||
so all of them are kept.
|
||||
"""
|
||||
keys = sorted({key for run in runs for key in run.params})
|
||||
keys = sorted({path for run in runs for path, _ in _leaves(run.params)})
|
||||
if len(runs) < 2:
|
||||
return keys
|
||||
return [
|
||||
key
|
||||
for key in keys
|
||||
if len({json.dumps(run.params.get(key), sort_keys=True) for run in runs}) > 1
|
||||
if len({json.dumps(_dig(run.params, key), sort_keys=True) for run in runs}) > 1
|
||||
]
|
||||
|
||||
|
||||
def _scored(runs: list[Run]) -> list[str]:
|
||||
"""A run's final numbers: every scalar its declared outputs carry."""
|
||||
"""A run's final numbers: every number its declared outputs carry, however
|
||||
deep it sits. A flag is not a number, and neither is a label."""
|
||||
return sorted(
|
||||
{
|
||||
key
|
||||
path
|
||||
for run in runs
|
||||
for key, value in run.result.items()
|
||||
for path, value in _leaves(run.result)
|
||||
if isinstance(value, (int, float)) and not isinstance(value, bool)
|
||||
}
|
||||
)
|
||||
@@ -526,6 +558,11 @@ def export_runs(
|
||||
by default the ones that vary across the selection, which is the sweep
|
||||
axis; ``params`` names them instead. ``metrics`` narrows the final numbers
|
||||
to a few of a run's declared outputs.
|
||||
|
||||
Both take dotted paths into a record a node returned:
|
||||
``metrics=final_metrics.train_loss,test_metrics.known.perfect`` selects
|
||||
three fields rather than two blobs, and the defaults reach the same
|
||||
depth.
|
||||
"""
|
||||
runs = _selected(session, flow, status, group, ids, since, until)
|
||||
inputs = [part for part in params.split(",") if part] or _varying(runs)
|
||||
@@ -539,8 +576,8 @@ def export_runs(
|
||||
def chunks() -> Iterator[list[dict[str, Any]]]:
|
||||
for run in runs:
|
||||
row = {column: _cell(getattr(run, column)) for column in RUN_COLUMNS}
|
||||
row.update((f"param.{k}", _cell(run.params.get(k))) for k in inputs)
|
||||
row.update((f"metric.{k}", _cell(run.result.get(k))) for k in scores)
|
||||
row.update((f"param.{k}", _cell(_dig(run.params, k))) for k in inputs)
|
||||
row.update((f"metric.{k}", _cell(_dig(run.result, k))) for k in scores)
|
||||
yield [row]
|
||||
|
||||
return _stream(format, "runs", columns, chunks())
|
||||
|
||||
@@ -270,9 +270,7 @@ class MetricsCollector:
|
||||
)
|
||||
)
|
||||
|
||||
def _finish_run(
|
||||
self, run: dict[str, Any], ts: float, status: str = ""
|
||||
) -> None:
|
||||
def _finish_run(self, run: dict[str, Any], ts: float, status: str = "") -> None:
|
||||
"""Close an open record. A run says how it ended; a cascade is told."""
|
||||
run["finished_at"] = datetime.fromtimestamp(ts, UTC)
|
||||
run["duration_ms"] = round((ts - run["started_ts"]) * 1000, 2)
|
||||
|
||||
@@ -12,6 +12,7 @@ from fastapi.routing import APIRoute
|
||||
from fluksio_worker.worker_main import ARTIFACT_DIR_ENV
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
|
||||
from fluksio import __version__
|
||||
from fluksio.api.main import api_router
|
||||
from fluksio.api.routes.alerts import read_config as read_alerts_config
|
||||
from fluksio.cloud import config as cloud_config
|
||||
@@ -283,6 +284,7 @@ _docs_enabled = settings.ENVIRONMENT != "production"
|
||||
|
||||
app = FastAPI(
|
||||
title=settings.PROJECT_NAME,
|
||||
version=__version__,
|
||||
openapi_url=f"{settings.API_V1_STR}/openapi.json" if _docs_enabled else None,
|
||||
docs_url="/docs" if _docs_enabled else None,
|
||||
redoc_url="/redoc" if _docs_enabled else None,
|
||||
|
||||
+111
-31
@@ -16,13 +16,14 @@ import json
|
||||
import pkgutil
|
||||
import sys
|
||||
import time
|
||||
from collections.abc import Iterator
|
||||
from collections.abc import Callable, Iterator
|
||||
from contextlib import contextmanager, nullcontext
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from fluksio import __version__
|
||||
from fluksio.sdk import FLOWS, Flow, SyncError
|
||||
from fluksio.sdk.client import (
|
||||
GLOBAL_DATA_DIR,
|
||||
@@ -634,6 +635,10 @@ def _status_screen(client: Client) -> Any:
|
||||
head.append(" " + " · ".join(str(p) for p in problems), style="yellow")
|
||||
phrase, style = _portal_phrase(portal)
|
||||
head.append(" " + phrase, style=style)
|
||||
# What it is running, since a client is upgraded on its own and a route
|
||||
# this one knows may not be there. Absent from an engine older than the
|
||||
# field itself, which is the answer in its own way.
|
||||
head.append(f" v{summary.get('version') or '?'}", style="dim")
|
||||
|
||||
counts = summary.get("flows") or {}
|
||||
nodes = summary.get("nodes") or {}
|
||||
@@ -873,7 +878,7 @@ def _selection(args: argparse.Namespace) -> dict[str, Any]:
|
||||
}
|
||||
|
||||
|
||||
def _write_rows(rows: list[dict[str, Any]], fmt: str, out: str) -> int:
|
||||
def _write_rows(rows: list[dict[str, Any]], fmt: str, out: str, hint: str = "") -> int:
|
||||
"""The exported rows, in the format asked for, to a file or to stdout.
|
||||
|
||||
The engine settles the columns over the whole selection before it sends
|
||||
@@ -881,7 +886,8 @@ def _write_rows(rows: list[dict[str, Any]], fmt: str, out: str) -> int:
|
||||
first row's are the header.
|
||||
"""
|
||||
if not rows:
|
||||
print("fluksio: nothing matched, so nothing was written", file=sys.stderr)
|
||||
empty = "fluksio: nothing matched, so nothing was written"
|
||||
print(empty + (f". {hint}" if hint else ""), file=sys.stderr)
|
||||
return 0
|
||||
if fmt == "parquet":
|
||||
try:
|
||||
@@ -904,44 +910,106 @@ def _write_rows(rows: list[dict[str, Any]], fmt: str, out: str) -> int:
|
||||
return 0
|
||||
|
||||
|
||||
def cmd_export_metrics(args: argparse.Namespace) -> int:
|
||||
"""Every selected run's numbers as one long table."""
|
||||
#: How many runs `--list` reads before giving up on finding a metric name.
|
||||
#: The names belong to the flow's nodes rather than to a run, so the newest
|
||||
#: one that recorded any is the whole vocabulary — the rest are for a
|
||||
#: selection whose newest runs failed before they measured anything.
|
||||
# ponytail: the first run with names wins; a name only an older run recorded
|
||||
# is not listed. A `distinct` over the selection would be exact and is a route
|
||||
# of its own.
|
||||
LIST_SCAN = 10
|
||||
|
||||
|
||||
def _list_names(client: Client, args: argparse.Namespace) -> int:
|
||||
"""The metric names these runs carry, since a name is flow-qualified.
|
||||
|
||||
`train_loss` is recorded as `train.train_loss`, and asking for the bare
|
||||
one matches nothing — so this is the answer to "what would match".
|
||||
"""
|
||||
filters = _selection(args)
|
||||
if "until" in filters:
|
||||
# The history spells the same bound `before`, where it is also the
|
||||
# cursor a page is taken from.
|
||||
filters["before"] = filters.pop("until")
|
||||
ids = args.run or [
|
||||
row["id"] for row in client.runs(flow=args.flow, limit=LIST_SCAN, **filters)
|
||||
]
|
||||
for run_id in ids[:LIST_SCAN]:
|
||||
names = sorted({point["name"] for point in client.metrics(run_id)})
|
||||
if names:
|
||||
_say("\n".join(names))
|
||||
return 0
|
||||
_say("No metrics recorded by these runs.")
|
||||
return 0
|
||||
|
||||
|
||||
def _too_old(client: Client) -> str:
|
||||
"""A route this client knows and the engine does not."""
|
||||
try:
|
||||
version = (client.summary() or {}).get("version") or ""
|
||||
except (ApiError, httpx.HTTPError):
|
||||
version = ""
|
||||
engine = f"the engine is {version}" if version else "the engine is older"
|
||||
return (
|
||||
f"this engine has no export endpoints — {engine} and this client is "
|
||||
f"{__version__}. Upgrade it: `pip install -U fluksio`"
|
||||
)
|
||||
|
||||
|
||||
def _export(
|
||||
args: argparse.Namespace, fetch: Callable[[Client], list[dict[str, Any]]]
|
||||
) -> int:
|
||||
"""Both exports: fetch what was asked for, then write it."""
|
||||
if args.format == "parquet" and not args.out:
|
||||
return _fail("--format parquet writes a file; name it with -o FILE")
|
||||
hint = (
|
||||
"`--list` names the metrics these runs carry"
|
||||
if getattr(args, "name", "")
|
||||
else ""
|
||||
)
|
||||
try:
|
||||
with _client_for(args, retries=0) as client:
|
||||
rows = client.export_metrics(
|
||||
flow=args.flow,
|
||||
ids=args.run,
|
||||
name=args.name,
|
||||
stride=args.stride,
|
||||
**_selection(args),
|
||||
)
|
||||
if getattr(args, "list_names", False):
|
||||
return _list_names(client, args)
|
||||
try:
|
||||
rows = fetch(client)
|
||||
except ApiError as exc:
|
||||
if exc.status != 404:
|
||||
raise
|
||||
return _fail(_too_old(client))
|
||||
except (SyncError, ApiError) as exc:
|
||||
return _fail(str(exc))
|
||||
except httpx.HTTPError as exc:
|
||||
return _unreachable(exc)
|
||||
return _write_rows(rows, args.format, args.out)
|
||||
return _write_rows(rows, args.format, args.out, hint)
|
||||
|
||||
|
||||
def cmd_export_metrics(args: argparse.Namespace) -> int:
|
||||
"""Every selected run's numbers as one long table."""
|
||||
return _export(
|
||||
args,
|
||||
lambda client: client.export_metrics(
|
||||
flow=args.flow,
|
||||
ids=args.run,
|
||||
name=args.name,
|
||||
stride=args.stride,
|
||||
**_selection(args),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def cmd_export_runs(args: argparse.Namespace) -> int:
|
||||
"""One row per run: its inputs, its final numbers, what it ran."""
|
||||
if args.format == "parquet" and not args.out:
|
||||
return _fail("--format parquet writes a file; name it with -o FILE")
|
||||
try:
|
||||
with _client_for(args, retries=0) as client:
|
||||
rows = client.export_runs(
|
||||
flow=args.flow,
|
||||
ids=args.run,
|
||||
params=args.params,
|
||||
metrics=args.metrics,
|
||||
**_selection(args),
|
||||
)
|
||||
except (SyncError, ApiError) as exc:
|
||||
return _fail(str(exc))
|
||||
except httpx.HTTPError as exc:
|
||||
return _unreachable(exc)
|
||||
return _write_rows(rows, args.format, args.out)
|
||||
return _export(
|
||||
args,
|
||||
lambda client: client.export_runs(
|
||||
flow=args.flow,
|
||||
ids=args.run,
|
||||
params=args.params,
|
||||
metrics=args.metrics,
|
||||
**_selection(args),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1148,6 +1216,12 @@ def add_parsers(subparsers: Any) -> None:
|
||||
default=1,
|
||||
help="keep every Nth point of each curve",
|
||||
)
|
||||
sub.add_argument(
|
||||
"--list",
|
||||
dest="list_names",
|
||||
action="store_true",
|
||||
help="print the metric names these runs carry, and stop",
|
||||
)
|
||||
sub.set_defaults(func=cmd_export_metrics)
|
||||
|
||||
sub = exports.add_parser(
|
||||
@@ -1158,9 +1232,15 @@ def add_parsers(subparsers: Any) -> None:
|
||||
"--params",
|
||||
default="",
|
||||
metavar="A,B",
|
||||
help="the inputs to put in columns (default: the ones that vary)",
|
||||
help=(
|
||||
"the inputs to put in columns, dotted into a record "
|
||||
"(default: the ones that vary)"
|
||||
),
|
||||
)
|
||||
sub.add_argument(
|
||||
"--metrics", default="", metavar="A,B", help="the final numbers to keep"
|
||||
"--metrics",
|
||||
default="",
|
||||
metavar="A,B",
|
||||
help="the final numbers to keep, e.g. final_metrics.train_loss",
|
||||
)
|
||||
sub.set_defaults(func=cmd_export_runs)
|
||||
|
||||
Reference in New Issue
Block a user