"""Artifacts over HTTP: the one way bytes get in and out of the store. A node on this host could reach the directory itself, but a node on a remote worker cannot — and having one path rather than two is what keeps a flow's code the same wherever it runs. """ import re import tempfile from pathlib import Path from typing import Any from fastapi import APIRouter, Depends, HTTPException, Query, Request from fastapi.responses import FileResponse from jwt.exceptions import InvalidTokenError from pydantic import BaseModel from sqlmodel import Session from starlette.concurrency import run_in_threadpool from fluksio.api.deps import user_from_token from fluksio.core import security from fluksio.core.config import settings from fluksio.core.db import engine from fluksio.flow.artifacts import ArtifactStore #: What may be echoed back as a response content type. The caller holds the #: reference and passes its media type, so this guards a header rather than #: trusting one — anything else is served as bytes. _MEDIA_TYPE = re.compile(r"^[\w.+-]+/[\w.+-]+$") def artifact_caller(request: Request) -> str: """Who may move artifacts: a signed-in person, or an attached worker. A worker's node stores its checkpoints through this endpoint, so its own credential has to open it — and only it. The token is no use anywhere else in the API, which is why this check is here rather than in the shared dependency every other route uses. """ header = request.headers.get("Authorization", "") token = header[7:] if header.lower().startswith("bearer ") else "" if not token: raise HTTPException(status_code=401, detail="Not authenticated") try: claims = security.decode_worker_token(token) except InvalidTokenError: pass else: return f"worker:{claims.get('sub')}" with Session(engine) as session: # With the request, so a credential that is scoped by route — a wall # panel's — is judged against this one rather than waved through. user = user_from_token(session, token, request) if user is None: raise HTTPException(status_code=401, detail="Not authenticated") return user.email router = APIRouter( prefix="/artifacts", tags=["artifacts"], dependencies=[Depends(artifact_caller)] ) class ArtifactRef(BaseModel): digest: str size: int media_type: str name: str = "" def _store(request: Request) -> ArtifactStore: store: ArtifactStore | None = getattr(request.app.state, "artifact_store", None) if store is None: raise HTTPException(status_code=503, detail="The artifact store is not ready") return store @router.put("", response_model=ArtifactRef) async def put_artifact( request: Request, name: str = Query(default=""), media_type: str = Query(default=""), ) -> Any: """Store the request body and answer with the reference to it. Spooled to disk as it arrives rather than buffered: a video segment is as legitimate a body here as a checkpoint, and neither should have to fit in memory twice. Capped, because nothing else here was: any account, and any worker credential, could otherwise fill the data volume. """ store = _store(request) cap = settings.MAX_ARTIFACT_BYTES declared = request.headers.get("content-length") if cap and declared and declared.isdigit() and int(declared) > cap: raise HTTPException( status_code=413, detail=f"An artifact may be at most {cap} bytes" ) handle = tempfile.NamedTemporaryFile(dir=store.root, delete=False) written = 0 try: with handle: async for chunk in request.stream(): written += len(chunk) # A chunked body declares no length, so the stream is what # actually holds the limit. if cap and written > cap: raise HTTPException( status_code=413, detail=f"An artifact may be at most {cap} bytes", ) # Off the event loop: this is a write syscall per chunk, for # as long as the upload lasts. await run_in_threadpool(handle.write, chunk) return await run_in_threadpool( store.put_file, Path(handle.name), media_type, name ) finally: Path(handle.name).unlink(missing_ok=True) @router.get("/{digest}") def get_artifact( digest: str, request: Request, media_type: str = Query(default="") ) -> Any: """Serve one artifact back. The caller passes the media type off the reference it holds, which is what lets a browser play a clip rather than download it; the store itself keeps only bytes. Ranged requests are answered because an audio or video element scrubbing through a file asks for them. """ store = _store(request) path = store.path(digest) if path is None: raise HTTPException(status_code=404, detail="No such artifact") return FileResponse( path, media_type=( media_type if _MEDIA_TYPE.match(media_type) else "application/octet-stream" ), # The digest is the content, so it is also the perfect validator. headers={"ETag": f'"{digest}"'}, )