Do not fail a node because the engine's own stdout is gone

The log tee wrote through to the real stream unguarded, and the worker
pool tees a returned call's logs there after reading its result and
before handing it back — so a dead stdout, which `fluksio serve` makes
possible by running the engine as a child of the dashboard holding that
pipe, failed the node with its outputs already in hand. The capture half
runs first, so swallowing the write loses nothing.

Also: `flow_events` catches the RuntimeError a peer leaving mid-send
raises, which is a disconnect by another route, and the remote agent no
longer raises out of the task when its subprocess died before it could
be written to — the read below reports that and ends the call.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01TXQv6KNyyvY7Z1etYTUUAd
This commit is contained in:
2026-08-31 07:52:33 +02:00
co-authored by Claude Opus 5
parent 9a5371d4c7
commit c09095d369
4 changed files with 40 additions and 4 deletions
+4 -1
View File
@@ -1026,7 +1026,10 @@ async def flow_events(websocket: WebSocket, token: str = "") -> None:
out.append(event)
if out:
await _send(websocket, out)
except WebSocketDisconnect:
except (WebSocketDisconnect, RuntimeError):
# A peer that goes away mid-send takes the RuntimeError route
# ("websocket.send after websocket.close") rather than the clean
# disconnect. Either way the socket is gone and the loop is over.
pass
finally:
receiver.cancel()
+12 -2
View File
@@ -34,10 +34,20 @@ class _Tee(io.TextIOBase):
sink = _sink.get()
if sink is not None and text:
sink(text)
return self._real.write(text)
try:
return self._real.write(text)
except OSError:
# A dead stdout — `fluksio serve` runs the engine as a child of the
# dashboard, which holds the far end of that pipe — must not fail
# the node whose output was being teed. The capture above has it,
# and there is nowhere left to report the loss to anyway.
return len(text)
def flush(self) -> None:
self._real.flush()
try:
self._real.flush()
except OSError:
pass
def isatty(self) -> bool:
return self._real.isatty()
+19
View File
@@ -122,6 +122,25 @@ def test_printing_outside_a_node_still_reaches_the_real_stream(capsys):
assert "server talking" in capsys.readouterr().out
def test_a_dead_real_stream_does_not_fail_the_node_being_teed():
"""`fluksio serve` runs the engine as a child of the dashboard."""
class Broken:
def write(self, text: str) -> int:
raise BrokenPipeError(32, "Broken pipe")
def flush(self) -> None:
raise BrokenPipeError(32, "Broken pipe")
collected: list[str] = []
tee = logs._Tee(Broken())
with logs.capture(collected.append):
assert tee.write("still teed") == len("still teed")
tee.flush()
assert collected == ["still teed"]
def test_installing_twice_does_not_stack_tees():
logs.install()
once = sys.stdout
+5 -1
View File
@@ -257,7 +257,11 @@ class Agent:
heartbeat = asyncio.create_task(beat())
try:
await loop.run_in_executor(None, worker.send, request)
# A subprocess that died before it could be written to is reported
# by the read below, which says so and ends the call — rather than
# raising here and leaving the engine waiting out its silence.
with contextlib.suppress(OSError):
await loop.run_in_executor(None, worker.send, request)
while True:
line = await loop.run_in_executor(None, worker.read_line)
if not line: