Files
ai-agent/sidecar/feed.py
Gabriel Vidal 3a0ef08dcf feat(worker): sidecar runs as a paired macOS worker — feed, pairing, launchd installer
- liveness falls back to psutil where there is no /proc (macOS)
- GET /feed + /feed/file: the hub pulls transcripts it has no mount for
- POST /pair trades a one-time code for the worker's bearer; /unpair drops it
- /health reports worker identity + permission mode
- worker/install-macos.sh: venv, launchd agent bound to the Tailscale IP,
  prints the pairing string (addr/code/claude login)
- Dockerfile copies every sidecar module (claude_cli was missing, so the
  in-container runner could not import)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-09-28 13:49:07 +02:00

127 lines
4.8 KiB
Python

"""
The transcript feed — how a hub mirrors a *remote* runner's transcripts.
On the homelab the backend reads the host's transcripts through a read-only
mount. A worker on another machine (the Orus MacBook) has no such mount: the
hub **pulls** instead, over the tailnet, with two bearer-guarded GETs:
GET /feed?since=<epoch> every transcript (``*.jsonl``, subagents
included) and subagent ``*.meta.json`` under the
default account's ``projects/`` whose mtime is
≥ ``since`` — ``{rel, mtime, size}`` each
GET /feed/file?rel=&offset= the file's bytes from ``offset`` (at most
``limit``), with its current size and mtime in
``X-Feed-Size`` / ``X-Feed-Mtime``
That is the whole protocol: the hub appends what grew and re-downloads a file
that shrank (fork surgery rewrites one). The worker never calls the hub — it
doesn't know where the hub is, and doesn't need to.
``rel`` is always relative to the projects dir and must name a ``.jsonl`` or
``.meta.json`` inside it; anything else (``..``, an absolute path, a symlink
out, another suffix) is a 404, so the feed can't be turned into a file reader
for the rest of the laptop.
"""
from __future__ import annotations
import os
import pathlib
import time
from typing import Callable
from fastapi import APIRouter, Header, HTTPException, Query, Response
SUFFIXES = (".jsonl", ".meta.json")
# One response's worth of bytes. The hub loops until it has caught up, so this
# only bounds memory on both ends (a long session's transcript is tens of MB).
MAX_CHUNK = 8 * 1024 * 1024
def list_files(root: pathlib.Path, since: float = 0.0) -> list[dict]:
"""Every feed file under ``root`` with mtime ≥ ``since`` (scandir walk —
the hub polls this every couple of seconds)."""
out: list[dict] = []
if not root.is_dir():
return out
base = str(root)
stack = [base]
while stack:
d = stack.pop()
try:
it = os.scandir(d)
except OSError:
continue
with it:
for e in it:
try:
if e.is_dir(follow_symlinks=False):
stack.append(e.path)
continue
if not e.name.endswith(SUFFIXES) or \
not e.is_file(follow_symlinks=False):
continue
st = e.stat(follow_symlinks=False)
except OSError:
continue
if st.st_mtime >= since:
out.append({"rel": os.path.relpath(e.path, base)
.replace(os.sep, "/"),
"mtime": st.st_mtime, "size": st.st_size})
return out
def resolve(root: pathlib.Path, rel: str) -> pathlib.Path:
"""``rel`` → a file inside ``root``, or 404 (traversal, wrong suffix,
missing)."""
rel = (rel or "").strip()
if not rel or rel.startswith("/") or "\\" in rel or "\0" in rel \
or not rel.endswith(SUFFIXES) \
or any(p in ("", ".", "..") for p in rel.split("/")):
raise HTTPException(404, "not a feed file")
try:
real_root = root.resolve()
p = (root / rel).resolve()
p.relative_to(real_root)
except (OSError, ValueError):
raise HTTPException(404, "not a feed file")
if not p.is_file():
raise HTTPException(404, "not a feed file")
return p
def read_chunk(p: pathlib.Path, offset: int, limit: int) -> tuple[bytes, os.stat_result]:
with open(p, "rb") as f:
st = os.fstat(f.fileno())
f.seek(max(0, offset))
return f.read(max(0, min(limit, MAX_CHUNK))), st
def make_router(auth: Callable[[str | None], None],
projects_dir: Callable[[], pathlib.Path]) -> APIRouter:
r = APIRouter()
@r.get("/feed")
def feed(since: float = Query(0.0),
authorization: str | None = Header(default=None)) -> dict:
auth(authorization)
now = time.time() # taken first: a file touched mid-walk is re-listed
root = projects_dir()
return {"now": now, "root": str(root),
"files": list_files(root, since)}
@r.get("/feed/file")
def feed_file(rel: str = Query(...), offset: int = Query(0, ge=0),
limit: int = Query(MAX_CHUNK, ge=1),
authorization: str | None = Header(default=None)) -> Response:
auth(authorization)
p = resolve(projects_dir(), rel)
try:
data, st = read_chunk(p, offset, limit)
except OSError:
raise HTTPException(404, "not a feed file")
return Response(data, media_type="application/octet-stream", headers={
"X-Feed-Size": str(st.st_size), "X-Feed-Mtime": repr(st.st_mtime)})
return r