Files
ai-agent/sidecar/launch.py
Gabriel Vidal cf01bc00f4 refactor(sidecar): split sidecar.py into focused modules
config / pids / validate / claude_accounts / launch / fork / routes_runs /
routes_accounts, mounted by a thin sidecar.py (same `sidecar:app`, same 24
routes, same startup hook). Mutable state keeps one owner module
(pids._procs, claude_accounts._account_status_cache). Includes the
_claude_args builder from 1de9426 and its test; the Dockerfile's explicit
COPY list gains the new modules.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-06 23:55:47 +02:00

176 lines
7.1 KiB
Python

"""
Launching and stopping a run: the detached ``claude -p`` / runner process, its
pid record and log, the launch probe, and the claude-only argv extras
(``--settings``, ``--remote-control``).
"""
import json
import os
import pathlib
import shlex
import signal
import subprocess
import sys
import time
from fastapi import HTTPException
import claude_cli
import pids
from claude_accounts import _run_env
from config import (CLAUDE_BIN, DEFAULT_ACCOUNT, DEFAULT_CWD, DEFAULT_MODEL,
DISALLOWED_TOOLS, LAUNCH_PROBE_S, LOG_DIR, PERMISSION_MODE,
REMOTE_CONTROL, STOP_GUARD, STOP_GUARD_ENABLED,
STOP_GUARD_TIMEOUT_S, STOP_WAIT_S)
def _claude_settings() -> dict:
"""The settings every claude run is launched with (``--settings`` merges
them over the user's own): the inbox opt-in that lets `/message` reach a
live run, and the Stop guard hook (see the constants above)."""
s: dict = {"crossSessionInbound": "accept"}
if STOP_GUARD_ENABLED and STOP_GUARD.is_file():
cmd = f"{shlex.quote(sys.executable)} {shlex.quote(str(STOP_GUARD))}"
s["hooks"] = {"Stop": [{"hooks": [{
"type": "command", "command": cmd,
"timeout": STOP_GUARD_TIMEOUT_S}]}]}
return s
def _claude_settings_args() -> list[str]:
return ["--settings", json.dumps(_claude_settings(), separators=(",", ":"))]
def _claude_args(prompt: str, flag: str, sid: str) -> list[str]:
"""What every claude run shares — the prompt, how it is keyed to its
session (``--session-id`` for a spawn, ``--resume`` for a continue or a
fork), the headless output + permission flags, the disallowed tools and
the ``--settings`` payload (stop guard, inbox opt-in). One builder so no
launch path can miss one of them."""
args = [CLAUDE_BIN, "-p", prompt, flag, sid,
"--output-format", "json",
"--permission-mode", PERMISSION_MODE]
if DISALLOWED_TOOLS:
args += ["--disallowed-tools", " ".join(DISALLOWED_TOOLS)]
return args + _claude_settings_args()
def _remote_control(account: str) -> list[str]:
"""`--remote-control` for default-account runs only: a work run would list
itself in the employer org's claude.ai session list."""
return ["--remote-control"] if REMOTE_CONTROL and account == DEFAULT_ACCOUNT \
else []
def _run_cwd(cwd: str | None) -> str:
"""The run's working dir (``~`` expanded — the backend sends an account's
default cwd host-relative), which must exist."""
cwd = os.path.expanduser(cwd) if cwd else DEFAULT_CWD
if not pathlib.Path(cwd).is_dir():
raise HTTPException(400, f"cwd does not exist: {cwd}")
return cwd
def _launch(args: list[str], sid: str, cwd: str, *,
model: str | None = None, append: bool = False,
account: str | None = None) -> dict:
"""Fire a detached ``claude -p`` run (a daemon) and persist its pid record.
The child leads its own session/process group (``start_new_session``) so the
run survives past this request *and* past the sidecar itself; the systemd
unit's ``KillMode=process`` keeps it alive across a sidecar restart. We
persist its identity (pid + ``/proc`` start-time) to ``logs/<sid>.pid`` so a
restarted sidecar can re-discover, list and ``/interrupt`` it, and never
signals a recycled PID. ``append`` keeps a resumed run's output alongside the
original spawn log instead of clobbering it.
The launch is *checked*, not fire-and-forget: we watch the child for
``LAUNCH_PROBE_S`` and raise 502 with the tail of its log if it exits
non-zero in that window (``No conversation found with session ID: …`` when a
``--resume`` can't find its transcript, a broken CLI, …). The caller then
sees a real error instead of a run that never produces a turn."""
LOG_DIR.mkdir(parents=True, exist_ok=True)
log_path = LOG_DIR / f"{sid}.log"
err_from = log_path.stat().st_size if append and log_path.exists() else 0
with open(log_path, "ab" if append else "wb") as log:
proc = subprocess.Popen(
args, cwd=cwd, stdin=subprocess.DEVNULL, stdout=log, stderr=log,
start_new_session=True,
env=_run_env(sid, account),
)
pids._procs[sid] = proc # kept so we reap it when it exits (see _reap)
# The child leads its own process group (start_new_session), so pid == pgid.
try:
pids._pidfile(sid).write_text(json.dumps({
"pid": proc.pid,
"startedAt": int(time.time()),
"starttime": pids._proc_starttime(proc.pid),
"model": model or DEFAULT_MODEL,
"cwd": cwd,
"account": account,
}))
except OSError:
pass
rc = _probe_launch(proc)
if rc is not None and rc != 0:
pids._procs.pop(sid, None)
pids._pidfile(sid).unlink(missing_ok=True)
msg, kind = claude_cli.run_failure(_log_text(log_path, err_from))
raise HTTPException(502, claude_cli.error_detail(
msg, kind=kind, account=account, exitCode=rc))
return {"sessionId": sid, "pid": proc.pid, "log": str(log_path)}
def _probe_launch(proc: subprocess.Popen) -> int | None:
"""Poll a just-started run for LAUNCH_PROBE_S. Its exit code if it died in
that window (0 = a legitimately quick, successful run), else None."""
deadline = time.monotonic() + LAUNCH_PROBE_S
while time.monotonic() < deadline:
rc = proc.poll()
if rc is not None:
return rc
time.sleep(0.1)
return None
def _stop_run(sid: str, rec: dict) -> None:
"""Stop a live run so a forced resume can take over its session.
SIGINT the process group (like the /interrupt endpoint — claude writes its
``[Request interrupted by user]`` marker and exits), wait up to STOP_WAIT_S
for it to actually die, and escalate to SIGKILL if it won't. Launching the
``--resume`` while the old run still lives would append duplicate turns to
the same transcript, so a run that survives even SIGKILL is a hard error."""
pid = rec["pid"]
try:
os.killpg(pid, signal.SIGINT)
except ProcessLookupError:
pass # died between the liveness check and the signal — that's fine
except OSError as e:
raise HTTPException(500, f"could not signal process: {e}")
deadline = time.monotonic() + STOP_WAIT_S
while time.monotonic() < deadline and pids._alive(rec):
time.sleep(0.15)
if pids._alive(rec):
try:
os.killpg(pid, signal.SIGKILL)
except OSError:
pass
deadline = time.monotonic() + 2.0
while time.monotonic() < deadline and pids._alive(rec):
time.sleep(0.15)
if pids._alive(rec):
raise HTTPException(409, "could not stop the running session")
pids._pidfile(sid).unlink(missing_ok=True)
def _log_text(log_path: pathlib.Path, since: int = 0) -> str:
"""What this run wrote to its log (from ``since``) — parsed by
``claude_cli.run_failure`` into the message to show the user."""
try:
with open(log_path, "rb") as f:
f.seek(since)
return f.read().decode(errors="replace")
except OSError:
return ""