Files
ai-agent/backend/files/routes.py
Gabriel Vidal 0ff9e40242 refactor(backend): split main.py into domain packages with app.state injection
main.py (5146 lines, 106 routes) becomes an assembly only: one package per
domain — core, files, settings, dashboards, conversations, diff,
notifications, forms, runs, accounts, workers, models, cron, agents,
projects, services, goals, memories, plans, templates — each exposing an
APIRouter; the flat domain modules move into their package behind a barrel
that keeps the old `import conversations` / `import projects` spellings.

The shared singletons (store, meta_store, hub, indexer, …) are built once by
core.state.build_state() and attached to app.state.ai; routes take them as
the `deps: State` dependency and helpers as an explicit `deps: AppState`.
conversations/pricing.py carries the per-model rates out of the parser.

Verified: route table and OpenAPI byte-identical; 90 read endpoints
golden-diffed against the monolith on a copy of the live data (identical);
write routes smoke-tested; 66 backend tests pass.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-06 23:55:47 +02:00

145 lines
5.5 KiB
Python

"""The file catalog (/api/bundle) and the single-file read/write routes."""
import pathlib
from fastapi import APIRouter, HTTPException
from pydantic import BaseModel
import schemas
from core.config import MD_EXTS, WORKSPACE
from core.http import _r
from core.state import AppState, State
from files.discovery import _is_context_file, _is_editable, _kind, _read_text
from indexer import INPUT_PRICE_PER_MTOK, MODEL, _estimate_tokens, _price
router = APIRouter()
# ── entry assembly ──────────────────────────────────────────────────────────
def _tokens_for(deps: AppState, path: str, text: str) -> tuple[int, float, bool]:
"""Pull the stored real token count/cost; fall back to a cheap estimate."""
row = deps.store.get_file(path)
if row and row["tokens"] is not None:
return row["tokens"], row["cost"] or 0.0, bool(row["estimated"])
est = _estimate_tokens(text)
return est, _price(est), True
def _file_entry(deps: AppState, ap: pathlib.Path, rel: pathlib.PurePosixPath, with_content: bool):
content = _read_text(ap)
is_md = rel.suffix.lower() in MD_EXTS
words = len(content.split())
chars = len(content)
tokens, cost, estimated = _tokens_for(deps, str(rel), content)
try:
bytes_ = ap.stat().st_size
except OSError:
bytes_ = chars
entry = {
"path": str(rel),
"name": rel.name,
"ext": rel.suffix,
"kind": _kind(rel),
"isMarkdown": is_md,
"bytes": bytes_,
"words": words,
"chars": chars,
"tokens": tokens,
"cost": cost,
"estimated": estimated,
"editable": _is_editable(rel),
}
if with_content:
entry["content"] = content
return entry
# ── path safety ─────────────────────────────────────────────────────────────
def _resolve(rel_path: str) -> tuple[pathlib.Path, pathlib.PurePosixPath]:
rel = pathlib.PurePosixPath(rel_path)
if rel.is_absolute() or ".." in rel.parts:
raise HTTPException(400, "invalid path")
ap = (WORKSPACE / rel).resolve()
try:
ap.relative_to(WORKSPACE)
except ValueError:
raise HTTPException(400, "path escapes workspace")
if not _is_context_file(rel):
raise HTTPException(403, "not an exposed context file")
return ap, rel
# ── API ─────────────────────────────────────────────────────────────────────
@router.get("/api/bundle", responses=_r(schemas.Bundle))
def bundle(deps: State):
"""The whole file catalog — metadata only, served from the index DB.
This used to read every context file's content per request (a 16 MB,
12-second response the app shell fetched on every load). File *content* now
loads lazily through ``GET /api/file`` when a file is opened; the words /
chars / token stats come from the ``files`` table, which the indexer keeps
fresh (recomputed only when a file's content actually changes)."""
files = []
for path, r in deps.store.all_files().items():
rel = pathlib.PurePosixPath(path)
tokens = r["tokens"]
estimated = bool(r["estimated"])
if tokens is None:
# not counted yet — estimate off the stored char count (~chars/4)
tokens = ((r["chars"] or 0) + 3) // 4
estimated = True
files.append({
"path": path,
"name": rel.name,
"ext": rel.suffix,
"kind": _kind(rel),
"isMarkdown": bool(r["is_markdown"]),
"bytes": r["size"] or 0,
"words": r["words"] or 0,
"chars": r["chars"] or 0,
"tokens": tokens,
"cost": r["cost"] or _price(tokens),
"estimated": estimated,
"editable": _is_editable(rel),
})
files.sort(key=lambda f: f["path"])
totals = {
"files": len(files),
"mdFiles": sum(1 for f in files if f["isMarkdown"]),
"words": sum(f["words"] for f in files),
"chars": sum(f["chars"] for f in files),
"tokens": sum(f["tokens"] for f in files),
"cost": sum(f["cost"] for f in files),
"estimated": sum(1 for f in files if f["estimated"]),
}
return {"workspace": str(WORKSPACE.name), "totals": totals, "files": files,
"pricing": {"model": MODEL, "inputPerMTok": INPUT_PRICE_PER_MTOK}}
@router.get("/api/file", responses=_r(schemas.FileEntry))
def get_file(deps: State, path: str):
ap, rel = _resolve(path)
if not ap.is_file():
raise HTTPException(404, "not found")
return _file_entry(deps, ap, rel, with_content=True)
class SaveBody(BaseModel):
path: str
content: str
@router.put("/api/file", responses=_r(schemas.FileEntry))
def put_file(deps: State, body: SaveBody):
ap, rel = _resolve(body.path)
if not _is_editable(rel):
raise HTTPException(403, "not editable")
if not ap.exists():
raise HTTPException(404, "not found")
ap.write_text(body.content, encoding="utf-8")
# Refresh on-disk metadata immediately (force the discovery walk so a
# brand-new file lands in the catalog at once); the real token count
# refreshes after the debounce.
deps.indexer.scan_files(force_walk=True)
return _file_entry(deps, ap, rel, with_content=True)