main.py (5146 lines, 106 routes) becomes an assembly only: one package per domain — core, files, settings, dashboards, conversations, diff, notifications, forms, runs, accounts, workers, models, cron, agents, projects, services, goals, memories, plans, templates — each exposing an APIRouter; the flat domain modules move into their package behind a barrel that keeps the old `import conversations` / `import projects` spellings. The shared singletons (store, meta_store, hub, indexer, …) are built once by core.state.build_state() and attached to app.state.ai; routes take them as the `deps: State` dependency and helpers as an explicit `deps: AppState`. conversations/pricing.py carries the per-model rates out of the parser. Verified: route table and OpenAPI byte-identical; 90 read endpoints golden-diffed against the monolith on a copy of the live data (identical); write routes smoke-tested; 66 backend tests pass. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
145 lines
5.5 KiB
Python
145 lines
5.5 KiB
Python
"""The file catalog (/api/bundle) and the single-file read/write routes."""
|
|
|
|
import pathlib
|
|
|
|
from fastapi import APIRouter, HTTPException
|
|
from pydantic import BaseModel
|
|
|
|
import schemas
|
|
from core.config import MD_EXTS, WORKSPACE
|
|
from core.http import _r
|
|
from core.state import AppState, State
|
|
from files.discovery import _is_context_file, _is_editable, _kind, _read_text
|
|
from indexer import INPUT_PRICE_PER_MTOK, MODEL, _estimate_tokens, _price
|
|
|
|
router = APIRouter()
|
|
|
|
|
|
# ── entry assembly ──────────────────────────────────────────────────────────
|
|
def _tokens_for(deps: AppState, path: str, text: str) -> tuple[int, float, bool]:
|
|
"""Pull the stored real token count/cost; fall back to a cheap estimate."""
|
|
row = deps.store.get_file(path)
|
|
if row and row["tokens"] is not None:
|
|
return row["tokens"], row["cost"] or 0.0, bool(row["estimated"])
|
|
est = _estimate_tokens(text)
|
|
return est, _price(est), True
|
|
|
|
|
|
def _file_entry(deps: AppState, ap: pathlib.Path, rel: pathlib.PurePosixPath, with_content: bool):
|
|
content = _read_text(ap)
|
|
is_md = rel.suffix.lower() in MD_EXTS
|
|
words = len(content.split())
|
|
chars = len(content)
|
|
tokens, cost, estimated = _tokens_for(deps, str(rel), content)
|
|
try:
|
|
bytes_ = ap.stat().st_size
|
|
except OSError:
|
|
bytes_ = chars
|
|
entry = {
|
|
"path": str(rel),
|
|
"name": rel.name,
|
|
"ext": rel.suffix,
|
|
"kind": _kind(rel),
|
|
"isMarkdown": is_md,
|
|
"bytes": bytes_,
|
|
"words": words,
|
|
"chars": chars,
|
|
"tokens": tokens,
|
|
"cost": cost,
|
|
"estimated": estimated,
|
|
"editable": _is_editable(rel),
|
|
}
|
|
if with_content:
|
|
entry["content"] = content
|
|
return entry
|
|
|
|
|
|
# ── path safety ─────────────────────────────────────────────────────────────
|
|
def _resolve(rel_path: str) -> tuple[pathlib.Path, pathlib.PurePosixPath]:
|
|
rel = pathlib.PurePosixPath(rel_path)
|
|
if rel.is_absolute() or ".." in rel.parts:
|
|
raise HTTPException(400, "invalid path")
|
|
ap = (WORKSPACE / rel).resolve()
|
|
try:
|
|
ap.relative_to(WORKSPACE)
|
|
except ValueError:
|
|
raise HTTPException(400, "path escapes workspace")
|
|
if not _is_context_file(rel):
|
|
raise HTTPException(403, "not an exposed context file")
|
|
return ap, rel
|
|
|
|
|
|
# ── API ─────────────────────────────────────────────────────────────────────
|
|
@router.get("/api/bundle", responses=_r(schemas.Bundle))
|
|
def bundle(deps: State):
|
|
"""The whole file catalog — metadata only, served from the index DB.
|
|
|
|
This used to read every context file's content per request (a 16 MB,
|
|
12-second response the app shell fetched on every load). File *content* now
|
|
loads lazily through ``GET /api/file`` when a file is opened; the words /
|
|
chars / token stats come from the ``files`` table, which the indexer keeps
|
|
fresh (recomputed only when a file's content actually changes)."""
|
|
files = []
|
|
for path, r in deps.store.all_files().items():
|
|
rel = pathlib.PurePosixPath(path)
|
|
tokens = r["tokens"]
|
|
estimated = bool(r["estimated"])
|
|
if tokens is None:
|
|
# not counted yet — estimate off the stored char count (~chars/4)
|
|
tokens = ((r["chars"] or 0) + 3) // 4
|
|
estimated = True
|
|
files.append({
|
|
"path": path,
|
|
"name": rel.name,
|
|
"ext": rel.suffix,
|
|
"kind": _kind(rel),
|
|
"isMarkdown": bool(r["is_markdown"]),
|
|
"bytes": r["size"] or 0,
|
|
"words": r["words"] or 0,
|
|
"chars": r["chars"] or 0,
|
|
"tokens": tokens,
|
|
"cost": r["cost"] or _price(tokens),
|
|
"estimated": estimated,
|
|
"editable": _is_editable(rel),
|
|
})
|
|
files.sort(key=lambda f: f["path"])
|
|
totals = {
|
|
"files": len(files),
|
|
"mdFiles": sum(1 for f in files if f["isMarkdown"]),
|
|
"words": sum(f["words"] for f in files),
|
|
"chars": sum(f["chars"] for f in files),
|
|
"tokens": sum(f["tokens"] for f in files),
|
|
"cost": sum(f["cost"] for f in files),
|
|
"estimated": sum(1 for f in files if f["estimated"]),
|
|
}
|
|
return {"workspace": str(WORKSPACE.name), "totals": totals, "files": files,
|
|
"pricing": {"model": MODEL, "inputPerMTok": INPUT_PRICE_PER_MTOK}}
|
|
|
|
|
|
@router.get("/api/file", responses=_r(schemas.FileEntry))
|
|
def get_file(deps: State, path: str):
|
|
ap, rel = _resolve(path)
|
|
if not ap.is_file():
|
|
raise HTTPException(404, "not found")
|
|
return _file_entry(deps, ap, rel, with_content=True)
|
|
|
|
|
|
class SaveBody(BaseModel):
|
|
path: str
|
|
content: str
|
|
|
|
|
|
@router.put("/api/file", responses=_r(schemas.FileEntry))
|
|
def put_file(deps: State, body: SaveBody):
|
|
ap, rel = _resolve(body.path)
|
|
if not _is_editable(rel):
|
|
raise HTTPException(403, "not editable")
|
|
if not ap.exists():
|
|
raise HTTPException(404, "not found")
|
|
ap.write_text(body.content, encoding="utf-8")
|
|
# Refresh on-disk metadata immediately (force the discovery walk so a
|
|
# brand-new file lands in the catalog at once); the real token count
|
|
# refreshes after the debounce.
|
|
deps.indexer.scan_files(force_walk=True)
|
|
return _file_entry(deps, ap, rel, with_content=True)
|