Files
ai-agent/backend/scaffold.py
Gabriel Vidal fe75a8debb fix(scaffold): strip the scaffold's HTML comments when publishing a summary
They are instructions for the model filling the scaffold, and the viewer renders
markdown — so the generated-at header showed up verbatim as the first line of the
published summary.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-09 21:55:28 +02:00

392 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Conversation scaffold: the algorithmic half of "complete this conversation".
The `Complete` CTA (see ``ctas.py``) used to be a hook that resumed a finished
session and asked the model to *re-read its own transcript* to summarise it —
paying for the whole conversation again to produce a paragraph. Almost all of
that work isn't reasoning, it's aggregation: which prompts were given, which
tools ran, what net change landed in git. This module does that part with plain
Python and hands the model a **scaffold** — a compact markdown document with
everything already gathered and two sections left blank for it to write.
GET /api/conversations/{id}/scaffold → text/markdown
What goes in:
- **Metadata** — session/transcript ids, cwd, timings, harness + models, turn
counts, real cost and tokens, the auto-detected projects/services/worktrees,
skills used, and the lifecycle stamps (committed/pushed/merged/deployed).
- **Prompts** — every user turn, in order, clipped. The intent of the
conversation, which no aggregate captures.
- **Work** — the tool calls *merged* rather than listed: bash bucketed by
program with counts, file edits bucketed by path, subagents with their
durations, the final task list. A 300-call conversation reduces to a page.
- **Changes** — one squashed diff across every commit the conversation
produced, from ``gitdiff.aggregate_diff`` (the same aggregation the diff
view shows), rendered back to unified-diff text and capped.
What stays blank: ``## Summary`` and ``## Analysis``. That's the model's half.
The filled document is published back with ``PUT /api/conversations/{id}/summary``
and stored on the conversation's meta sidecar (``meta.summary``), which is what
the viewer renders under the thread.
"""
import re
from datetime import datetime
import gitdiff
# Hard caps. A scaffold is meant to fit comfortably in a subagent's context —
# an uncapped diff of a big refactor would defeat the whole point of not
# re-reading the transcript.
MAX_DIFF_LINES = 1200
MAX_PROMPT_CHARS = 1500
MAX_PROMPTS = 40
MAX_BASH_EXAMPLES = 3
MAX_ROWS = 25
# Tool names that touch a file, mapped to the input key holding its path.
_FILE_TOOLS = {"Read": "file_path", "Write": "file_path", "Edit": "file_path",
"NotebookEdit": "notebook_path"}
# `cd foo &&`, `sudo`, env assignments — noise in front of the program that the
# command is actually *about*. The quoted alternatives matter: a real command
# opens with things like `FOO="$(grep -m1 … | cut -d= -f2)" ./run`, and matching
# only `\S+` would stop inside the value and report `-m1` as the program.
_CMD_PREFIX_RE = re.compile(r"""
^(?:\s*(?:
cd\s+(?:'[^']*'|"[^"]*"|\S+)\s*&& |
(?:sudo|env|command|exec)\b |
[A-Za-z_][A-Za-z0-9_]*=
(?:'[^']*'|"(?:[^"\\]|\\.)*"|\$\([^)]*\)|\S*)
)\s*)+""", re.X)
def _fmt_dt(iso: str | None) -> str:
if not iso:
return ""
try:
return datetime.fromisoformat(iso.replace("Z", "+00:00")) \
.astimezone().strftime("%Y-%m-%d %H:%M")
except ValueError:
return iso
def _duration(start: str | None, end: str | None) -> str:
try:
a = datetime.fromisoformat((start or "").replace("Z", "+00:00"))
b = datetime.fromisoformat((end or "").replace("Z", "+00:00"))
except ValueError:
return ""
secs = int((b - a).total_seconds())
if secs < 0:
return ""
h, rem = divmod(secs, 3600)
m, s = divmod(rem, 60)
return f"{h}h {m}m" if h else (f"{m}m {s}s" if m else f"{s}s")
def _clip(text: str, n: int) -> str:
text = (text or "").strip()
return text if len(text) <= n else text[:n].rstrip() + " […]"
def _program(cmd: str) -> str:
"""The program a shell command is *about* — `rtk`/`npx`-style wrappers and
`cd …&&` prefixes stripped, so `rtk git status` counts as git."""
cmd = _CMD_PREFIX_RE.sub("", (cmd or "").strip())
for p in cmd.split():
if p.startswith("-"): # a flag left over from a stripped wrapper
continue
base = p.split("/")[-1]
if base in ("rtk", "npx", "uvx", "time", "nohup", "proxy"):
continue
return base or "?"
return "?"
def _table(rows: list[tuple[str, str]]) -> list[str]:
"""A two-column markdown table, skipping rows with an empty value."""
out = ["| | |", "|---|---|"]
for k, v in rows:
if v and v != "":
out.append(f"| **{k}** | {v} |")
return out if len(out) > 2 else []
def _counted(pairs: list[tuple[str, int]], limit: int = MAX_ROWS) -> str:
"""`git ×12 · npm ×3 · …` — busiest first, with an overflow tail."""
top = sorted(pairs, key=lambda kv: -kv[1])[:limit]
rest = len(pairs) - len(top)
s = " · ".join(f"`{k}` ×{n}" if n > 1 else f"`{k}`" for k, n in top)
return s + (f" · +{rest} more" if rest > 0 else "")
# ── sections ────────────────────────────────────────────────────────────────
def _metadata(cid: str, sid: str, summary: dict, meta: dict) -> list[str]:
tok = summary.get("tokens") or 0
ctx = summary.get("contextTokens") or 0
models = ", ".join(summary.get("models") or
([summary["model"]] if summary.get("model") else []))
skills = _counted([(k, (v or {}).get("count") or 0)
for k, v in (summary.get("skills") or {}).items()])
worktrees = " · ".join(
f"`{w.get('name') or w.get('dir')}`"
+ (" (removed)" if w.get("removedAt") else "")
for w in (summary.get("worktreesAuto") or []))
auto = summary.get("lifecycleAuto") or {}
stamps = " · ".join(
k for k in ("committed", "pushed", "merged", "deployed", "notified")
if meta.get(k) or auto.get(k))
projects = ", ".join(dict.fromkeys((meta.get("projects") or [])
+ (summary.get("projectsAuto") or [])))
rows = [
("Title", summary.get("title") or ""),
("Session", f"`{sid}`"),
("Transcript", f"`{cid}`"),
("Working dir", f"`{summary.get('cwd')}`" if summary.get("cwd") else ""),
("Ran", f"{_fmt_dt(summary.get('startedAt'))}{_fmt_dt(summary.get('endedAt'))}"
f" ({_duration(summary.get('startedAt'), summary.get('endedAt'))})"),
("Harness / model", f"{meta.get('harness') or 'claude'} · {models or ''}"),
("Turns", f"{summary.get('userTurns') or 0} user · "
f"{summary.get('assistantTurns') or 0} assistant"),
("Cost", f"${summary.get('cost') or 0:.2f} · {tok:,} tokens"
+ (f" · {ctx:,} context" if ctx else "")),
("Projects", projects or ""),
("Services", ", ".join(summary.get("servicesAuto") or []) or ""),
("Worktrees", worktrees or ""),
("Skills", skills or ""),
("Lifecycle", stamps or ""),
]
return ["## Metadata", "", *_table(rows), ""]
def _prompts(thread: list[dict]) -> list[str]:
msgs = [it for it in thread
if it.get("role") == "user" and it.get("kind") == "text"
and (it.get("text") or "").strip()]
if not msgs:
return []
out = ["## Prompts", ""]
for i, it in enumerate(msgs[:MAX_PROMPTS], 1):
body = _clip(it.get("text") or "", MAX_PROMPT_CHARS)
# Indent as a blockquote so multi-line prompts stay one list item.
out.append(f"{i}. " + body.replace("\n", "\n "))
if len(msgs) > MAX_PROMPTS:
out.append(f"…and {len(msgs) - MAX_PROMPTS} more prompts.")
out.append("")
return out
def _work(thread: list[dict], summary: dict) -> list[str]:
calls = [it for it in thread if it.get("kind") == "tool_use"]
if not calls and not (summary.get("tasks") or []):
return []
bash: dict[str, int] = {}
bash_examples: dict[str, list[str]] = {}
files: dict[str, int] = {}
others: dict[str, int] = {}
agents: list[str] = []
errors = 0
for it in calls:
name = it.get("name") or "?"
inp = it.get("input") if isinstance(it.get("input"), dict) else {}
if it.get("isError"):
errors += 1
if name == "Bash":
cmd = " ".join(str(inp.get("command") or "").split())
prog = _program(cmd)
bash[prog] = bash.get(prog, 0) + 1
ex = bash_examples.setdefault(prog, [])
# Dedupe on the *clipped* form: three long commands that differ only
# past the cut would otherwise render as three identical bullets.
short = _clip(cmd, 120)
if cmd and len(ex) < MAX_BASH_EXAMPLES and short not in ex:
ex.append(short)
elif name in _FILE_TOOLS:
path = str(inp.get(_FILE_TOOLS[name]) or "?")
files[path] = files.get(path, 0) + 1
elif name in ("Task", "Agent"):
desc = inp.get("description") or inp.get("subagent_type") or "subagent"
ms = it.get("durationMs")
agents.append(f"`{inp.get('subagent_type') or 'agent'}` — {desc}"
+ (f" ({round(ms / 1000)}s)" if ms else ""))
else:
others[name] = others.get(name, 0) + 1
out = ["## Work", ""]
out.append(f"{len(calls)} tool call{'s' if len(calls) != 1 else ''}"
+ (f", {errors} returned an error" if errors else "") + ".")
out.append("")
if bash:
out.append(f"### Shell — {sum(bash.values())} calls")
out.append("")
out.append(_counted(list(bash.items())))
out.append("")
for prog, _n in sorted(bash.items(), key=lambda kv: -kv[1])[:8]:
for ex in bash_examples.get(prog, []):
out.append(f"- `{ex}`")
out.append("")
if files:
edits = sum(files.values())
out.append(f"### Files — {edits} call{'s' if edits != 1 else ''} "
f"across {len(files)} file{'s' if len(files) != 1 else ''}")
out.append("")
out.append(_counted(list(files.items())))
out.append("")
if others:
out.append("### Other tools")
out.append("")
out.append(_counted(list(others.items())))
out.append("")
if agents:
out.append(f"### Subagents — {len(agents)}")
out.append("")
out += [f"- {a}" for a in agents[:MAX_ROWS]]
out.append("")
tasks = summary.get("tasks") or []
if tasks:
out.append("### Task list")
out.append("")
mark = {"completed": "x", "cancelled": "-"}
for t in tasks[:MAX_ROWS]:
out.append(f"- [{mark.get(t.get('status'), ' ')}] {t.get('subject')}")
out.append("")
return out
def _render_diff(files: list[dict]) -> tuple[list[str], bool]:
"""Re-render the parsed file/hunk model back to unified-diff text, capped."""
lines: list[str] = []
truncated = False
for f in files:
if len(lines) >= MAX_DIFF_LINES:
truncated = True
break
head = f"--- a/{f.get('oldPath') or f.get('path')}\n+++ b/{f.get('path')}"
lines.append(head)
if f.get("binary"):
lines.append("Binary file differs")
continue
for h in f.get("hunks") or []:
if len(lines) >= MAX_DIFF_LINES:
truncated = True
break
lines.append(h.get("header") or "@@")
for ln in h.get("lines") or []:
if len(lines) >= MAX_DIFF_LINES:
truncated = True
break
sign = {"add": "+", "del": "-"}.get(ln.get("type"), " ")
lines.append(sign + (ln.get("text") or ""))
return lines, truncated
def _changes(store, summary: dict, meta: dict, sid: str) -> list[str]:
"""One squashed diff across every commit the conversation produced."""
try:
data = gitdiff.conversation_commits(store, summary, meta, sid)
except Exception:
return []
commits = data.get("commits") or []
if not commits:
return ["## Changes", "", "No commits from this conversation.", ""]
by_repo: dict[str, list[str]] = {}
for c in commits:
by_repo.setdefault(c["repo"], []).append(c["sha"])
files: list[dict] = []
add = dele = 0
for token, shas in by_repo.items():
agg = gitdiff.aggregate_diff(token, shas, title="", subtitle="")
if agg:
files += agg["files"]
add += agg["additions"]
dele += agg["deletions"]
out = ["## Changes", "",
f"{len(commits)} commit{'s' if len(commits) != 1 else ''} · "
f"{len(files)} file{'s' if len(files) != 1 else ''} · "
f"+{add} {dele}", ""]
for c in commits[:MAX_ROWS]:
out.append(f"- `{c.get('short') or c.get('sha', '')[:8]}` "
f"{c.get('subject') or ''} — *{c.get('repo')}*")
out.append("")
if files:
out.append("### Files")
out.append("")
for f in files[:MAX_ROWS]:
out.append(f"- `{f.get('path')}` ({f.get('status')}, "
f"+{f.get('additions')} {f.get('deletions')})")
if len(files) > MAX_ROWS:
out.append(f"- …and {len(files) - MAX_ROWS} more files")
out.append("")
body, truncated = _render_diff(files)
out.append("### Squashed diff")
out.append("")
out.append("```diff")
out += body
out.append("```")
if truncated:
out.append("")
out.append(f"*Diff truncated at {MAX_DIFF_LINES} lines — read the "
f"files directly if you need the rest.*")
out.append("")
return out
# The two sections the model fills in. Kept as constants so `publish` can check
# whether they were actually written (an unedited scaffold is a no-op, not a
# summary) and so the CLI's `--check` says the same thing the backend does.
SUMMARY_HEADING = "## Summary"
ANALYSIS_HEADING = "## Analysis"
_SUMMARY_TODO = "<!-- fill in: what was asked for, what was built, what shipped -->"
_ANALYSIS_TODO = ("<!-- fill in: key decisions and why, what's worth remembering, "
"what is still open -->")
def build(cid: str, sid: str, summary: dict, meta: dict, store) -> str:
"""The scaffold markdown for one conversation. ``summary`` must be a **full**
parse (``thread`` present) — the merged tool sections come from it."""
thread = summary.get("thread") or []
generated = datetime.now().astimezone().strftime("%Y-%m-%d %H:%M")
out = [
f"<!-- conv-scaffold — generated {generated}. Fill in the Summary and "
f"Analysis sections, then run `conv-scaffold publish <this file>`. -->",
"",
f"# {summary.get('title') or 'Conversation'}",
"",
]
out += _metadata(cid, sid, summary, meta)
out += _prompts(thread)
out += _work(thread, summary)
out += _changes(store, summary, meta, sid)
out += [SUMMARY_HEADING, "", _SUMMARY_TODO, "",
ANALYSIS_HEADING, "", _ANALYSIS_TODO, ""]
return "\n".join(out)
def clean_for_publish(markdown: str) -> str:
"""Strip the HTML comments out of a filled scaffold.
They are scaffolding *for the model* — the generated-at header telling it to
fill the two sections, and any placeholder it left behind — and the viewer
renders markdown, not HTML, so they'd otherwise show up verbatim in the
published summary as a stray line of instructions."""
out = re.sub(r"<!--.*?-->", "", markdown or "", flags=re.S)
# Collapse the blank-line runs the removals leave behind.
return re.sub(r"\n{3,}", "\n\n", out).strip() + "\n"
def is_filled(markdown: str) -> bool:
"""True when the Summary section has real prose in it — the guard that stops
an untouched scaffold from being published as a conversation's summary."""
text = markdown or ""
if _SUMMARY_TODO in text:
return False
i = text.find(SUMMARY_HEADING)
if i < 0:
return False
body = text[i + len(SUMMARY_HEADING):]
end = body.find("\n## ")
body = (body if end < 0 else body[:end])
body = re.sub(r"<!--.*?-->", "", body, flags=re.S)
return len(body.strip()) >= 20