Files
ai-agent/backend/scaffold.py
Gabriel Vidal 8dccce2334 feat(ctas): replace the conversation-hook scheduler with end-of-conversation CTAs
A hook resumed every finished run on a timer — unasked-for, and expensive: a
`--resume` never hits the finished run's prompt cache, so each firing re-created
the whole transcript as fresh input tokens. The prompts stay, the scheduler goes.

A CTA is a prompt file plus a button under the last message of a finished
conversation: click loads its prompt into the resume box (editable before it is
sent), long-press fires it at the session in the background. The same prompts are
selectable as `cta:<id>` tags in every composer.

The seeded `complete` CTA also does the post-processing the hook never could
afford: `conv-scaffold` reduces the conversation to markdown — metadata, its
prompts, merged bash/tool/subagent summaries, one squashed diff across every
commit it produced — and one subagent fills in the Summary/Analysis and publishes
it onto `meta.summary`, rendered under the thread.

- backend: ctas.py (store + seeded Complete), scaffold.py (+ /scaffold, PUT
  /summary), /api/ctas CRUD/run/prompt/reorder; hooks.py and its scheduler,
  pending queue and routes deleted (read-only /api/claude-hooks stays)
- frontend: ConversationCtas section, Settings → CTAs, cta tags, CtaBadge
- scripts/conv-scaffold.sh: scaffold + publish CLI

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-09 21:46:27 +02:00

380 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Conversation scaffold: the algorithmic half of "complete this conversation".
The `Complete` CTA (see ``ctas.py``) used to be a hook that resumed a finished
session and asked the model to *re-read its own transcript* to summarise it —
paying for the whole conversation again to produce a paragraph. Almost all of
that work isn't reasoning, it's aggregation: which prompts were given, which
tools ran, what net change landed in git. This module does that part with plain
Python and hands the model a **scaffold** — a compact markdown document with
everything already gathered and two sections left blank for it to write.
GET /api/conversations/{id}/scaffold → text/markdown
What goes in:
- **Metadata** — session/transcript ids, cwd, timings, harness + models, turn
counts, real cost and tokens, the auto-detected projects/services/worktrees,
skills used, and the lifecycle stamps (committed/pushed/merged/deployed).
- **Prompts** — every user turn, in order, clipped. The intent of the
conversation, which no aggregate captures.
- **Work** — the tool calls *merged* rather than listed: bash bucketed by
program with counts, file edits bucketed by path, subagents with their
durations, the final task list. A 300-call conversation reduces to a page.
- **Changes** — one squashed diff across every commit the conversation
produced, from ``gitdiff.aggregate_diff`` (the same aggregation the diff
view shows), rendered back to unified-diff text and capped.
What stays blank: ``## Summary`` and ``## Analysis``. That's the model's half.
The filled document is published back with ``PUT /api/conversations/{id}/summary``
and stored on the conversation's meta sidecar (``meta.summary``), which is what
the viewer renders under the thread.
"""
import re
from datetime import datetime
import gitdiff
# Hard caps. A scaffold is meant to fit comfortably in a subagent's context —
# an uncapped diff of a big refactor would defeat the whole point of not
# re-reading the transcript.
MAX_DIFF_LINES = 1200
MAX_PROMPT_CHARS = 1500
MAX_PROMPTS = 40
MAX_BASH_EXAMPLES = 3
MAX_ROWS = 25
# Tool names that touch a file, mapped to the input key holding its path.
_FILE_TOOLS = {"Read": "file_path", "Write": "file_path", "Edit": "file_path",
"NotebookEdit": "notebook_path"}
# `cd foo &&`, `sudo`, env assignments — noise in front of the program that the
# command is actually *about*. The quoted alternatives matter: a real command
# opens with things like `FOO="$(grep -m1 … | cut -d= -f2)" ./run`, and matching
# only `\S+` would stop inside the value and report `-m1` as the program.
_CMD_PREFIX_RE = re.compile(r"""
^(?:\s*(?:
cd\s+(?:'[^']*'|"[^"]*"|\S+)\s*&& |
(?:sudo|env|command|exec)\b |
[A-Za-z_][A-Za-z0-9_]*=
(?:'[^']*'|"(?:[^"\\]|\\.)*"|\$\([^)]*\)|\S*)
)\s*)+""", re.X)
def _fmt_dt(iso: str | None) -> str:
if not iso:
return ""
try:
return datetime.fromisoformat(iso.replace("Z", "+00:00")) \
.astimezone().strftime("%Y-%m-%d %H:%M")
except ValueError:
return iso
def _duration(start: str | None, end: str | None) -> str:
try:
a = datetime.fromisoformat((start or "").replace("Z", "+00:00"))
b = datetime.fromisoformat((end or "").replace("Z", "+00:00"))
except ValueError:
return ""
secs = int((b - a).total_seconds())
if secs < 0:
return ""
h, rem = divmod(secs, 3600)
m, s = divmod(rem, 60)
return f"{h}h {m}m" if h else (f"{m}m {s}s" if m else f"{s}s")
def _clip(text: str, n: int) -> str:
text = (text or "").strip()
return text if len(text) <= n else text[:n].rstrip() + " […]"
def _program(cmd: str) -> str:
"""The program a shell command is *about* — `rtk`/`npx`-style wrappers and
`cd …&&` prefixes stripped, so `rtk git status` counts as git."""
cmd = _CMD_PREFIX_RE.sub("", (cmd or "").strip())
for p in cmd.split():
if p.startswith("-"): # a flag left over from a stripped wrapper
continue
base = p.split("/")[-1]
if base in ("rtk", "npx", "uvx", "time", "nohup", "proxy"):
continue
return base or "?"
return "?"
def _table(rows: list[tuple[str, str]]) -> list[str]:
"""A two-column markdown table, skipping rows with an empty value."""
out = ["| | |", "|---|---|"]
for k, v in rows:
if v and v != "":
out.append(f"| **{k}** | {v} |")
return out if len(out) > 2 else []
def _counted(pairs: list[tuple[str, int]], limit: int = MAX_ROWS) -> str:
"""`git ×12 · npm ×3 · …` — busiest first, with an overflow tail."""
top = sorted(pairs, key=lambda kv: -kv[1])[:limit]
rest = len(pairs) - len(top)
s = " · ".join(f"`{k}` ×{n}" if n > 1 else f"`{k}`" for k, n in top)
return s + (f" · +{rest} more" if rest > 0 else "")
# ── sections ────────────────────────────────────────────────────────────────
def _metadata(cid: str, sid: str, summary: dict, meta: dict) -> list[str]:
tok = summary.get("tokens") or 0
ctx = summary.get("contextTokens") or 0
models = ", ".join(summary.get("models") or
([summary["model"]] if summary.get("model") else []))
skills = _counted([(k, (v or {}).get("count") or 0)
for k, v in (summary.get("skills") or {}).items()])
worktrees = " · ".join(
f"`{w.get('name') or w.get('dir')}`"
+ (" (removed)" if w.get("removedAt") else "")
for w in (summary.get("worktreesAuto") or []))
auto = summary.get("lifecycleAuto") or {}
stamps = " · ".join(
k for k in ("committed", "pushed", "merged", "deployed", "notified")
if meta.get(k) or auto.get(k))
projects = ", ".join(dict.fromkeys((meta.get("projects") or [])
+ (summary.get("projectsAuto") or [])))
rows = [
("Title", summary.get("title") or ""),
("Session", f"`{sid}`"),
("Transcript", f"`{cid}`"),
("Working dir", f"`{summary.get('cwd')}`" if summary.get("cwd") else ""),
("Ran", f"{_fmt_dt(summary.get('startedAt'))}{_fmt_dt(summary.get('endedAt'))}"
f" ({_duration(summary.get('startedAt'), summary.get('endedAt'))})"),
("Harness / model", f"{meta.get('harness') or 'claude'} · {models or ''}"),
("Turns", f"{summary.get('userTurns') or 0} user · "
f"{summary.get('assistantTurns') or 0} assistant"),
("Cost", f"${summary.get('cost') or 0:.2f} · {tok:,} tokens"
+ (f" · {ctx:,} context" if ctx else "")),
("Projects", projects or ""),
("Services", ", ".join(summary.get("servicesAuto") or []) or ""),
("Worktrees", worktrees or ""),
("Skills", skills or ""),
("Lifecycle", stamps or ""),
]
return ["## Metadata", "", *_table(rows), ""]
def _prompts(thread: list[dict]) -> list[str]:
msgs = [it for it in thread
if it.get("role") == "user" and it.get("kind") == "text"
and (it.get("text") or "").strip()]
if not msgs:
return []
out = ["## Prompts", ""]
for i, it in enumerate(msgs[:MAX_PROMPTS], 1):
body = _clip(it.get("text") or "", MAX_PROMPT_CHARS)
# Indent as a blockquote so multi-line prompts stay one list item.
out.append(f"{i}. " + body.replace("\n", "\n "))
if len(msgs) > MAX_PROMPTS:
out.append(f"…and {len(msgs) - MAX_PROMPTS} more prompts.")
out.append("")
return out
def _work(thread: list[dict], summary: dict) -> list[str]:
calls = [it for it in thread if it.get("kind") == "tool_use"]
if not calls and not (summary.get("tasks") or []):
return []
bash: dict[str, int] = {}
bash_examples: dict[str, list[str]] = {}
files: dict[str, int] = {}
others: dict[str, int] = {}
agents: list[str] = []
errors = 0
for it in calls:
name = it.get("name") or "?"
inp = it.get("input") if isinstance(it.get("input"), dict) else {}
if it.get("isError"):
errors += 1
if name == "Bash":
cmd = " ".join(str(inp.get("command") or "").split())
prog = _program(cmd)
bash[prog] = bash.get(prog, 0) + 1
ex = bash_examples.setdefault(prog, [])
# Dedupe on the *clipped* form: three long commands that differ only
# past the cut would otherwise render as three identical bullets.
short = _clip(cmd, 120)
if cmd and len(ex) < MAX_BASH_EXAMPLES and short not in ex:
ex.append(short)
elif name in _FILE_TOOLS:
path = str(inp.get(_FILE_TOOLS[name]) or "?")
files[path] = files.get(path, 0) + 1
elif name in ("Task", "Agent"):
desc = inp.get("description") or inp.get("subagent_type") or "subagent"
ms = it.get("durationMs")
agents.append(f"`{inp.get('subagent_type') or 'agent'}` — {desc}"
+ (f" ({round(ms / 1000)}s)" if ms else ""))
else:
others[name] = others.get(name, 0) + 1
out = ["## Work", ""]
out.append(f"{len(calls)} tool call{'s' if len(calls) != 1 else ''}"
+ (f", {errors} returned an error" if errors else "") + ".")
out.append("")
if bash:
out.append(f"### Shell — {sum(bash.values())} calls")
out.append("")
out.append(_counted(list(bash.items())))
out.append("")
for prog, _n in sorted(bash.items(), key=lambda kv: -kv[1])[:8]:
for ex in bash_examples.get(prog, []):
out.append(f"- `{ex}`")
out.append("")
if files:
edits = sum(files.values())
out.append(f"### Files — {edits} call{'s' if edits != 1 else ''} "
f"across {len(files)} file{'s' if len(files) != 1 else ''}")
out.append("")
out.append(_counted(list(files.items())))
out.append("")
if others:
out.append("### Other tools")
out.append("")
out.append(_counted(list(others.items())))
out.append("")
if agents:
out.append(f"### Subagents — {len(agents)}")
out.append("")
out += [f"- {a}" for a in agents[:MAX_ROWS]]
out.append("")
tasks = summary.get("tasks") or []
if tasks:
out.append("### Task list")
out.append("")
mark = {"completed": "x", "cancelled": "-"}
for t in tasks[:MAX_ROWS]:
out.append(f"- [{mark.get(t.get('status'), ' ')}] {t.get('subject')}")
out.append("")
return out
def _render_diff(files: list[dict]) -> tuple[list[str], bool]:
"""Re-render the parsed file/hunk model back to unified-diff text, capped."""
lines: list[str] = []
truncated = False
for f in files:
if len(lines) >= MAX_DIFF_LINES:
truncated = True
break
head = f"--- a/{f.get('oldPath') or f.get('path')}\n+++ b/{f.get('path')}"
lines.append(head)
if f.get("binary"):
lines.append("Binary file differs")
continue
for h in f.get("hunks") or []:
if len(lines) >= MAX_DIFF_LINES:
truncated = True
break
lines.append(h.get("header") or "@@")
for ln in h.get("lines") or []:
if len(lines) >= MAX_DIFF_LINES:
truncated = True
break
sign = {"add": "+", "del": "-"}.get(ln.get("type"), " ")
lines.append(sign + (ln.get("text") or ""))
return lines, truncated
def _changes(store, summary: dict, meta: dict, sid: str) -> list[str]:
"""One squashed diff across every commit the conversation produced."""
try:
data = gitdiff.conversation_commits(store, summary, meta, sid)
except Exception:
return []
commits = data.get("commits") or []
if not commits:
return ["## Changes", "", "No commits from this conversation.", ""]
by_repo: dict[str, list[str]] = {}
for c in commits:
by_repo.setdefault(c["repo"], []).append(c["sha"])
files: list[dict] = []
add = dele = 0
for token, shas in by_repo.items():
agg = gitdiff.aggregate_diff(token, shas, title="", subtitle="")
if agg:
files += agg["files"]
add += agg["additions"]
dele += agg["deletions"]
out = ["## Changes", "",
f"{len(commits)} commit{'s' if len(commits) != 1 else ''} · "
f"{len(files)} file{'s' if len(files) != 1 else ''} · "
f"+{add} {dele}", ""]
for c in commits[:MAX_ROWS]:
out.append(f"- `{c.get('short') or c.get('sha', '')[:8]}` "
f"{c.get('subject') or ''} — *{c.get('repo')}*")
out.append("")
if files:
out.append("### Files")
out.append("")
for f in files[:MAX_ROWS]:
out.append(f"- `{f.get('path')}` ({f.get('status')}, "
f"+{f.get('additions')} {f.get('deletions')})")
if len(files) > MAX_ROWS:
out.append(f"- …and {len(files) - MAX_ROWS} more files")
out.append("")
body, truncated = _render_diff(files)
out.append("### Squashed diff")
out.append("")
out.append("```diff")
out += body
out.append("```")
if truncated:
out.append("")
out.append(f"*Diff truncated at {MAX_DIFF_LINES} lines — read the "
f"files directly if you need the rest.*")
out.append("")
return out
# The two sections the model fills in. Kept as constants so `publish` can check
# whether they were actually written (an unedited scaffold is a no-op, not a
# summary) and so the CLI's `--check` says the same thing the backend does.
SUMMARY_HEADING = "## Summary"
ANALYSIS_HEADING = "## Analysis"
_SUMMARY_TODO = "<!-- fill in: what was asked for, what was built, what shipped -->"
_ANALYSIS_TODO = ("<!-- fill in: key decisions and why, what's worth remembering, "
"what is still open -->")
def build(cid: str, sid: str, summary: dict, meta: dict, store) -> str:
"""The scaffold markdown for one conversation. ``summary`` must be a **full**
parse (``thread`` present) — the merged tool sections come from it."""
thread = summary.get("thread") or []
generated = datetime.now().astimezone().strftime("%Y-%m-%d %H:%M")
out = [
f"<!-- conv-scaffold — generated {generated}. Fill in the Summary and "
f"Analysis sections, then run `conv-scaffold publish <this file>`. -->",
"",
f"# {summary.get('title') or 'Conversation'}",
"",
]
out += _metadata(cid, sid, summary, meta)
out += _prompts(thread)
out += _work(thread, summary)
out += _changes(store, summary, meta, sid)
out += [SUMMARY_HEADING, "", _SUMMARY_TODO, "",
ANALYSIS_HEADING, "", _ANALYSIS_TODO, ""]
return "\n".join(out)
def is_filled(markdown: str) -> bool:
"""True when the Summary section has real prose in it — the guard that stops
an untouched scaffold from being published as a conversation's summary."""
text = markdown or ""
if _SUMMARY_TODO in text:
return False
i = text.find(SUMMARY_HEADING)
if i < 0:
return False
body = text[i + len(SUMMARY_HEADING):]
end = body.find("\n## ")
body = (body if end < 0 else body[:end])
body = re.sub(r"<!--.*?-->", "", body, flags=re.S)
return len(body.strip()) >= 20