Files
ai-agent/backend/conversations/pricing.py
Gabriel Vidal 0ff9e40242 refactor(backend): split main.py into domain packages with app.state injection
main.py (5146 lines, 106 routes) becomes an assembly only: one package per
domain — core, files, settings, dashboards, conversations, diff,
notifications, forms, runs, accounts, workers, models, cron, agents,
projects, services, goals, memories, plans, templates — each exposing an
APIRouter; the flat domain modules move into their package behind a barrel
that keeps the old `import conversations` / `import projects` spellings.

The shared singletons (store, meta_store, hub, indexer, …) are built once by
core.state.build_state() and attached to app.state.ai; routes take them as
the `deps: State` dependency and helpers as an explicit `deps: AppState`.
conversations/pricing.py carries the per-model rates out of the parser.

Verified: route table and OpenAPI byte-identical; 90 read endpoints
golden-diffed against the monolith on a copy of the live data (identical);
write routes smoke-tested; 66 backend tests pass.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
2026-10-06 23:55:47 +02:00

109 lines
4.3 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Per-model pricing and the usage arithmetic every cost figure is built on.
Claude models use the built-in table; a pi-harness session is priced from the
OpenRouter catalogue (``models.openrouter``) when its id is known there.
"""
import models
from models import openrouter
# Per-model price in USD / million tokens (input, output). Cache tiers are
# multiples of the input price (read ≈ 0.1×, 5m write ≈ 1.25×, 1h write ≈ 2×).
# `qwen` is the *last-resort* estimate for a pi-harness OpenRouter session whose
# model isn't in the catalogue; `qwen-local` is the EVOX2 LM Studio provider,
# which is free. These are the fallback estimates — a pi transcript record
# carrying the runner-mirrored `costUSD` (pi's own per-message provider cost)
# overrides them (see `feed`).
PRICES = {"opus": (5.0, 25.0), "sonnet": (3.0, 15.0), "haiku": (1.0, 5.0),
"fable": (10.0, 50.0), "qwen": (0.15, 1.0), "qwen-local": (0.0, 0.0)}
# What a cache read costs, as a multiple of the input rate — Anthropic's flat
# 10%. OpenRouter models carry their own absolute cache price instead (see
# `rates_for`), because a cache read there is per-model policy, not a fixed cut.
CACHE_READ_MULT = 0.1
def rates_for(model: str) -> tuple[float, float, float]:
"""(input, output, cache-read) price in USD / million tokens for a model.
**A pi-harness session is priced from the OpenRouter catalogue** (APPE's
daily models.dev sync — see `openrouter.py`), so a run on any of the several
hundred pickable models is costed at that model's real rates, cache read
included. Only an id the catalogue doesn't know falls through to the flat
qwen estimate below. Claude models keep the built-in table and the 10%
cache-read cut.
A model the pi picker declares as EVOX2-local is free and skips the
catalogue entirely — its id may well carry a vendor prefix
(`ggml-org/qwen3.8-27b`), which would otherwise read as OpenRouter's.
"""
m = (model or "").lower()
local = models.pi_provider(model) == "evox2"
if "/" in m and not local: # vendor-prefixed ⇒ an OpenRouter model
hit = openrouter.rates(model)
if hit:
return hit
if "sonnet" in m:
pi, po = PRICES["sonnet"]
elif "haiku" in m:
pi, po = PRICES["haiku"]
elif "fable" in m or "mythos" in m:
pi, po = PRICES["fable"]
elif "qwen" in m or "/" in m:
# Anything the picker calls EVOX2-local is free; every other qwen or
# vendor-prefixed id ran on OpenRouter (priced above when known).
pi, po = PRICES["qwen-local"] if local or "/" not in m else PRICES["qwen"]
else:
pi, po = PRICES["opus"]
return pi, po, pi * CACHE_READ_MULT
def price_for(model: str) -> tuple[float, float]:
"""(input, output) $/Mtok — `rates_for` without the cache-read rate."""
pi, po, _ = rates_for(model)
return pi, po
def norm_usage(u: dict) -> dict:
cc = u.get("cache_creation") or {}
c5 = cc.get("ephemeral_5m_input_tokens", 0) or 0
c1 = cc.get("ephemeral_1h_input_tokens", 0) or 0
flat = u.get("cache_creation_input_tokens", 0) or 0
split = c5 + c1
return {
"input": u.get("input_tokens", 0) or 0,
"output": u.get("output_tokens", 0) or 0,
"cacheRead": u.get("cache_read_input_tokens", 0) or 0,
"cacheWriteTokens": split if split else flat,
"cacheWriteUnits": (c5 * 1.25 + c1 * 2.0) if split else flat * 1.25,
}
def add_usage(a: dict, b: dict) -> dict:
return {k: a[k] + b[k] for k in a}
def zero_usage() -> dict:
return {"input": 0, "output": 0, "cacheRead": 0,
"cacheWriteTokens": 0, "cacheWriteUnits": 0.0}
def usage_tokens(u: dict) -> int:
return int(u["input"] + u["cacheWriteTokens"] + u["cacheRead"] + u["output"])
def usage_cost(u: dict, model: str) -> float:
pi, po, pc = rates_for(model)
pi /= 1e6
po /= 1e6
pc /= 1e6
return (u["input"] * pi + u["cacheWriteUnits"] * pi
+ u["cacheRead"] * pc + u["output"] * po)
def usage_cache_cost(u: dict, model: str) -> float:
"""The share of a turn's cost that is replayed cached prompt (billed at 10%
of the input rate on Claude, at the model's own cache rate on OpenRouter).
The rest is fresh tokens the model actually had to read or write this turn."""
_, _, pc = rates_for(model)
return u["cacheRead"] * (pc / 1e6)