main.py (5146 lines, 106 routes) becomes an assembly only: one package per domain — core, files, settings, dashboards, conversations, diff, notifications, forms, runs, accounts, workers, models, cron, agents, projects, services, goals, memories, plans, templates — each exposing an APIRouter; the flat domain modules move into their package behind a barrel that keeps the old `import conversations` / `import projects` spellings. The shared singletons (store, meta_store, hub, indexer, …) are built once by core.state.build_state() and attached to app.state.ai; routes take them as the `deps: State` dependency and helpers as an explicit `deps: AppState`. conversations/pricing.py carries the per-model rates out of the parser. Verified: route table and OpenAPI byte-identical; 90 read endpoints golden-diffed against the monolith on a copy of the live data (identical); write routes smoke-tested; 66 backend tests pass. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
109 lines
4.3 KiB
Python
109 lines
4.3 KiB
Python
"""Per-model pricing and the usage arithmetic every cost figure is built on.
|
||
|
||
Claude models use the built-in table; a pi-harness session is priced from the
|
||
OpenRouter catalogue (``models.openrouter``) when its id is known there.
|
||
"""
|
||
|
||
import models
|
||
from models import openrouter
|
||
|
||
# Per-model price in USD / million tokens (input, output). Cache tiers are
|
||
# multiples of the input price (read ≈ 0.1×, 5m write ≈ 1.25×, 1h write ≈ 2×).
|
||
# `qwen` is the *last-resort* estimate for a pi-harness OpenRouter session whose
|
||
# model isn't in the catalogue; `qwen-local` is the EVOX2 LM Studio provider,
|
||
# which is free. These are the fallback estimates — a pi transcript record
|
||
# carrying the runner-mirrored `costUSD` (pi's own per-message provider cost)
|
||
# overrides them (see `feed`).
|
||
PRICES = {"opus": (5.0, 25.0), "sonnet": (3.0, 15.0), "haiku": (1.0, 5.0),
|
||
"fable": (10.0, 50.0), "qwen": (0.15, 1.0), "qwen-local": (0.0, 0.0)}
|
||
# What a cache read costs, as a multiple of the input rate — Anthropic's flat
|
||
# 10%. OpenRouter models carry their own absolute cache price instead (see
|
||
# `rates_for`), because a cache read there is per-model policy, not a fixed cut.
|
||
CACHE_READ_MULT = 0.1
|
||
|
||
|
||
def rates_for(model: str) -> tuple[float, float, float]:
|
||
"""(input, output, cache-read) price in USD / million tokens for a model.
|
||
|
||
**A pi-harness session is priced from the OpenRouter catalogue** (APPE's
|
||
daily models.dev sync — see `openrouter.py`), so a run on any of the several
|
||
hundred pickable models is costed at that model's real rates, cache read
|
||
included. Only an id the catalogue doesn't know falls through to the flat
|
||
qwen estimate below. Claude models keep the built-in table and the 10%
|
||
cache-read cut.
|
||
|
||
A model the pi picker declares as EVOX2-local is free and skips the
|
||
catalogue entirely — its id may well carry a vendor prefix
|
||
(`ggml-org/qwen3.8-27b`), which would otherwise read as OpenRouter's.
|
||
"""
|
||
m = (model or "").lower()
|
||
local = models.pi_provider(model) == "evox2"
|
||
if "/" in m and not local: # vendor-prefixed ⇒ an OpenRouter model
|
||
hit = openrouter.rates(model)
|
||
if hit:
|
||
return hit
|
||
if "sonnet" in m:
|
||
pi, po = PRICES["sonnet"]
|
||
elif "haiku" in m:
|
||
pi, po = PRICES["haiku"]
|
||
elif "fable" in m or "mythos" in m:
|
||
pi, po = PRICES["fable"]
|
||
elif "qwen" in m or "/" in m:
|
||
# Anything the picker calls EVOX2-local is free; every other qwen or
|
||
# vendor-prefixed id ran on OpenRouter (priced above when known).
|
||
pi, po = PRICES["qwen-local"] if local or "/" not in m else PRICES["qwen"]
|
||
else:
|
||
pi, po = PRICES["opus"]
|
||
return pi, po, pi * CACHE_READ_MULT
|
||
|
||
|
||
def price_for(model: str) -> tuple[float, float]:
|
||
"""(input, output) $/Mtok — `rates_for` without the cache-read rate."""
|
||
pi, po, _ = rates_for(model)
|
||
return pi, po
|
||
|
||
|
||
def norm_usage(u: dict) -> dict:
|
||
cc = u.get("cache_creation") or {}
|
||
c5 = cc.get("ephemeral_5m_input_tokens", 0) or 0
|
||
c1 = cc.get("ephemeral_1h_input_tokens", 0) or 0
|
||
flat = u.get("cache_creation_input_tokens", 0) or 0
|
||
split = c5 + c1
|
||
return {
|
||
"input": u.get("input_tokens", 0) or 0,
|
||
"output": u.get("output_tokens", 0) or 0,
|
||
"cacheRead": u.get("cache_read_input_tokens", 0) or 0,
|
||
"cacheWriteTokens": split if split else flat,
|
||
"cacheWriteUnits": (c5 * 1.25 + c1 * 2.0) if split else flat * 1.25,
|
||
}
|
||
|
||
|
||
def add_usage(a: dict, b: dict) -> dict:
|
||
return {k: a[k] + b[k] for k in a}
|
||
|
||
|
||
def zero_usage() -> dict:
|
||
return {"input": 0, "output": 0, "cacheRead": 0,
|
||
"cacheWriteTokens": 0, "cacheWriteUnits": 0.0}
|
||
|
||
|
||
def usage_tokens(u: dict) -> int:
|
||
return int(u["input"] + u["cacheWriteTokens"] + u["cacheRead"] + u["output"])
|
||
|
||
|
||
def usage_cost(u: dict, model: str) -> float:
|
||
pi, po, pc = rates_for(model)
|
||
pi /= 1e6
|
||
po /= 1e6
|
||
pc /= 1e6
|
||
return (u["input"] * pi + u["cacheWriteUnits"] * pi
|
||
+ u["cacheRead"] * pc + u["output"] * po)
|
||
|
||
|
||
def usage_cache_cost(u: dict, model: str) -> float:
|
||
"""The share of a turn's cost that is replayed cached prompt (billed at 10%
|
||
of the input rate on Claude, at the model's own cache rate on OpenRouter).
|
||
The rest is fresh tokens the model actually had to read or write this turn."""
|
||
_, _, pc = rates_for(model)
|
||
return u["cacheRead"] * (pc / 1e6)
|