Coverage for src/keel/cost.py: 100%
159 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""Keel token counts and USD cost estimates.
3Totals the token counts the activity records under ``.keel/activity`` carry — or a
4placeholder where they carry none — and prices them from built-in model pricing tables,
5without any external billing API.
7**The report says which of its figures are measured.** A record that carries a
8non-zero ``prompt_tokens`` / ``completion_tokens``, or a
9:data:`keel.activity.USAGE_FIELD` entry with non-zero counts, is *measured*; one that
10carries neither is priced at :data:`ASSUMED_PROMPT_TOKENS` /
11:data:`ASSUMED_COMPLETION_TOKENS` and counted as *estimated* (#1359) — a report whose
12dollar figure is a run count times a constant must not read as a bill.
13:class:`CostReport` carries the split, and both renderings state it.
15The one keel writer of counts is ``keel delegate run --activity-run-id`` (#1373): a
16hosted-API delegate's reported usage, one entry per call, priced at that call's model.
17A measured run is therefore measured *for its delegate calls*: an agent host's own
18tokens, and a CLI delegate's, are never in a record, and the report says so. The dollar
19figures stay estimates either way — they are keel's pricing table applied to counts,
20not the provider's bill.
21"""
23from __future__ import annotations
25import itertools
26import re
27from dataclasses import dataclass
28from pathlib import Path
29from typing import Any
31from . import activity, agents
32from .agents import LOCAL_TRANSPORTS
34# Model pricing in USD per 1,000,000 tokens: (prompt_price_per_m, completion_price_per_m)
35MODEL_PRICING: dict[str, tuple[float, float]] = {
36 # Anthropic
37 "claude-3-7-sonnet": (3.00, 15.00),
38 "claude-3-5-sonnet": (3.00, 15.00),
39 "claude-3-5-haiku": (0.80, 4.00),
40 "claude-3-opus": (15.00, 75.00),
41 "claude": (3.00, 15.00),
42 # Google Gemini
43 "gemini-2.5-pro": (1.25, 5.00),
44 "gemini-2.5-flash": (0.15, 0.60),
45 "gemini-1.5-pro": (1.25, 5.00),
46 "gemini-1.5-flash": (0.075, 0.30),
47 "gemini": (0.15, 0.60),
48 # OpenAI
49 "gpt-4o": (2.50, 10.00),
50 "gpt-4o-mini": (0.15, 0.60),
51 "o1": (15.00, 60.00),
52 "o3-mini": (1.10, 4.40),
53 "codex": (2.50, 10.00),
54 # DeepSeek
55 "deepseek-chat": (0.27, 1.10),
56 "deepseek-reasoner": (0.55, 2.19),
57 "deepseek": (0.27, 1.10),
58 # Local
59 "ollama": (0.00, 0.00),
60 "local": (0.00, 0.00),
61}
63DEFAULT_FALLBACK_PRICE = (1.00, 3.00)
65#: The token counts a record without its own is priced at: a placeholder per run, not a
66#: measurement. Named, because the report has to name it — #1359 found the numbers
67#: inlined in the aggregation loop and nowhere in the output, so a report built entirely
68#: from them printed a dollar figure with nothing to say it was one.
69ASSUMED_PROMPT_TOKENS = 1500
70ASSUMED_COMPLETION_TOKENS = 400
71FRONTIER_BENCHMARK_PRICE = (15.00, 75.00) # Used for computing savings vs Claude Opus/o1
73# #930's `_PRICING_KEYS_BY_LENGTH` is gone with the substring scan it ordered.
74# Longest-first existed so `gpt-4o-mini` would not be captured by `gpt-4o`;
75# exact resolution makes that impossible by construction, and a constant nothing
76# reads is a comment pretending to be a guard. The property it protected is
77# asserted on the outcome in `tests/test_cost.py`.
80#: Cloud re-hosters put the vendor in front of an otherwise ordinary model id.
81#: Bedrock uses ``anthropic.claude-…-v1:0``, Vertex
82#: ``publishers/anthropic/models/…``, OpenRouter ``anthropic/claude-…``. The old
83#: code split on the first ``:`` and kept the **right** side, so a Bedrock id
84#: normalised to ``0`` — not an approximation, a total loss of the model name
85#: (#941).
86_REHOST_VENDORS = frozenset(
87 {
88 "anthropic",
89 "openai",
90 "google",
91 "meta",
92 "mistral",
93 "cohere",
94 "amazon",
95 "ai21",
96 "deepseek",
97 "qwen",
98 "x-ai",
99 "perplexity",
100 }
101)
103#: A release stamp or revision that carries no pricing signal: ``-20240229``,
104#: ``-v1``, ``-latest``, ``-preview``.
105_MODEL_STAMP = re.compile(r"-(?:\d{6,8}|v\d+|latest|preview|exp)(?=-|$)")
107#: The bridge #942 asked for: the vocabulary :func:`keel.agents.model_base`
108#: emits, mapped onto the pricing keys that already exist.
109#:
110#: keel writes a versionless ``model:<base>`` label on every PR — ``opus-4-8``,
111#: ``sonnet-4-5`` — and **none** of those bases appears in ``MODEL_PRICING``,
112#: whose keys are 2024-era product names. So `keel cost-report` priced keel's own
113#: runs at the ``DEFAULT_FALLBACK_PRICE``: roughly 5 % of true Opus spend, with
114#: the difference then claimed as savings (#944).
115#:
116#: **These are aliases onto prices already in the table, not new prices.** An
117#: alias says "this label names that tier", which is a fact about naming and is
118#: checkable. Whether the tier's *numbers* are still current is a separate,
119#: operator-owned question — #942's "refresh the keys" — and inventing figures
120#: here would bury that question under a plausible-looking table.
121MODEL_ALIASES: dict[str, str] = {
122 # Anthropic — keel's own attribution bases and the vendor's current ids.
123 "opus": "claude-3-opus",
124 "opus-4": "claude-3-opus",
125 "opus-4-5": "claude-3-opus",
126 "opus-4-8": "claude-3-opus",
127 "claude-opus": "claude-3-opus",
128 "claude-opus-4": "claude-3-opus",
129 "sonnet": "claude-3-7-sonnet",
130 "sonnet-4": "claude-3-7-sonnet",
131 "sonnet-4-5": "claude-3-7-sonnet",
132 "claude-sonnet": "claude-3-7-sonnet",
133 "claude-sonnet-4": "claude-3-7-sonnet",
134 "haiku": "claude-3-5-haiku",
135 "haiku-4-5": "claude-3-5-haiku",
136 "claude-haiku": "claude-3-5-haiku",
137 # OpenAI. Only the CLI's own label, which names the same product as `codex`.
138 "codex-cli": "codex",
139}
141# Deliberately absent: `gpt-5 -> gpt-4o`, `gemini-2 -> gemini-2.5-pro` and
142# friends. Those are not naming facts, they are price guesses across tiers — and
143# the first draft of this map proved the point by resolving
144# `gemini-2.5-flash-lite` to the *pro* price, 8x its own. An unknown model is
145# reported as unpriced, which `calculate_cost_report` counts; a wrong tier is
146# reported as a number, which nobody counts.
149def _bare_model_id(model: str) -> str:
150 """Strip re-hosting decoration down to the vendor's own model id.
152 Order matters: the path form is unwrapped before the colon is read, because
153 a Bedrock id carries *both* (``anthropic.claude-3-opus-20240229-v1:0``).
154 """
155 raw = model.lower().strip()
156 # Local inference is free, and `MODEL_PRICING` prices the *tier* at 0.00 —
157 # so `ollama:`/`local:` collapse to the transport on purpose. This is the one
158 # place where pricing and attribution legitimately want different answers:
159 # #955's label must name the model, this must name the free tier.
160 if raw.startswith(tuple(f"{p}:" for p in LOCAL_TRANSPORTS)):
161 return raw.split(":", 1)[0]
162 # `<vendor>-api:model` is a transport prefix, and which names are transports
163 # is defined once, in `agents` (#955). Sharing the definition is what stops
164 # the pricing key and the attribution label drifting apart again.
165 raw = agents.strip_transport(raw)
166 if "/" in raw: # openrouter `vendor/model`, vertex `publishers/v/models/model`
167 raw = raw.rsplit("/", 1)[1]
168 if ":" in raw:
169 head, tail = raw.split(":", 1)
170 # `…-v1:0` is a Bedrock revision; `google:gemini-2.5-pro` is an
171 # unrecognised re-hoster, whose model is still on the right.
172 raw = head if tail.isdigit() else tail
173 if "." in raw and raw.split(".", 1)[0] in _REHOST_VENDORS:
174 raw = raw.split(".", 1)[1]
175 return _MODEL_STAMP.sub("", raw)
178def normalize_model_name(model: str) -> str:
179 """Resolve a model id to a pricing key, or ``"default"`` when unknown.
181 Three changes from the substring match this replaces, one per finding:
183 * **Re-hosted ids are unwrapped first** (#941). See :func:`_bare_model_id`.
184 * **The key is the attribution base**, produced by
185 :func:`keel.agents.model_base` — the same function that writes the
186 ``model:<base>`` label onto a PR. #942's finding was that the pricing table
187 and the attribution convention were two vocabularies with nothing
188 connecting them, so keel priced its own runs at the fallback. They are one
189 vocabulary now, and :mod:`tests.test_cost_vocabulary` asserts it.
190 * **Matching is on token boundaries** (#943). A key matches only as a whole
191 run of ``-``-separated segments, so ``o1-mini`` no longer resolves to
192 ``o1`` — a 13.6x overcharge on a widely used model. Longest-first (#930)
193 still decides between two keys that both match.
195 An unrecognised model returns ``"default"`` rather than the raw string, so a
196 caller can tell "priced" from "guessed"; :func:`calculate_cost_report` counts
197 the guesses instead of quietly folding them into the total.
198 """
199 raw = _bare_model_id(model)
200 if not raw:
201 return "default"
202 for candidate in (raw, agents.model_base(raw), _family_root(raw)):
203 if candidate in MODEL_PRICING:
204 return candidate
205 aliased = MODEL_ALIASES.get(candidate)
206 if aliased in MODEL_PRICING:
207 return aliased
208 return "default"
211def _family_root(raw: str) -> str:
212 """``claude-opus-4-5`` -> ``claude-opus``: the id with its version run removed.
214 A naming rule, not a price guess. Every member of a vendor's named family
215 shares a tier, so enumerating `-4`, `-4-5`, `-4-8` in
216 :data:`MODEL_ALIASES` would be a list that goes stale on the next release —
217 which is the failure #942 is about, re-created one level down.
219 Stops at the first purely numeric segment, so ``gpt-4o`` (whose ``4o`` is not
220 numeric) and ``gemini-2.5-flash`` (whose price differs per member) are left
221 whole and fall through to ``default`` rather than borrowing a sibling's
222 price.
223 """
224 segments = raw.split("-")
225 head = list(itertools.takewhile(lambda part: not part.isdigit(), segments))
226 return "-".join(head) if head and len(head) < len(segments) else raw
229def estimate_token_cost(prompt_tokens: int, completion_tokens: int, model: str = "") -> float:
230 """Calculate estimated USD cost for token usage based on model pricing."""
231 key = normalize_model_name(model)
232 prompt_rate, completion_rate = MODEL_PRICING.get(key, DEFAULT_FALLBACK_PRICE)
233 cost = (prompt_tokens * prompt_rate + completion_tokens * completion_rate) / 1_000_000.0
234 return round(cost, 6)
237def estimate_benchmark_cost(prompt_tokens: int, completion_tokens: int) -> float:
238 """Calculate benchmark frontier model cost for computing tiered routing savings."""
239 p_rate, c_rate = FRONTIER_BENCHMARK_PRICE
240 return round((prompt_tokens * p_rate + completion_tokens * c_rate) / 1_000_000.0, 6)
243@dataclass(frozen=True)
244class CostReport:
245 total_runs: int
246 total_prompt_tokens: int
247 total_completion_tokens: int
248 total_tokens: int
249 total_cost_usd: float
250 estimated_savings_usd: float
251 model_breakdown: dict[str, dict[str, Any]]
252 top_performer: str | None
253 #: Runs whose model could not be priced. Their tokens and fallback cost are
254 #: in the totals; they are excluded from the savings figure, and this is the
255 #: number that says how much of the report is a guess (#944).
256 unpriced_runs: int = 0
257 #: Runs whose record carried its own token counts (#1359).
258 measured_runs: int = 0
259 #: Runs priced at :data:`ASSUMED_PROMPT_TOKENS` / :data:`ASSUMED_COMPLETION_TOKENS`
260 #: because their record carried none. ``measured_runs + estimated_runs ==
261 #: total_runs``.
262 estimated_runs: int = 0
264 @property
265 def token_basis(self) -> str:
266 """``none`` | ``measured`` | ``estimated`` | ``mixed`` — what the token totals rest on.
268 Keyed on ``measured_runs``, so a report that does not say how many runs were
269 measured reads as ``estimated``: the unflattering direction, as with #944's
270 unpriced runs.
271 """
272 if not self.total_runs:
273 return "none"
274 if not self.measured_runs:
275 return "estimated"
276 if self.measured_runs >= self.total_runs:
277 return "measured"
278 return "mixed"
280 def to_dict(self) -> dict[str, Any]:
281 return {
282 "total_runs": self.total_runs,
283 "total_prompt_tokens": self.total_prompt_tokens,
284 "total_completion_tokens": self.total_completion_tokens,
285 "total_tokens": self.total_tokens,
286 "total_cost_usd": round(self.total_cost_usd, 4),
287 "estimated_savings_usd": round(self.estimated_savings_usd, 4),
288 "model_breakdown": self.model_breakdown,
289 "top_performer": self.top_performer,
290 "unpriced_runs": self.unpriced_runs,
291 # Additive (#1359): every key above keeps its name and meaning.
292 "token_basis": self.token_basis,
293 "measured_runs": self.measured_runs,
294 "estimated_runs": self.estimated_runs,
295 "assumed_tokens_per_run": {
296 "prompt_tokens": ASSUMED_PROMPT_TOKENS,
297 "completion_tokens": ASSUMED_COMPLETION_TOKENS,
298 },
299 }
302def calculate_cost_report(records: list[dict[str, Any]]) -> CostReport:
303 """Aggregate token metrics and compute USD costs from activity records.
305 A record whose model cannot be priced is counted in the totals — the tokens
306 were spent either way — but excluded from ``estimated_savings_usd`` and
307 reported in ``unpriced_runs``.
309 Two reasons, both from #944. The savings figure is
310 ``frontier_benchmark - actual``, so *understating* a cost *inflates* the
311 saving: every pricing error propagated into the headline at double weight.
312 And a missing ``model`` used to default to ``gemini-2.5-flash``, one of the
313 cheapest entries — so **missing attribution read as maximum savings**, which
314 is the flattering direction, on a repo whose own ledger has ``model: None``
315 for every record.
316 """
317 total_prompt = 0
318 total_completion = 0
319 total_actual_cost = 0.0
320 priced_actual_cost = 0.0
321 total_benchmark_cost = 0.0
322 unpriced_runs = 0
323 estimated_runs = 0
324 models_data: dict[str, dict[str, Any]] = {}
326 for rec in records:
327 parts = _measured_parts(rec)
328 if not parts:
329 # No counts on the record: price it at the placeholder, and count it, so the
330 # report can say how much of itself is the placeholder (#1359).
331 parts = [(rec.get("model") or "", ASSUMED_PROMPT_TOKENS, ASSUMED_COMPLETION_TOKENS)]
332 estimated_runs += 1
334 run_unpriced = False
335 run_models: set[str] = set()
336 for model, p_tok, c_tok in parts:
337 total_prompt += p_tok
338 total_completion += c_tok
340 m_key = normalize_model_name(model)
341 cost = estimate_token_cost(p_tok, c_tok, model)
342 total_actual_cost += cost
343 if m_key == "default":
344 run_unpriced = True
345 else:
346 # Only a run whose real price is known can evidence a saving against
347 # the frontier benchmark.
348 priced_actual_cost += cost
349 total_benchmark_cost += estimate_benchmark_cost(p_tok, c_tok)
351 stats = models_data.setdefault(
352 m_key, {"runs": 0, "prompt_tokens": 0, "completion_tokens": 0, "cost_usd": 0.0}
353 )
354 # A model's `runs` counts the runs that used it, once each: a ship run whose
355 # implementer and reviewer both called the same model is one run of it.
356 if m_key not in run_models:
357 stats["runs"] += 1
358 run_models.add(m_key)
359 stats["prompt_tokens"] += p_tok
360 stats["completion_tokens"] += c_tok
361 stats["cost_usd"] = round(stats["cost_usd"] + cost, 4)
362 unpriced_runs += run_unpriced
364 total_tokens = total_prompt + total_completion
365 # Like with like: both sides of the subtraction cover exactly the priced
366 # runs. Benchmarking a subset against a total would understate the saving
367 # rather than inflate it, which is safer but still wrong.
368 savings = max(0.0, total_benchmark_cost - priced_actual_cost)
370 top_perf = None
371 if models_data:
372 # Top performer is model with most runs
373 top_perf = max(models_data.items(), key=lambda item: item[1]["runs"])[0]
375 return CostReport(
376 total_runs=len(records),
377 total_prompt_tokens=total_prompt,
378 total_completion_tokens=total_completion,
379 total_tokens=total_tokens,
380 total_cost_usd=round(total_actual_cost, 4),
381 estimated_savings_usd=round(savings, 4),
382 model_breakdown=models_data,
383 top_performer=top_perf,
384 unpriced_runs=unpriced_runs,
385 measured_runs=len(records) - estimated_runs,
386 estimated_runs=estimated_runs,
387 )
390def _measured_parts(rec: dict[str, Any]) -> list[tuple[str, int, int]]:
391 """Every ``(model, prompt_tokens, completion_tokens)`` a record carries; ``[]`` if none.
393 Two sources, both measured: the record's own top-level counts (any writer's, priced at
394 the record's ``model``), and each :data:`keel.activity.USAGE_FIELD` entry keel's
395 delegates recorded (#1373), priced at that call's model. A part whose counts are both
396 zero is not a measurement, as before; neither is a malformed entry — a record read
397 from disk has been validated, but this also takes records built in memory.
398 """
399 parts: list[tuple[str, int, int]] = []
400 p_tok = int(rec.get("prompt_tokens") or 0)
401 c_tok = int(rec.get("completion_tokens") or 0)
402 if p_tok or c_tok:
403 parts.append((rec.get("model") or "", p_tok, c_tok))
404 usage = rec.get(activity.USAGE_FIELD)
405 for entry in usage.values() if isinstance(usage, dict) else ():
406 if activity.usage_entry_issue(entry) is None and (
407 entry["prompt_tokens"] or entry["completion_tokens"]
408 ):
409 parts.append((entry["model"], entry["prompt_tokens"], entry["completion_tokens"]))
410 return parts
413def _token_basis_lines(report: CostReport) -> list[str]:
414 """The report's statement of what its token figures rest on (#1359).
416 Placed directly under the run count, above the first figure it qualifies, because a
417 caveat printed after the dollar amount is read after the dollar amount.
418 """
419 assumption = (
420 f"{ASSUMED_PROMPT_TOKENS:,} prompt / {ASSUMED_COMPLETION_TOKENS:,} completion "
421 "tokens per run"
422 )
423 indent = " " * 26
424 basis = report.token_basis
425 if basis == "none":
426 return []
427 # What a measured count covers, said wherever one is counted: keel records what a
428 # hosted-API delegate reports (#1373), and nothing of the agent host running the run.
429 scope = [
430 f"{indent}Measured counts are the ones the records carry: keel records what a",
431 f"{indent}hosted-API delegate reports, never an agent host's own tokens.",
432 ]
433 if basis == "measured":
434 whole = (
435 "the 1 run carries token counts"
436 if report.total_runs == 1
437 else f"all {report.total_runs} runs carry token counts"
438 )
439 return [f" Token Basis : measured ({whole})", *scope]
440 if basis == "estimated":
441 return [
442 f" Token Basis : ESTIMATED at {assumption}",
443 f"{indent}No record carries measured token counts, so every figure below rests",
444 f"{indent}on that placeholder. Read it as a run count, not a bill.",
445 ]
446 return [
447 f" Token Basis : {report.measured_runs} measured, "
448 f"{report.estimated_runs} ESTIMATED at {assumption}",
449 f"{indent}The estimated runs carry no token counts; their share of every figure",
450 f"{indent}below is that assumption, not a measurement.",
451 *scope,
452 ]
455def render_cost_report(report: CostReport) -> str:
456 """Render human-readable markdown / CLI report."""
457 lines = [
458 "Keel Token & Cost Report",
459 "────────────────────────────────────────────────────────",
460 f" Total Runs Tracked : {report.total_runs}",
461 *_token_basis_lines(report),
462 f" Total Tokens : {report.total_tokens:,} "
463 f"(Prompt: {report.total_prompt_tokens:,} / "
464 f"Completion: {report.total_completion_tokens:,})",
465 f" Estimated Spend (USD) : ${report.total_cost_usd:.4f}",
466 f" Estimated Savings : ${report.estimated_savings_usd:.4f} "
467 "(priced runs only, vs the frontier benchmark)",
468 ]
469 if report.unpriced_runs:
470 # Printed whenever it is non-zero, because the spend figure above is
471 # partly a fallback guess and the reader cannot tell from the number.
472 lines.append(
473 f" Unpriced Runs : {report.unpriced_runs} "
474 "(model not in the pricing table; excluded from savings)"
475 )
476 if report.top_performer:
477 lines.append(f" Top Dispatched Model : {report.top_performer}")
479 if report.model_breakdown:
480 lines.append("")
481 lines.append(" Model Breakdown:")
482 for m, stats in sorted(report.model_breakdown.items(), key=lambda x: -x[1]["runs"]):
483 lines.append(
484 f" - {m:<18}: {stats['runs']:>3} runs | "
485 f"{stats['prompt_tokens'] + stats['completion_tokens']:>9,} tokens | "
486 f"${stats['cost_usd']:.4f}"
487 )
489 return "\n".join(lines)
492def generate_cost_report(root: str | Path = ".") -> CostReport:
493 """Read all activity records from .keel/activity and compile CostReport."""
494 root_path = Path(root).resolve()
495 act_dir = root_path / activity.DEFAULT_ACTIVITY_DIR
496 records: list[dict[str, Any]] = []
498 if act_dir.exists() and act_dir.is_dir():
499 records = activity.read_all_activity(act_dir)
501 return calculate_cost_report(records)