Coverage for src/keel/cost.py: 100%

159 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""Keel token counts and USD cost estimates. 

2 

3Totals the token counts the activity records under ``.keel/activity`` carry — or a 

4placeholder where they carry none — and prices them from built-in model pricing tables, 

5without any external billing API. 

6 

7**The report says which of its figures are measured.** A record that carries a 

8non-zero ``prompt_tokens`` / ``completion_tokens``, or a 

9:data:`keel.activity.USAGE_FIELD` entry with non-zero counts, is *measured*; one that 

10carries neither is priced at :data:`ASSUMED_PROMPT_TOKENS` / 

11:data:`ASSUMED_COMPLETION_TOKENS` and counted as *estimated* (#1359) — a report whose 

12dollar figure is a run count times a constant must not read as a bill. 

13:class:`CostReport` carries the split, and both renderings state it. 

14 

15The one keel writer of counts is ``keel delegate run --activity-run-id`` (#1373): a 

16hosted-API delegate's reported usage, one entry per call, priced at that call's model. 

17A measured run is therefore measured *for its delegate calls*: an agent host's own 

18tokens, and a CLI delegate's, are never in a record, and the report says so. The dollar 

19figures stay estimates either way — they are keel's pricing table applied to counts, 

20not the provider's bill. 

21""" 

22 

23from __future__ import annotations 

24 

25import itertools 

26import re 

27from dataclasses import dataclass 

28from pathlib import Path 

29from typing import Any 

30 

31from . import activity, agents 

32from .agents import LOCAL_TRANSPORTS 

33 

34# Model pricing in USD per 1,000,000 tokens: (prompt_price_per_m, completion_price_per_m) 

35MODEL_PRICING: dict[str, tuple[float, float]] = { 

36 # Anthropic 

37 "claude-3-7-sonnet": (3.00, 15.00), 

38 "claude-3-5-sonnet": (3.00, 15.00), 

39 "claude-3-5-haiku": (0.80, 4.00), 

40 "claude-3-opus": (15.00, 75.00), 

41 "claude": (3.00, 15.00), 

42 # Google Gemini 

43 "gemini-2.5-pro": (1.25, 5.00), 

44 "gemini-2.5-flash": (0.15, 0.60), 

45 "gemini-1.5-pro": (1.25, 5.00), 

46 "gemini-1.5-flash": (0.075, 0.30), 

47 "gemini": (0.15, 0.60), 

48 # OpenAI 

49 "gpt-4o": (2.50, 10.00), 

50 "gpt-4o-mini": (0.15, 0.60), 

51 "o1": (15.00, 60.00), 

52 "o3-mini": (1.10, 4.40), 

53 "codex": (2.50, 10.00), 

54 # DeepSeek 

55 "deepseek-chat": (0.27, 1.10), 

56 "deepseek-reasoner": (0.55, 2.19), 

57 "deepseek": (0.27, 1.10), 

58 # Local 

59 "ollama": (0.00, 0.00), 

60 "local": (0.00, 0.00), 

61} 

62 

63DEFAULT_FALLBACK_PRICE = (1.00, 3.00) 

64 

65#: The token counts a record without its own is priced at: a placeholder per run, not a 

66#: measurement. Named, because the report has to name it — #1359 found the numbers 

67#: inlined in the aggregation loop and nowhere in the output, so a report built entirely 

68#: from them printed a dollar figure with nothing to say it was one. 

69ASSUMED_PROMPT_TOKENS = 1500 

70ASSUMED_COMPLETION_TOKENS = 400 

71FRONTIER_BENCHMARK_PRICE = (15.00, 75.00) # Used for computing savings vs Claude Opus/o1 

72 

73# #930's `_PRICING_KEYS_BY_LENGTH` is gone with the substring scan it ordered. 

74# Longest-first existed so `gpt-4o-mini` would not be captured by `gpt-4o`; 

75# exact resolution makes that impossible by construction, and a constant nothing 

76# reads is a comment pretending to be a guard. The property it protected is 

77# asserted on the outcome in `tests/test_cost.py`. 

78 

79 

80#: Cloud re-hosters put the vendor in front of an otherwise ordinary model id. 

81#: Bedrock uses ``anthropic.claude-…-v1:0``, Vertex 

82#: ``publishers/anthropic/models/…``, OpenRouter ``anthropic/claude-…``. The old 

83#: code split on the first ``:`` and kept the **right** side, so a Bedrock id 

84#: normalised to ``0`` — not an approximation, a total loss of the model name 

85#: (#941). 

86_REHOST_VENDORS = frozenset( 

87 { 

88 "anthropic", 

89 "openai", 

90 "google", 

91 "meta", 

92 "mistral", 

93 "cohere", 

94 "amazon", 

95 "ai21", 

96 "deepseek", 

97 "qwen", 

98 "x-ai", 

99 "perplexity", 

100 } 

101) 

102 

103#: A release stamp or revision that carries no pricing signal: ``-20240229``, 

104#: ``-v1``, ``-latest``, ``-preview``. 

105_MODEL_STAMP = re.compile(r"-(?:\d{6,8}|v\d+|latest|preview|exp)(?=-|$)") 

106 

107#: The bridge #942 asked for: the vocabulary :func:`keel.agents.model_base` 

108#: emits, mapped onto the pricing keys that already exist. 

109#: 

110#: keel writes a versionless ``model:<base>`` label on every PR — ``opus-4-8``, 

111#: ``sonnet-4-5`` — and **none** of those bases appears in ``MODEL_PRICING``, 

112#: whose keys are 2024-era product names. So `keel cost-report` priced keel's own 

113#: runs at the ``DEFAULT_FALLBACK_PRICE``: roughly 5 % of true Opus spend, with 

114#: the difference then claimed as savings (#944). 

115#: 

116#: **These are aliases onto prices already in the table, not new prices.** An 

117#: alias says "this label names that tier", which is a fact about naming and is 

118#: checkable. Whether the tier's *numbers* are still current is a separate, 

119#: operator-owned question — #942's "refresh the keys" — and inventing figures 

120#: here would bury that question under a plausible-looking table. 

121MODEL_ALIASES: dict[str, str] = { 

122 # Anthropic — keel's own attribution bases and the vendor's current ids. 

123 "opus": "claude-3-opus", 

124 "opus-4": "claude-3-opus", 

125 "opus-4-5": "claude-3-opus", 

126 "opus-4-8": "claude-3-opus", 

127 "claude-opus": "claude-3-opus", 

128 "claude-opus-4": "claude-3-opus", 

129 "sonnet": "claude-3-7-sonnet", 

130 "sonnet-4": "claude-3-7-sonnet", 

131 "sonnet-4-5": "claude-3-7-sonnet", 

132 "claude-sonnet": "claude-3-7-sonnet", 

133 "claude-sonnet-4": "claude-3-7-sonnet", 

134 "haiku": "claude-3-5-haiku", 

135 "haiku-4-5": "claude-3-5-haiku", 

136 "claude-haiku": "claude-3-5-haiku", 

137 # OpenAI. Only the CLI's own label, which names the same product as `codex`. 

138 "codex-cli": "codex", 

139} 

140 

141# Deliberately absent: `gpt-5 -> gpt-4o`, `gemini-2 -> gemini-2.5-pro` and 

142# friends. Those are not naming facts, they are price guesses across tiers — and 

143# the first draft of this map proved the point by resolving 

144# `gemini-2.5-flash-lite` to the *pro* price, 8x its own. An unknown model is 

145# reported as unpriced, which `calculate_cost_report` counts; a wrong tier is 

146# reported as a number, which nobody counts. 

147 

148 

149def _bare_model_id(model: str) -> str: 

150 """Strip re-hosting decoration down to the vendor's own model id. 

151 

152 Order matters: the path form is unwrapped before the colon is read, because 

153 a Bedrock id carries *both* (``anthropic.claude-3-opus-20240229-v1:0``). 

154 """ 

155 raw = model.lower().strip() 

156 # Local inference is free, and `MODEL_PRICING` prices the *tier* at 0.00 — 

157 # so `ollama:`/`local:` collapse to the transport on purpose. This is the one 

158 # place where pricing and attribution legitimately want different answers: 

159 # #955's label must name the model, this must name the free tier. 

160 if raw.startswith(tuple(f"{p}:" for p in LOCAL_TRANSPORTS)): 

161 return raw.split(":", 1)[0] 

162 # `<vendor>-api:model` is a transport prefix, and which names are transports 

163 # is defined once, in `agents` (#955). Sharing the definition is what stops 

164 # the pricing key and the attribution label drifting apart again. 

165 raw = agents.strip_transport(raw) 

166 if "/" in raw: # openrouter `vendor/model`, vertex `publishers/v/models/model` 

167 raw = raw.rsplit("/", 1)[1] 

168 if ":" in raw: 

169 head, tail = raw.split(":", 1) 

170 # `…-v1:0` is a Bedrock revision; `google:gemini-2.5-pro` is an 

171 # unrecognised re-hoster, whose model is still on the right. 

172 raw = head if tail.isdigit() else tail 

173 if "." in raw and raw.split(".", 1)[0] in _REHOST_VENDORS: 

174 raw = raw.split(".", 1)[1] 

175 return _MODEL_STAMP.sub("", raw) 

176 

177 

178def normalize_model_name(model: str) -> str: 

179 """Resolve a model id to a pricing key, or ``"default"`` when unknown. 

180 

181 Three changes from the substring match this replaces, one per finding: 

182 

183 * **Re-hosted ids are unwrapped first** (#941). See :func:`_bare_model_id`. 

184 * **The key is the attribution base**, produced by 

185 :func:`keel.agents.model_base` — the same function that writes the 

186 ``model:<base>`` label onto a PR. #942's finding was that the pricing table 

187 and the attribution convention were two vocabularies with nothing 

188 connecting them, so keel priced its own runs at the fallback. They are one 

189 vocabulary now, and :mod:`tests.test_cost_vocabulary` asserts it. 

190 * **Matching is on token boundaries** (#943). A key matches only as a whole 

191 run of ``-``-separated segments, so ``o1-mini`` no longer resolves to 

192 ``o1`` — a 13.6x overcharge on a widely used model. Longest-first (#930) 

193 still decides between two keys that both match. 

194 

195 An unrecognised model returns ``"default"`` rather than the raw string, so a 

196 caller can tell "priced" from "guessed"; :func:`calculate_cost_report` counts 

197 the guesses instead of quietly folding them into the total. 

198 """ 

199 raw = _bare_model_id(model) 

200 if not raw: 

201 return "default" 

202 for candidate in (raw, agents.model_base(raw), _family_root(raw)): 

203 if candidate in MODEL_PRICING: 

204 return candidate 

205 aliased = MODEL_ALIASES.get(candidate) 

206 if aliased in MODEL_PRICING: 

207 return aliased 

208 return "default" 

209 

210 

211def _family_root(raw: str) -> str: 

212 """``claude-opus-4-5`` -> ``claude-opus``: the id with its version run removed. 

213 

214 A naming rule, not a price guess. Every member of a vendor's named family 

215 shares a tier, so enumerating `-4`, `-4-5`, `-4-8` in 

216 :data:`MODEL_ALIASES` would be a list that goes stale on the next release — 

217 which is the failure #942 is about, re-created one level down. 

218 

219 Stops at the first purely numeric segment, so ``gpt-4o`` (whose ``4o`` is not 

220 numeric) and ``gemini-2.5-flash`` (whose price differs per member) are left 

221 whole and fall through to ``default`` rather than borrowing a sibling's 

222 price. 

223 """ 

224 segments = raw.split("-") 

225 head = list(itertools.takewhile(lambda part: not part.isdigit(), segments)) 

226 return "-".join(head) if head and len(head) < len(segments) else raw 

227 

228 

229def estimate_token_cost(prompt_tokens: int, completion_tokens: int, model: str = "") -> float: 

230 """Calculate estimated USD cost for token usage based on model pricing.""" 

231 key = normalize_model_name(model) 

232 prompt_rate, completion_rate = MODEL_PRICING.get(key, DEFAULT_FALLBACK_PRICE) 

233 cost = (prompt_tokens * prompt_rate + completion_tokens * completion_rate) / 1_000_000.0 

234 return round(cost, 6) 

235 

236 

237def estimate_benchmark_cost(prompt_tokens: int, completion_tokens: int) -> float: 

238 """Calculate benchmark frontier model cost for computing tiered routing savings.""" 

239 p_rate, c_rate = FRONTIER_BENCHMARK_PRICE 

240 return round((prompt_tokens * p_rate + completion_tokens * c_rate) / 1_000_000.0, 6) 

241 

242 

243@dataclass(frozen=True) 

244class CostReport: 

245 total_runs: int 

246 total_prompt_tokens: int 

247 total_completion_tokens: int 

248 total_tokens: int 

249 total_cost_usd: float 

250 estimated_savings_usd: float 

251 model_breakdown: dict[str, dict[str, Any]] 

252 top_performer: str | None 

253 #: Runs whose model could not be priced. Their tokens and fallback cost are 

254 #: in the totals; they are excluded from the savings figure, and this is the 

255 #: number that says how much of the report is a guess (#944). 

256 unpriced_runs: int = 0 

257 #: Runs whose record carried its own token counts (#1359). 

258 measured_runs: int = 0 

259 #: Runs priced at :data:`ASSUMED_PROMPT_TOKENS` / :data:`ASSUMED_COMPLETION_TOKENS` 

260 #: because their record carried none. ``measured_runs + estimated_runs == 

261 #: total_runs``. 

262 estimated_runs: int = 0 

263 

264 @property 

265 def token_basis(self) -> str: 

266 """``none`` | ``measured`` | ``estimated`` | ``mixed`` — what the token totals rest on. 

267 

268 Keyed on ``measured_runs``, so a report that does not say how many runs were 

269 measured reads as ``estimated``: the unflattering direction, as with #944's 

270 unpriced runs. 

271 """ 

272 if not self.total_runs: 

273 return "none" 

274 if not self.measured_runs: 

275 return "estimated" 

276 if self.measured_runs >= self.total_runs: 

277 return "measured" 

278 return "mixed" 

279 

280 def to_dict(self) -> dict[str, Any]: 

281 return { 

282 "total_runs": self.total_runs, 

283 "total_prompt_tokens": self.total_prompt_tokens, 

284 "total_completion_tokens": self.total_completion_tokens, 

285 "total_tokens": self.total_tokens, 

286 "total_cost_usd": round(self.total_cost_usd, 4), 

287 "estimated_savings_usd": round(self.estimated_savings_usd, 4), 

288 "model_breakdown": self.model_breakdown, 

289 "top_performer": self.top_performer, 

290 "unpriced_runs": self.unpriced_runs, 

291 # Additive (#1359): every key above keeps its name and meaning. 

292 "token_basis": self.token_basis, 

293 "measured_runs": self.measured_runs, 

294 "estimated_runs": self.estimated_runs, 

295 "assumed_tokens_per_run": { 

296 "prompt_tokens": ASSUMED_PROMPT_TOKENS, 

297 "completion_tokens": ASSUMED_COMPLETION_TOKENS, 

298 }, 

299 } 

300 

301 

302def calculate_cost_report(records: list[dict[str, Any]]) -> CostReport: 

303 """Aggregate token metrics and compute USD costs from activity records. 

304 

305 A record whose model cannot be priced is counted in the totals — the tokens 

306 were spent either way — but excluded from ``estimated_savings_usd`` and 

307 reported in ``unpriced_runs``. 

308 

309 Two reasons, both from #944. The savings figure is 

310 ``frontier_benchmark - actual``, so *understating* a cost *inflates* the 

311 saving: every pricing error propagated into the headline at double weight. 

312 And a missing ``model`` used to default to ``gemini-2.5-flash``, one of the 

313 cheapest entries — so **missing attribution read as maximum savings**, which 

314 is the flattering direction, on a repo whose own ledger has ``model: None`` 

315 for every record. 

316 """ 

317 total_prompt = 0 

318 total_completion = 0 

319 total_actual_cost = 0.0 

320 priced_actual_cost = 0.0 

321 total_benchmark_cost = 0.0 

322 unpriced_runs = 0 

323 estimated_runs = 0 

324 models_data: dict[str, dict[str, Any]] = {} 

325 

326 for rec in records: 

327 parts = _measured_parts(rec) 

328 if not parts: 

329 # No counts on the record: price it at the placeholder, and count it, so the 

330 # report can say how much of itself is the placeholder (#1359). 

331 parts = [(rec.get("model") or "", ASSUMED_PROMPT_TOKENS, ASSUMED_COMPLETION_TOKENS)] 

332 estimated_runs += 1 

333 

334 run_unpriced = False 

335 run_models: set[str] = set() 

336 for model, p_tok, c_tok in parts: 

337 total_prompt += p_tok 

338 total_completion += c_tok 

339 

340 m_key = normalize_model_name(model) 

341 cost = estimate_token_cost(p_tok, c_tok, model) 

342 total_actual_cost += cost 

343 if m_key == "default": 

344 run_unpriced = True 

345 else: 

346 # Only a run whose real price is known can evidence a saving against 

347 # the frontier benchmark. 

348 priced_actual_cost += cost 

349 total_benchmark_cost += estimate_benchmark_cost(p_tok, c_tok) 

350 

351 stats = models_data.setdefault( 

352 m_key, {"runs": 0, "prompt_tokens": 0, "completion_tokens": 0, "cost_usd": 0.0} 

353 ) 

354 # A model's `runs` counts the runs that used it, once each: a ship run whose 

355 # implementer and reviewer both called the same model is one run of it. 

356 if m_key not in run_models: 

357 stats["runs"] += 1 

358 run_models.add(m_key) 

359 stats["prompt_tokens"] += p_tok 

360 stats["completion_tokens"] += c_tok 

361 stats["cost_usd"] = round(stats["cost_usd"] + cost, 4) 

362 unpriced_runs += run_unpriced 

363 

364 total_tokens = total_prompt + total_completion 

365 # Like with like: both sides of the subtraction cover exactly the priced 

366 # runs. Benchmarking a subset against a total would understate the saving 

367 # rather than inflate it, which is safer but still wrong. 

368 savings = max(0.0, total_benchmark_cost - priced_actual_cost) 

369 

370 top_perf = None 

371 if models_data: 

372 # Top performer is model with most runs 

373 top_perf = max(models_data.items(), key=lambda item: item[1]["runs"])[0] 

374 

375 return CostReport( 

376 total_runs=len(records), 

377 total_prompt_tokens=total_prompt, 

378 total_completion_tokens=total_completion, 

379 total_tokens=total_tokens, 

380 total_cost_usd=round(total_actual_cost, 4), 

381 estimated_savings_usd=round(savings, 4), 

382 model_breakdown=models_data, 

383 top_performer=top_perf, 

384 unpriced_runs=unpriced_runs, 

385 measured_runs=len(records) - estimated_runs, 

386 estimated_runs=estimated_runs, 

387 ) 

388 

389 

390def _measured_parts(rec: dict[str, Any]) -> list[tuple[str, int, int]]: 

391 """Every ``(model, prompt_tokens, completion_tokens)`` a record carries; ``[]`` if none. 

392 

393 Two sources, both measured: the record's own top-level counts (any writer's, priced at 

394 the record's ``model``), and each :data:`keel.activity.USAGE_FIELD` entry keel's 

395 delegates recorded (#1373), priced at that call's model. A part whose counts are both 

396 zero is not a measurement, as before; neither is a malformed entry — a record read 

397 from disk has been validated, but this also takes records built in memory. 

398 """ 

399 parts: list[tuple[str, int, int]] = [] 

400 p_tok = int(rec.get("prompt_tokens") or 0) 

401 c_tok = int(rec.get("completion_tokens") or 0) 

402 if p_tok or c_tok: 

403 parts.append((rec.get("model") or "", p_tok, c_tok)) 

404 usage = rec.get(activity.USAGE_FIELD) 

405 for entry in usage.values() if isinstance(usage, dict) else (): 

406 if activity.usage_entry_issue(entry) is None and ( 

407 entry["prompt_tokens"] or entry["completion_tokens"] 

408 ): 

409 parts.append((entry["model"], entry["prompt_tokens"], entry["completion_tokens"])) 

410 return parts 

411 

412 

413def _token_basis_lines(report: CostReport) -> list[str]: 

414 """The report's statement of what its token figures rest on (#1359). 

415 

416 Placed directly under the run count, above the first figure it qualifies, because a 

417 caveat printed after the dollar amount is read after the dollar amount. 

418 """ 

419 assumption = ( 

420 f"{ASSUMED_PROMPT_TOKENS:,} prompt / {ASSUMED_COMPLETION_TOKENS:,} completion " 

421 "tokens per run" 

422 ) 

423 indent = " " * 26 

424 basis = report.token_basis 

425 if basis == "none": 

426 return [] 

427 # What a measured count covers, said wherever one is counted: keel records what a 

428 # hosted-API delegate reports (#1373), and nothing of the agent host running the run. 

429 scope = [ 

430 f"{indent}Measured counts are the ones the records carry: keel records what a", 

431 f"{indent}hosted-API delegate reports, never an agent host's own tokens.", 

432 ] 

433 if basis == "measured": 

434 whole = ( 

435 "the 1 run carries token counts" 

436 if report.total_runs == 1 

437 else f"all {report.total_runs} runs carry token counts" 

438 ) 

439 return [f" Token Basis : measured ({whole})", *scope] 

440 if basis == "estimated": 

441 return [ 

442 f" Token Basis : ESTIMATED at {assumption}", 

443 f"{indent}No record carries measured token counts, so every figure below rests", 

444 f"{indent}on that placeholder. Read it as a run count, not a bill.", 

445 ] 

446 return [ 

447 f" Token Basis : {report.measured_runs} measured, " 

448 f"{report.estimated_runs} ESTIMATED at {assumption}", 

449 f"{indent}The estimated runs carry no token counts; their share of every figure", 

450 f"{indent}below is that assumption, not a measurement.", 

451 *scope, 

452 ] 

453 

454 

455def render_cost_report(report: CostReport) -> str: 

456 """Render human-readable markdown / CLI report.""" 

457 lines = [ 

458 "Keel Token & Cost Report", 

459 "────────────────────────────────────────────────────────", 

460 f" Total Runs Tracked : {report.total_runs}", 

461 *_token_basis_lines(report), 

462 f" Total Tokens : {report.total_tokens:,} " 

463 f"(Prompt: {report.total_prompt_tokens:,} / " 

464 f"Completion: {report.total_completion_tokens:,})", 

465 f" Estimated Spend (USD) : ${report.total_cost_usd:.4f}", 

466 f" Estimated Savings : ${report.estimated_savings_usd:.4f} " 

467 "(priced runs only, vs the frontier benchmark)", 

468 ] 

469 if report.unpriced_runs: 

470 # Printed whenever it is non-zero, because the spend figure above is 

471 # partly a fallback guess and the reader cannot tell from the number. 

472 lines.append( 

473 f" Unpriced Runs : {report.unpriced_runs} " 

474 "(model not in the pricing table; excluded from savings)" 

475 ) 

476 if report.top_performer: 

477 lines.append(f" Top Dispatched Model : {report.top_performer}") 

478 

479 if report.model_breakdown: 

480 lines.append("") 

481 lines.append(" Model Breakdown:") 

482 for m, stats in sorted(report.model_breakdown.items(), key=lambda x: -x[1]["runs"]): 

483 lines.append( 

484 f" - {m:<18}: {stats['runs']:>3} runs | " 

485 f"{stats['prompt_tokens'] + stats['completion_tokens']:>9,} tokens | " 

486 f"${stats['cost_usd']:.4f}" 

487 ) 

488 

489 return "\n".join(lines) 

490 

491 

492def generate_cost_report(root: str | Path = ".") -> CostReport: 

493 """Read all activity records from .keel/activity and compile CostReport.""" 

494 root_path = Path(root).resolve() 

495 act_dir = root_path / activity.DEFAULT_ACTIVITY_DIR 

496 records: list[dict[str, Any]] = [] 

497 

498 if act_dir.exists() and act_dir.is_dir(): 

499 records = activity.read_all_activity(act_dir) 

500 

501 return calculate_cost_report(records)