Coverage for src/keel/artifacts.py: 100%

327 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""Canonical Markdown renderers for ship artifacts. 

2 

3These helpers keep public GitHub artifacts deterministic and consumer-neutral. 

4Adapters should post the rendered bodies verbatim instead of hand-writing PR 

5descriptions, review verdicts, jury verdicts, or extension result summaries. 

6""" 

7 

8from __future__ import annotations 

9 

10from typing import Any 

11 

12from . import evidence 

13 

14SCHEMA_VERSION = "keel.artifacts.v1" 

15EXTENSION_RESULT_MARKER = "<!-- keel.extension-result.v1 -->" 

16ISSUE_UPDATE_MARKER = "<!-- keel.issue-update.v1 -->" 

17STEP_HANDOFF_MARKER = "<!-- keel.step-handoff.v1 -->" 

18RUN_CONTROL_HALT_MARKER = "<!-- keel.run-control-halt.v1 -->" 

19REVIEW_CYCLE_SUMMARY_MARKER = "keel.review-cycle-summary.v1" 

20#: The verdict a jury verdict renders with when nobody supplied the panel's consensus 

21#: (#1429). It reads as **not** an approval — :func:`keel.evidence.jury_verdict_token` 

22#: takes its first word, ``PANEL_CONSENSUS`` — so a template posted unfilled holds the 

23#: merge with ``jury-verdict-not-approved`` instead of approving for a panel that may 

24#: have rejected the change. 

25JURY_CONSENSUS_PLACEHOLDER = ( 

26 "<PANEL_CONSENSUS \u2014 replace with the panel's APPROVE / REQUEST_CHANGES>" 

27) 

28COVERAGE_DELTA_MARKER = "keel.coverage-delta.v1" 

29DEPS_AUDIT_MARKER = "keel.deps-audit.v1" 

30FLAKE_AUDIT_MARKER = "keel.flake-audit.v1" 

31SCAN_FINDING_MARKER = "keel.scan-finding.v1" 

32TRIAGE_AUDIT_MARKER = "keel.triage-audit.v1" 

33 

34#: Severity buckets that drive the consolidated histogram + merge recommendation, 

35#: in must-fix → advisory order. ``critical`` folds into ``blocker`` (must-fix). 

36SEVERITY_ORDER = ("blocker", "major", "minor", "nit") 

37_SEVERITY_ALIASES = {"critical": "blocker"} 

38 

39#: Dependency-advisory severity buckets, most → least severe. 

40DEPS_SEVERITY_ORDER = ("critical", "high", "moderate", "low") 

41 

42 

43def contract_as_dict() -> dict[str, Any]: 

44 """Return the canonical artifact renderer contract for ship-like flows.""" 

45 return { 

46 "schema_version": SCHEMA_VERSION, 

47 "consumer_neutral": True, 

48 "deterministic": True, 

49 "renderers": { 

50 "pr_body": "keel.artifacts.render_pr_body", 

51 "issue_update": "keel.artifacts.render_issue_update", 

52 "review_verdict": "keel.artifacts.render_review_verdict", 

53 "jury_verdict": "keel.artifacts.render_jury_verdict", 

54 "review_cycle_summary": "keel.artifacts.render_review_cycle_summary", 

55 "extension_result": "keel.artifacts.render_extension_result", 

56 "step_handoff": "keel.artifacts.render_step_handoff", 

57 "run_control_halt": "keel.artifacts.render_run_control_halt", 

58 "ship_provenance": "keel.artifacts.render_ship_provenance", 

59 }, 

60 "markers": { 

61 "review_verdict": evidence.REVIEW_VERDICT_MARKER, 

62 "jury_verdict": evidence.JURY_VERDICT_MARKER, 

63 "ship_provenance": evidence.SHIP_PROVENANCE_MARKER, 

64 "review_cycle_summary": REVIEW_CYCLE_SUMMARY_MARKER, 

65 "issue_update": ISSUE_UPDATE_MARKER, 

66 "extension_result": EXTENSION_RESULT_MARKER, 

67 "step_handoff": STEP_HANDOFF_MARKER, 

68 "run_control_halt": RUN_CONTROL_HALT_MARKER, 

69 }, 

70 "adapter_rule": "post rendered markdown verbatim when available", 

71 } 

72 

73 

74def render_pr_body( 

75 *, 

76 issue_number: int | None = None, 

77 issue_intake: dict[str, Any] | None = None, 

78 changed_files: list[str] | tuple[str, ...] | None = (), 

79 testing: list[str] | tuple[str, ...] = (), 

80 fix_evidence: list[str] | tuple[str, ...] | None = (), 

81 docs_impact: str | None = None, 

82) -> str: 

83 """Render the canonical PR body used by ship implementers.""" 

84 intake = issue_intake if isinstance(issue_intake, dict) else {} 

85 lines = [ 

86 "## Summary", 

87 f"- {_value(intake.get('deliverable'), 'Implement the requested change.')}", 

88 "", 

89 "## Context / Root Cause", 

90 _value(intake.get("objective"), "See the linked issue for context."), 

91 "", 

92 "## Changes Made", 

93 ] 

94 # `None` is "git could not be read", which must not render as "nothing changed". 

95 if changed_files is None: 

96 lines.append("- The changed-file list could not be read from git.") 

97 else: 

98 files = [file for file in changed_files if isinstance(file, str)] 

99 if files: 

100 lines.extend(f"- Updated `{file}`." for file in files) 

101 else: 

102 lines.append("- No changed files recorded yet.") 

103 lines.extend(["", "## Testing"]) 

104 tests = [item for item in testing if isinstance(item, str) and item.strip()] 

105 lines.extend(f"- {item.strip()}" for item in tests) if tests else lines.append( 

106 "- Not run yet; update this section before marking the PR ready." 

107 ) 

108 # A fix's own section, rendered as a prompt rather than a claim. Coverage cannot say 

109 # whether a test *guards* a change — `fail_under = 100` is enforced, so "maintained 100% 

110 # coverage" is true of every merged PR before it is written. An audit of 14 closed fixes 

111 # found three whose tests passed with the fix removed, all three offering coverage as 

112 # evidence (#1289). The unit is the behaviour, not the git hunk: #871's guarded and 

113 # unguarded arms shared one hunk, so a per-hunk claim passed while half the fix was 

114 # unpinned. No caller supplies `fix_evidence` yet — the only one, 

115 # `contracts.ship_result_as_dict`, does not pass it — so from `keel ship --json` this is 

116 # always the prompt, and the 

117 # implementer replaces it. The parameter exists so a future caller can. 

118 lines.extend(["", "## Fix evidence"]) 

119 evidence = [item for item in fix_evidence or () if isinstance(item, str) and item.strip()] 

120 if evidence: 

121 lines.extend(f"- {item.strip()}" for item in evidence) 

122 else: 

123 lines.append( 

124 "- Not stated yet. For each behaviour this change touches — each arm of a " 

125 "conditional, each call site — name a test that fails as an assertion when that " 

126 "one change is reverted, and list any behaviour left unpinned with the reason. " 

127 "`N/A — <docs | pure refactor | dependency bump | packaging>` if there is nothing " 

128 "to revert-test." 

129 ) 

130 lines.extend( 

131 [ 

132 "", 

133 "## Docs Impact", 

134 _value(docs_impact, "Docs Impact: none — no operator-facing behavior changed."), 

135 "", 

136 _closing_reference(issue_number), 

137 ] 

138 ) 

139 return "\n".join(lines).rstrip() + "\n" 

140 

141 

142def render_issue_update( 

143 *, 

144 issue_number: int | None = None, 

145 pull_request: int | None = None, 

146 status: str = "in-progress", 

147 summary: str | None = None, 

148 next_step: str | None = None, 

149) -> str: 

150 """Render a stable issue progress/update comment.""" 

151 lines = [ 

152 ISSUE_UPDATE_MARKER, 

153 "", 

154 "## Ship update", 

155 "", 

156 f"- **Issue:** {_issue(issue_number)}", 

157 f"- **Pull request:** {_pr(pull_request)}", 

158 f"- **Status:** {_value(status, 'in-progress')}", 

159 f"- **Summary:** {_value(summary, 'No summary recorded.')}", 

160 f"- **Next step:** {_value(next_step, 'Continue the ship workflow.')}", 

161 ] 

162 return "\n".join(lines) + "\n" 

163 

164 

165def render_review_verdict( 

166 *, 

167 reviewer: str, 

168 head_sha: str | None, 

169 verdict: str = "ABSTAIN", 

170 scope: str | None = None, 

171 findings: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

172 testing: str | None = None, 

173 vendor: str | None = None, 

174 model: str | None = None, 

175) -> str: 

176 """Render a head-bound reviewer verdict comment accepted by evidence verification. 

177 

178 When ``vendor`` (and optionally ``model``) is supplied, structured 

179 ``vendor:`` / ``model:`` provenance lines are emitted so evidence 

180 verification can enforce vendor distinctness across required verdicts. The 

181 fields use the same vendor/model conventions as ``keel.provenance`` and are 

182 omitted entirely when not supplied, so the default rendering is unchanged. 

183 

184 **Pass a real ``scope`` or real ``findings``.** The defaults — "Full 

185 changed-file diff and relevant contracts" and "none" — name nothing, and 

186 :func:`keel.evidence.verdict_substance` refuses a verdict that names nothing 

187 (#926). That is deliberate: 75 of 75 verdicts across 25 pull requests were 

188 this template with the defaults left in, and the gate could not tell them 

189 apart from a review that caught a blocker. Any one of these is enough: 

190 

191 * a path (``src/keel/evidence.py``), a ``file.py:42``, a backticked token, 

192 or a called ``module.function()``; 

193 * **two** of the unbackticked forms — a bare filename; a dotted 

194 ``module.symbol`` that carries a mark prose does not use (an underscore, 

195 an internal capital, a run of capitals, or a capitalised segment), so 

196 ``Config.parse`` and ``cache.cache_key`` read and ``foo.bar`` does not; 

197 or a lowercase ``snake_case`` identifier. One alone does not count, because ``Node.js`` and 

198 ``evidence.py`` are spelled the same way and so are ``GitHub.com`` and 

199 ``Config.parse``; naming two things is what a review does and a mention 

200 does not; 

201 * a "Checked X, Y and Z" clause. That one verb keeps a free-form object, 

202 because it predates the rule and the corpus has real reviews under it 

203 naming their objects in English ("Checked the formula syntax, the 

204 version URL and the checksum placeholder"). #1106 tried to widen it to 

205 traced/read/ran/inspected/verified; those could not keep a free-form 

206 object without readmitting the receipt, and requiring their object to 

207 name something made the branch decide nothing at all — the object is 

208 part of the prose, which already takes that test. So they are ordinary 

209 prose: name two things, or one in backticks. 

210 

211 A genuinely clean review stays expressible; it just has to say what it 

212 looked at. 

213 

214 ``verdict`` is written as given on the ``Verdict:`` line, and that line is what the 

215 gate reads (#1426): the verdict counts toward a merge only when its first word is in 

216 :data:`keel.evidence.APPROVING_VERDICTS` (``APPROVE`` / ``LGTM`` / ``PASS``), and one 

217 that requests changes holds the merge. Keep the line; a verdict without it is not an 

218 approval. 

219 """ 

220 lines = [ 

221 evidence.REVIEW_VERDICT_MARKER, 

222 f"reviewer: {_slug(reviewer)}", 

223 f"head: {_value(head_sha, '<head-sha>')}", 

224 ] 

225 if isinstance(vendor, str) and vendor.strip(): 

226 lines.append(f"vendor: {_slug(vendor)}") 

227 if isinstance(model, str) and model.strip(): 

228 lines.append(f"model: {_slug(model)}") 

229 lines.extend( 

230 [ 

231 "", 

232 f"Verdict: {_value(verdict, 'ABSTAIN')}", 

233 "", 

234 f"Scope reviewed: {_value(scope, 'Full changed-file diff and relevant contracts.')}", 

235 "", 

236 "Findings:", 

237 ] 

238 ) 

239 lines.extend(_finding_lines(findings)) 

240 lines.extend(["", f"Testing noted: {_value(testing, 'See PR Testing section.')}"]) 

241 return "\n".join(lines) + "\n" 

242 

243 

244def render_jury_verdict( 

245 *, 

246 head_sha: str | None, 

247 participants: list[str] | tuple[str, ...] = (), 

248 verdict: str | None = "ABSTAIN", 

249 findings_summary: list[str] | tuple[str, ...] = (), 

250 remaining_risks: str | None = None, 

251 participating_vendors: int | None = None, 

252 panelists: int | None = None, 

253) -> str: 

254 """Render a head-bound jury verdict comment accepted by evidence verification. 

255 

256 The verdict declares ``vendors: <N>`` — the distinct vendors that actually 

257 took part. That line is the only channel by which the vendor count reaches a 

258 CI evidence check: the run ledger and the jury artifact both live under the 

259 gitignored ``.keel/state/``, so a hosted runner cannot read them, while PR 

260 comments are always visible. When ``participating_vendors`` is omitted it is 

261 inferred from ``participants``, so a caller that already lists them does not 

262 have to count twice. 

263 

264 ``panelists: <N>`` travels the same channel for the same reason (#1015). When 

265 the panel **is** the review, the number of ballots is the reviewer count the 

266 evidence gate has to require, and it is knowable only once the panel has run. 

267 An undeclared panel size leaves the gate on its floor (the minimum vendor 

268 count) rather than requiring nothing, so omitting it fails closed. Omitted, 

269 it is inferred from ``participants``. 

270 

271 ``verdict`` is the panel's consensus, written on the ``AI Jury verdict:`` line the 

272 evidence gate reads (#1429). Missing or blank, it renders ``ABSTAIN`` — a panel that 

273 stated no consensus did not approve — where it used to render ``LGTM``, an approval 

274 nobody gave. 

275 """ 

276 people = [ 

277 person.strip() for person in participants if isinstance(person, str) and person.strip() 

278 ] 

279 vendors = participating_vendors if participating_vendors is not None else len(people) 

280 seats = panelists if panelists is not None else len(people) 

281 lines = [ 

282 evidence.JURY_VERDICT_MARKER, 

283 f"head: {_value(head_sha, '<head-sha>')}", 

284 f"vendors: {vendors}", 

285 f"panelists: {seats}", 

286 "", 

287 f"AI Jury verdict: {_value(verdict, 'ABSTAIN')}.", 

288 "", 

289 f"Participants: {', '.join(people) if people else 'not recorded'}.", 

290 "", 

291 JURY_SUMMARY_HEADING, 

292 ] 

293 # One line per item, whitespace collapsed: :func:`jury_verdict_summary` reads the 

294 # summary back line by line (#1437), so a message carrying its own newline must not 

295 # end the item early — or start a line that reads as an item of its own. 

296 summaries = [ 

297 " ".join(item.split()) 

298 for item in findings_summary 

299 if isinstance(item, str) and item.strip() 

300 ] 

301 lines.extend(f"- {item}" for item in summaries) if summaries else lines.append("- none") 

302 lines.extend(["", f"{JURY_RISKS_LABEL} {_value(remaining_risks, 'none identified')}."]) 

303 return "\n".join(lines) + "\n" 

304 

305 

306#: The two labels that bracket a jury verdict's findings summary, as 

307#: :func:`render_jury_verdict` writes them and :func:`jury_verdict_summary` reads them. 

308JURY_SUMMARY_HEADING = "Findings summary:" 

309JURY_RISKS_LABEL = "Remaining risks:" 

310 

311 

312def jury_verdict_summary(body: str) -> tuple[str, ...] | None: 

313 """The findings-summary items of a jury verdict :func:`render_jury_verdict` wrote (#1437). 

314 

315 The inverse of the renderer, written against it: every ``- <item>`` line between 

316 :data:`JURY_SUMMARY_HEADING` and :data:`JURY_RISKS_LABEL`, with the renderer's 

317 ``- none`` read as no items. A verdict keel rendered from a panel lists its verified 

318 findings here as ``<severity>: <message>`` (:func:`keel.jury.jury_verdict`), which is 

319 how ``keel ship`` reuses the head's posted panel without convening another. 

320 

321 Read to the risks line rather than to the first blank line, so the parse can only ever 

322 see *more* of the comment than the renderer put in the summary — never stop short of a 

323 finding. A body without the heading is not one keel rendered, and its summary cannot be 

324 read: ``None``, which is never the same answer as "no findings". 

325 """ 

326 lines = body.splitlines() 

327 try: 

328 start = lines.index(JURY_SUMMARY_HEADING) + 1 

329 except ValueError: 

330 return None 

331 items: list[str] = [] 

332 for line in lines[start:]: 

333 if line.startswith(JURY_RISKS_LABEL): 

334 return () if items == ["none"] else tuple(items) 

335 if line.startswith("- "): 

336 items.append(line[2:].strip()) 

337 # No closing risks line: not the shape the renderer writes, so not a summary keel can 

338 # vouch for — unreadable, and the caller convenes the panel instead of reusing it. 

339 return None 

340 

341 

342def render_ship_provenance( 

343 *, 

344 run_id: str | None = None, 

345 issue: int | None = None, 

346 head_sha: str | None = None, 

347 implementer_attribution: dict[str, Any] | None = None, 

348) -> str: 

349 """Render the ship-provenance comment a live run posts on its own PR (#1013). 

350 

351 This comment is the run stamping *itself*: it says which ship run produced the 

352 PR, for which issue, at which head, and — verbatim from 

353 :func:`keel.agents.attribution` — what the implementer's attribution labels are. 

354 :func:`keel.evidence.gate_decision` arms the evidence gate on the marker ahead of 

355 the branch-name regex, so a ship run whose branch is named anything at all still 

356 reads as a keel run instead of as an unreviewed drive-by PR. 

357 

358 ``implementer_attribution`` is the dict :func:`keel.agents.attribution` (or 

359 :func:`keel.agents.profile_attribution`) returns. Pass it through unchanged: the 

360 whole point of the artifact is that the labels are *core's*, not prose's. 

361 """ 

362 record = implementer_attribution if isinstance(implementer_attribution, dict) else {} 

363 lines = [ 

364 evidence.SHIP_PROVENANCE_MARKER, 

365 f"run-id: {_value(run_id, 'not recorded')}", 

366 f"issue: {_issue(issue)}", 

367 f"head: {_value(head_sha, '<head-sha>')}", 

368 f"agent-label: {_value(record.get('agent_label'), 'not recorded')}", 

369 f"model-label: {_value(record.get('model_label'), 'not recorded')}", 

370 f"system: {_value(record.get('system'), 'not recorded')}", 

371 ] 

372 profile = record.get("delegate_profile") 

373 if isinstance(profile, str) and profile.strip(): 

374 lines.append(f"delegate-profile: {profile.strip()}") 

375 lines.extend( 

376 [ 

377 "", 

378 ( 

379 "Provenance stamp for a keel run: this pull request came out of the backbone, " 

380 "so the evidence gate applies to it." 

381 ), 

382 "", 

383 ( 

384 "The attribution labels above come from `keel attribution` — apply them to the " 

385 "PR verbatim rather than composing them by hand." 

386 ), 

387 ] 

388 ) 

389 return "\n".join(lines) + "\n" 

390 

391 

392def render_review_cycle_summary( 

393 *, 

394 reviewers: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

395 head_sha: str | None = None, 

396 run_id: str | None = None, 

397) -> str: 

398 """Render the deterministic multi-reviewer review-cycle summary comment. 

399 

400 Emits one section per reviewer (codename · focus · verdict · a 

401 ``Severity | File:Line | Description | Suggested Fix`` finding table) followed 

402 by a consolidated summary whose severity histogram — not the verdict strings — 

403 is the source of truth for the merge recommendation. The output is byte-stable 

404 for a given input so the orchestrator posts it verbatim instead of improvising 

405 a layout. When ``run_id`` is supplied an invisible ``keel.run-id`` marker is 

406 appended so an idempotent re-post edits the existing comment in place. 

407 """ 

408 clean = [reviewer for reviewer in reviewers if isinstance(reviewer, dict)] 

409 lines = [ 

410 REVIEW_CYCLE_SUMMARY_MARKER, 

411 f"head: {_value(head_sha, '<head-sha>')}", 

412 "", 

413 ] 

414 for index, reviewer in enumerate(clean): 

415 if index: 

416 lines.extend(["", "---", ""]) 

417 lines.extend(_cycle_reviewer_lines(reviewer)) 

418 if clean: 

419 lines.extend(["", "---", ""]) 

420 lines.extend(_cycle_summary_lines(clean)) 

421 if isinstance(run_id, str) and run_id.strip(): 

422 lines.extend(["", f"<!-- keel.run-id: {run_id.strip()} -->"]) 

423 return "\n".join(lines) + "\n" 

424 

425 

426def render_coverage_delta( 

427 *, 

428 codename: str, 

429 base_sha: str | None = None, 

430 head_sha: str | None = None, 

431 areas: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

432) -> str: 

433 """Render the deterministic per-PR coverage delta comment. 

434 

435 The literal first line is the caller-supplied ``codename`` (e.g. 

436 ``COVERAGE-<PR>-<UTC>``) — the load-bearing anchor the adapter finds by prefix 

437 to update the comment in place, so nothing precedes it. The adapter supplies 

438 the timestamped codename, so the renderer stays pure and byte-stable for a 

439 given input. 

440 """ 

441 lines = [ 

442 codename, 

443 "", 

444 f"Coverage delta: base@{_value(base_sha, '<base>')} → head@{_value(head_sha, '<head>')}", 

445 ] 

446 for area in areas: 

447 if isinstance(area, dict): 

448 lines.extend(["", *_coverage_area_lines(area)]) 

449 lines.extend(["", f"<!-- {COVERAGE_DELTA_MARKER} -->"]) 

450 return "\n".join(lines) + "\n" 

451 

452 

453def render_deps_audit( 

454 *, 

455 codename: str, 

456 ecosystems: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

457 licences: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

458 skipped: list[str] | tuple[str, ...] = (), 

459 security_only: bool = False, 

460) -> str: 

461 """Render the deterministic dependency-audit comment for the tracking issue. 

462 

463 The literal first line is the caller-supplied ``codename`` (e.g. 

464 ``DEPS-AUDIT-<DATE>-<UTC>``); a fresh comment is appended per run, found later 

465 by that prefix. Under ``security_only`` the licence-drift section is omitted. 

466 """ 

467 counts = _deps_counts(ecosystems) 

468 lines = [ 

469 codename, 

470 "", 

471 " | ".join(f"{severity}: {counts[severity]}" for severity in DEPS_SEVERITY_ORDER), 

472 ] 

473 for ecosystem in ecosystems: 

474 if isinstance(ecosystem, dict): 

475 lines.extend(["", *_deps_ecosystem_lines(ecosystem)]) 

476 if not security_only: 

477 lines.extend(["", *_deps_licence_lines(licences)]) 

478 skipped_items = _string_list(skipped) 

479 if skipped_items: 

480 lines.extend(["", "## Skipped", ""]) 

481 lines.extend(f"- {item}" for item in skipped_items) 

482 lines.extend(["", f"<!-- {DEPS_AUDIT_MARKER} -->"]) 

483 return "\n".join(lines) + "\n" 

484 

485 

486def render_flake_audit( 

487 *, 

488 codename: str, 

489 summary: dict[str, Any] | None = None, 

490 new_flakes: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

491 tracked: list[dict[str, Any]] | tuple[dict[str, Any], ...] = (), 

492 limitations: list[str] | tuple[str, ...] = (), 

493) -> str: 

494 """Render the deterministic flake-audit report comment. 

495 

496 The literal first line is the caller-supplied ``codename`` (e.g. 

497 ``FLAKE-AUDIT-<DATE>-<UTC>``). Newly-classified flakes render as a table (or a 

498 single italic line when none cleared the threshold); already-tracked flakes 

499 and honest limitations render only when present. 

500 """ 

501 stats = summary if isinstance(summary, dict) else {} 

502 lines = [ 

503 codename, 

504 "", 

505 ( 

506 f"runs examined: {_count(stats.get('runs'))} · " 

507 f"distinct failing tests: {_count(stats.get('distinct'))} · " 

508 f"classified flakes: {_count(stats.get('classified'))} · " 

509 f"newly-opened issues: {_count(stats.get('opened'))}" 

510 ), 

511 "", 

512 "## Newly classified flakes", 

513 "", 

514 ] 

515 flakes = _dict_list(new_flakes) 

516 if flakes: 

517 lines.append("| Test | Fail rate | Failures | Sample runs | Signature |") 

518 lines.append("| --- | --- | --- | --- | --- |") 

519 lines.extend( 

520 _table_row( 

521 [ 

522 _cell(_value(flake.get("test"), "—")), 

523 _cell(_value(flake.get("fail_rate"), "—")), 

524 _cell(str(_count(flake.get("failures")))), 

525 _cell(", ".join(_string_list(flake.get("samples"))) or "—"), 

526 _cell(_value(flake.get("signature"), "—")), 

527 ] 

528 ) 

529 for flake in flakes 

530 ) 

531 else: 

532 lines.append("_no new flakes above threshold_") 

533 tracked_rows = _dict_list(tracked) 

534 if tracked_rows: 

535 lines.extend(["", "## Already tracked (deduped)", ""]) 

536 lines.extend( 

537 f"- {_value(row.get('test'), '—')} — see {_issue_ref(row.get('issue'))}" 

538 for row in tracked_rows 

539 ) 

540 limitation_items = _string_list(limitations) 

541 if limitation_items: 

542 lines.extend(["", "## Limitations", ""]) 

543 lines.extend(f"- {item}" for item in limitation_items) 

544 lines.extend(["", f"<!-- {FLAKE_AUDIT_MARKER} -->"]) 

545 return "\n".join(lines) + "\n" 

546 

547 

548def _table_row(cells: list[str]) -> str: 

549 return "| " + " | ".join(cells) + " |" 

550 

551 

552def _dict_list(raw: Any) -> list[dict[str, Any]]: 

553 if not isinstance(raw, (list, tuple)): 

554 return [] 

555 return [item for item in raw if isinstance(item, dict)] 

556 

557 

558def _count(value: Any) -> int: 

559 return value if isinstance(value, int) and not isinstance(value, bool) else 0 

560 

561 

562def _issue_ref(value: Any) -> str: 

563 if isinstance(value, int) and not isinstance(value, bool): 

564 return f"#{value}" 

565 return _value(value, "?") 

566 

567 

568def _scalar(value: Any, fallback: str) -> str: 

569 if isinstance(value, bool): 

570 return fallback 

571 if isinstance(value, int): 

572 return str(value) 

573 return _value(value, fallback) 

574 

575 

576def _is_number(value: Any) -> bool: 

577 return isinstance(value, (int, float)) and not isinstance(value, bool) 

578 

579 

580def _fmt_pct(value: Any) -> str: 

581 return f"{value:.1f}%" if _is_number(value) else "—" 

582 

583 

584def _fmt_delta(base: Any, head: Any) -> tuple[str, bool]: 

585 if _is_number(base) and _is_number(head): 

586 delta = head - base 

587 return f"{delta:+.1f}%", abs(delta) >= 0.5 

588 return "—", False 

589 

590 

591def _coverage_row(unit: str, base: Any, head: Any, files: Any, *, is_overall: bool = False) -> str: 

592 delta_text, bold = _fmt_delta(base, head) 

593 files_text = _cell(str(files)) if str(files).strip() else "" 

594 cells = [_cell(unit), _fmt_pct(base), _fmt_pct(head), delta_text, files_text] 

595 if bold: 

596 cells = [f"**{cell}**" if cell else "" for cell in cells] 

597 elif is_overall: 

598 cells[0] = f"**{cells[0]}**" 

599 return _table_row(cells) 

600 

601 

602def _coverage_area_lines(area: dict[str, Any]) -> list[str]: 

603 name = _value(area.get("name"), "area") 

604 if area.get("skipped"): 

605 return [f"_{name} coverage skipped: {_value(area.get('skip_reason'), 'not run')}_"] 

606 lines = [ 

607 f"## {name}", 

608 "", 

609 "| Unit | Base % | Head % | Δ | Files |", 

610 "| --- | --- | --- | --- | --- |", 

611 ] 

612 lines.extend( 

613 _coverage_row( 

614 _value(row.get("unit"), "—"), row.get("base"), row.get("head"), row.get("files", "") 

615 ) 

616 for row in _dict_list(area.get("rows")) 

617 ) 

618 overall = area.get("overall") 

619 if isinstance(overall, dict): 

620 lines.append( 

621 _coverage_row("overall", overall.get("base"), overall.get("head"), "", is_overall=True) 

622 ) 

623 return lines 

624 

625 

626def _deps_sev_rank(severity: str) -> int: 

627 lowered = severity.lower() 

628 return ( 

629 DEPS_SEVERITY_ORDER.index(lowered) 

630 if lowered in DEPS_SEVERITY_ORDER 

631 else len(DEPS_SEVERITY_ORDER) 

632 ) 

633 

634 

635def _deps_counts(ecosystems: Any) -> dict[str, int]: 

636 counts = dict.fromkeys(DEPS_SEVERITY_ORDER, 0) 

637 for ecosystem in _dict_list(ecosystems): 

638 for finding in _dict_list(ecosystem.get("findings")): 

639 severity = _value(finding.get("severity"), "low").lower() 

640 if severity in counts: 

641 counts[severity] += 1 

642 return counts 

643 

644 

645def _deps_ecosystem_lines(ecosystem: dict[str, Any]) -> list[str]: 

646 name = _value(ecosystem.get("name"), "ecosystem") 

647 findings = _dict_list(ecosystem.get("findings")) 

648 if not findings: 

649 threshold = _value(ecosystem.get("threshold"), "low") 

650 return [f"_No {name} findings at or above {threshold} severity._"] 

651 findings = sorted(findings, key=lambda f: _deps_sev_rank(_value(f.get("severity"), "low"))) 

652 lines = [ 

653 f"## {name}", 

654 "", 

655 "| Package | Version | Severity | Advisory | Fix available |", 

656 "| --- | --- | --- | --- | --- |", 

657 ] 

658 lines.extend( 

659 _table_row( 

660 [ 

661 _cell(_value(finding.get("package"), "—")), 

662 _cell(_value(finding.get("version"), "—")), 

663 _cell(_value(finding.get("severity"), "low")), 

664 _cell(_value(finding.get("advisory"), "—")), 

665 _cell(_value(finding.get("fix_available"), "—")), 

666 ] 

667 ) 

668 for finding in findings 

669 ) 

670 return lines 

671 

672 

673def _deps_licence_lines(licences: Any) -> list[str]: 

674 rows = _dict_list(licences) 

675 if not rows: 

676 return ["_licences: no drift_"] 

677 lines = [ 

678 "## Licences", 

679 "", 

680 "| Status | Package | Baseline | Current |", 

681 "| --- | --- | --- | --- |", 

682 ] 

683 lines.extend( 

684 _table_row( 

685 [ 

686 _cell(_value(row.get("status"), "—")), 

687 _cell(_value(row.get("package"), "—")), 

688 _cell(_value(row.get("baseline"), "—")), 

689 _cell(_value(row.get("current"), "—")), 

690 ] 

691 ) 

692 for row in rows 

693 ) 

694 return lines 

695 

696 

697def render_scan_finding_issue( 

698 *, 

699 problem: str | None = None, 

700 location: str | None = None, 

701 severity: str | None = None, 

702 justification: str | None = None, 

703 evidence: str | None = None, 

704 suggested_fix: str | None = None, 

705 source: str | None = None, 

706 regression_of: int | None = None, 

707) -> str: 

708 """Render the deterministic issue body for a scan finding. 

709 

710 Shared by ``regression`` and ``review-all-day`` — the body carries the 

711 problem statement, ``path:line`` location, severity + justification, fenced 

712 evidence, and suggested fix, plus a provenance marker. When ``regression_of`` 

713 is supplied the grep-able ``regression-of: #N`` cross-reference is the body's 

714 literal last line. 

715 """ 

716 lines = [ 

717 "## Problem", 

718 "", 

719 _value(problem, "A scan finding was reported without a problem statement."), 

720 "", 

721 "## Location", 

722 "", 

723 f"`{_value(location, 'unknown')}`", 

724 "", 

725 "## Severity", 

726 "", 

727 f"{_value(severity, 'minor')} — {_value(justification, 'no justification recorded')}", 

728 "", 

729 "## Evidence", 

730 "", 

731 "```", 

732 _value(evidence, "none provided"), 

733 "```", 

734 "", 

735 "## Suggested fix", 

736 "", 

737 _value(suggested_fix, "none proposed"), 

738 "", 

739 f"Found by keel {_value(source, 'scan')}.", 

740 "", 

741 f"<!-- {SCAN_FINDING_MARKER} -->", 

742 ] 

743 if isinstance(regression_of, int) and not isinstance(regression_of, bool): 

744 lines.extend(["", f"regression-of: #{regression_of}"]) 

745 return "\n".join(lines) + "\n" 

746 

747 

748def render_triage_audit( 

749 *, 

750 issue: int | None = None, 

751 role: str | None = None, 

752 priority: str | None = None, 

753 status: str | None = None, 

754 tier: int | str | None = None, 

755 rationale: str | None = None, 

756 run_id: str | None = None, 

757) -> str: 

758 """Render the deterministic, label-only triage audit comment. 

759 

760 One comment per triaged issue: the applied role / priority / status labels and 

761 risk tier on one line, then the classifier's rationale. When ``run_id`` is 

762 supplied an idempotent re-post edits the existing comment in place. 

763 """ 

764 labels = " · ".join( 

765 [ 

766 f"role: {_value(role, 'unassigned')}", 

767 f"priority: {_value(priority, 'unset')}", 

768 f"status: {_value(status, 'unset')}", 

769 f"tier: {_scalar(tier, 'n/a')}", 

770 ] 

771 ) 

772 lines = [ 

773 TRIAGE_AUDIT_MARKER, 

774 f"keel triage — {_issue_ref(issue)}: {labels}", 

775 "", 

776 _value(rationale, "Classified from the existing label set."), 

777 ] 

778 if isinstance(run_id, str) and run_id.strip(): 

779 lines.extend(["", f"<!-- keel.run-id: {run_id.strip()} -->"]) 

780 return "\n".join(lines) + "\n" 

781 

782 

783def render_extension_result( 

784 *, 

785 slot: str, 

786 extension_id: str, 

787 status: str, 

788 mode: str, 

789 summary: str | None = None, 

790 artifacts: list[str] | tuple[str, ...] = (), 

791 follow_ups: list[str] | tuple[str, ...] = (), 

792) -> str: 

793 """Render a canonical extension result block/comment.""" 

794 lines = [ 

795 EXTENSION_RESULT_MARKER, 

796 "", 

797 "## Extension result", 

798 "", 

799 f"- **Slot:** `{_value(slot, 'unknown')}`", 

800 f"- **Extension:** `{_value(extension_id, 'unknown')}`", 

801 f"- **Status:** {_value(status, 'not-recorded')}", 

802 f"- **Mode:** {_value(mode, 'advisory')}", 

803 f"- **Summary:** {_value(summary, 'No summary recorded.')}", 

804 "- **Artifacts:**", 

805 ] 

806 artifact_lines = _string_bullets(artifacts) 

807 lines.extend(artifact_lines if artifact_lines else [" - none"]) 

808 lines.append("- **Follow-ups:**") 

809 follow_up_lines = _string_bullets(follow_ups) 

810 lines.extend(follow_up_lines if follow_up_lines else [" - none"]) 

811 return "\n".join(lines) + "\n" 

812 

813 

814def render_step_handoff( 

815 *, 

816 step_id: str, 

817 step_name: str | None = None, 

818 status: str = "complete", 

819 summary: str | None = None, 

820 next_step: str | None = None, 

821 evidence_ids: list[str] | tuple[str, ...] = (), 

822) -> str: 

823 """Render the canonical structured handoff between backbone steps.""" 

824 lines = [ 

825 STEP_HANDOFF_MARKER, 

826 "", 

827 "## Step handoff", 

828 "", 

829 f"- **Step:** `{_value(step_id, 'unknown')}`", 

830 f"- **Name:** {_value(step_name, 'not recorded')}", 

831 f"- **Status:** {_value(status, 'complete')}", 

832 f"- **Summary:** {_value(summary, 'No summary recorded.')}", 

833 f"- **Next step:** {_value(next_step, 'Continue the backbone plan.')}", 

834 "- **Evidence:**", 

835 ] 

836 evidence_lines = _string_bullets(evidence_ids) 

837 lines.extend(evidence_lines if evidence_lines else [" - none"]) 

838 return "\n".join(lines) + "\n" 

839 

840 

841def render_run_control_halt( 

842 *, 

843 control: str, 

844 reason: str, 

845 scope: str | None = None, 

846 observed: int | str | None = None, 

847 limit: int | str | None = None, 

848 action: str | None = None, 

849) -> str: 

850 """Render a stable hard-halt reason emitted by run controls.""" 

851 lines = [ 

852 RUN_CONTROL_HALT_MARKER, 

853 "", 

854 "## Run control halt", 

855 "", 

856 f"- **Control:** `{_value(control, 'unknown')}`", 

857 f"- **Reason:** {_value(reason, 'No reason recorded.')}", 

858 f"- **Scope:** {_value(scope, 'run')}", 

859 f"- **Observed:** {_value(observed, 'not recorded')}", 

860 f"- **Limit:** {_value(limit, 'not recorded')}", 

861 f"- **Action:** {_value(action, 'halt')}", 

862 ] 

863 return "\n".join(lines) + "\n" 

864 

865 

866def _finding_lines(findings: list[dict[str, Any]] | tuple[dict[str, Any], ...]) -> list[str]: 

867 if not findings: 

868 return ["- none"] 

869 lines: list[str] = [] 

870 for finding in findings: 

871 severity = _value(finding.get("severity") if isinstance(finding, dict) else None, "nit") 

872 message = _value(finding.get("message") if isinstance(finding, dict) else None, "") 

873 if message: 

874 lines.append(f"- {severity}: {message}") 

875 return lines or ["- none"] 

876 

877 

878def _string_bullets(values: list[str] | tuple[str, ...]) -> list[str]: 

879 return [f" - {value.strip()}" for value in values if isinstance(value, str) and value.strip()] 

880 

881 

882def _string_list(values: Any) -> list[str]: 

883 if not isinstance(values, (list, tuple)): 

884 return [] 

885 return [value.strip() for value in values if isinstance(value, str) and value.strip()] 

886 

887 

888def _canonical_severity(severity: str) -> str: 

889 lowered = severity.strip().lower() 

890 return _SEVERITY_ALIASES.get(lowered, lowered) 

891 

892 

893def _severity_rank(severity: str) -> int: 

894 canonical = _canonical_severity(severity) 

895 return SEVERITY_ORDER.index(canonical) if canonical in SEVERITY_ORDER else len(SEVERITY_ORDER) 

896 

897 

898def _cell(value: str) -> str: 

899 """Escape a non-empty finding string for a Markdown table cell. 

900 

901 Callers pass values already normalised through ``_value`` (never blank), so 

902 escaping the table delimiter and folding newlines keeps the row intact. 

903 """ 

904 return value.replace("\n", " ").replace("|", "\\|") 

905 

906 

907def _cycle_findings(raw: Any) -> list[dict[str, str]]: 

908 if not isinstance(raw, (list, tuple)): 

909 return [] 

910 findings = [ 

911 { 

912 "severity": _value(item.get("severity"), "nit"), 

913 "location": _value(item.get("location"), "—"), 

914 "description": _value(item.get("description"), "—"), 

915 "suggested_fix": _value(item.get("suggested_fix"), "—"), 

916 } 

917 for item in raw 

918 if isinstance(item, dict) 

919 ] 

920 findings.sort(key=lambda finding: _severity_rank(finding["severity"])) 

921 return findings 

922 

923 

924#: What a review-cycle reviewer entry with no ``verdict`` renders and counts as (#1439). It 

925#: used to be ``LGTM``, so a reviewer who said nothing read as approving and the merge 

926#: recommendation could come out "approve"; a missing verdict is no verdict, as 

927#: :func:`render_jury_verdict` reads one since #1432. 

928_CYCLE_NO_VERDICT = "ABSTAIN" 

929 

930 

931def _cycle_verdict(reviewer: dict[str, Any]) -> str: 

932 """A review-cycle reviewer's verdict, or :data:`_CYCLE_NO_VERDICT` when it gave none.""" 

933 return _value(reviewer.get("verdict"), _CYCLE_NO_VERDICT) 

934 

935 

936def _cycle_reviewer_lines(reviewer: dict[str, Any]) -> list[str]: 

937 lines = [ 

938 f"## Reviewer: {_value(reviewer.get('codename'), 'Reviewer')} " 

939 f"(Focus: {_value(reviewer.get('focus'), 'general review')})", 

940 "", 

941 f"Verdict: {_cycle_verdict(reviewer)}", 

942 "", 

943 ] 

944 findings = _cycle_findings(reviewer.get("findings")) 

945 if findings: 

946 lines.append("| Severity | File:Line | Description | Suggested Fix |") 

947 lines.append("| --- | --- | --- | --- |") 

948 lines.extend( 

949 f"| {_cell(finding['severity'])} | {_cell(finding['location'])} | " 

950 f"{_cell(finding['description'])} | {_cell(finding['suggested_fix'])} |" 

951 for finding in findings 

952 ) 

953 else: 

954 lines.append("No findings.") 

955 clean_areas = _string_list(reviewer.get("clean_areas")) 

956 if clean_areas: 

957 lines.extend(["", f"Clean areas: {', '.join(clean_areas)}"]) 

958 return lines 

959 

960 

961def _cycle_histogram(reviewers: list[dict[str, Any]]) -> dict[str, int]: 

962 histogram = dict.fromkeys(SEVERITY_ORDER, 0) 

963 for reviewer in reviewers: 

964 for finding in _cycle_findings(reviewer.get("findings")): 

965 canonical = _canonical_severity(finding["severity"]) 

966 if canonical in histogram: 

967 histogram[canonical] += 1 

968 return histogram 

969 

970 

971def _aggregate_clean_areas(reviewers: list[dict[str, Any]]) -> list[str]: 

972 # Optimize deduplication: O(N) using C-level dict.fromkeys instead of O(N^2) list lookups 

973 return list( 

974 dict.fromkeys( 

975 area for reviewer in reviewers for area in _string_list(reviewer.get("clean_areas")) 

976 ) 

977 ) 

978 

979 

980def _merge_recommendation(reviewers: list[dict[str, Any]], histogram: dict[str, int]) -> str: 

981 needs_fixes = any( 

982 not _cycle_verdict(reviewer).lower().startswith("lgtm") for reviewer in reviewers 

983 ) 

984 if needs_fixes or histogram["blocker"] > 0: 

985 return "❌ block" 

986 if histogram["major"] + histogram["minor"] > 0: 

987 return "⚠️ request changes" 

988 if histogram["nit"] > 0: 

989 return "✅ approve (cosmetic nits)" 

990 return "✅ approve" 

991 

992 

993def _cycle_summary_lines(reviewers: list[dict[str, Any]]) -> list[str]: 

994 histogram = _cycle_histogram(reviewers) 

995 lines = [ 

996 "## Consolidated Summary", 

997 "", 

998 "Severity Histogram: " 

999 + " · ".join(f"{severity} {histogram[severity]}" for severity in SEVERITY_ORDER), 

1000 "", 

1001 "Reviewer verdicts:", 

1002 ] 

1003 if reviewers: 

1004 lines.extend( 

1005 f"- {_value(reviewer.get('codename'), 'Reviewer')}: {_cycle_verdict(reviewer)}" 

1006 for reviewer in reviewers 

1007 ) 

1008 else: 

1009 lines.append("- none") 

1010 areas = _aggregate_clean_areas(reviewers) 

1011 lines.extend(["", f"Clean areas: {', '.join(areas) if areas else 'none reported'}"]) 

1012 lines.extend(["", f"Merge recommendation: {_merge_recommendation(reviewers, histogram)}"]) 

1013 return lines 

1014 

1015 

1016def _closing_reference(issue_number: int | None) -> str: 

1017 return f"Closes #{issue_number}" if isinstance(issue_number, int) else "Refs #<issue-number>" 

1018 

1019 

1020def _issue(issue_number: int | None) -> str: 

1021 return f"#{issue_number}" if isinstance(issue_number, int) else "not recorded" 

1022 

1023 

1024def _pr(pull_request: int | None) -> str: 

1025 return f"#{pull_request}" if isinstance(pull_request, int) else "not opened" 

1026 

1027 

1028def slug(value: str) -> str: 

1029 """Stable, deterministic slug for reviewer/run-id sub-keys (public alias).""" 

1030 clean = "".join(ch.lower() if ch.isalnum() else "-" for ch in value.strip()) 

1031 return "-".join(part for part in clean.split("-") if part) or "reviewer" 

1032 

1033 

1034def _slug(value: str) -> str: 

1035 return slug(value) 

1036 

1037 

1038def _value(value: Any, fallback: str) -> str: 

1039 if isinstance(value, str) and value.strip(): 

1040 return value.strip() 

1041 return fallback