Coverage for src/keel/evidence.py: 100%

689 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""Deterministic pre-merge evidence verification. 

2 

3The ship adapter is agentic, but the artifacts it must leave behind are not: 

4reviewer verdict comments/reviews, the jury verdict a gating jury requires, and the stable 

5closure comment marker. This module keeps the check pure so CI can enforce it 

6without trusting prose in an agent prompt. 

7 

8Classification is **header-anchored**: :func:`marker_in_header` decides what a 

9comment is from its first non-empty line and nothing else, so a marker a reviewer 

10quotes in prose is content rather than a classification signal (#1026). The ship 

11assessment heading is anchored the same way, from its own header line (#1035). 

12""" 

13 

14from __future__ import annotations 

15 

16import hashlib 

17import re 

18from collections.abc import Collection, Sequence 

19from dataclasses import dataclass 

20from typing import Any 

21 

22from . import agents, closure, juryavail 

23from . import team as team_policy 

24 

25SCHEMA_VERSION = "keel.evidence.v1" 

26#: ``reviewers.panel`` when the cross-vendor jury *is* the review for this tier 

27#: (``knobs.team``'s ``review.by_tier.<n>: jury``), and the minimum distinct 

28#: vendors such a panel must span. Both come from :mod:`keel.team`, the leaf 

29#: module that owns the team vocabulary, so the gate and the policy cannot drift. 

30JURY_PANEL = team_policy.JURY_PANEL 

31DEFAULT_MINIMUM_JURY_VENDORS = team_policy.DEFAULT_MIN_VENDORS 

32AGENT_LABEL_PREFIX = "agent:" 

33MODEL_LABEL_PREFIX = "model:" 

34REVIEW_VERDICT_MARKER = "keel.review-verdict.v1" 

35#: The ``Verdict:`` tokens that **approve** a change (#1426). A review verdict counts 

36#: toward ``review-verdict-N`` only when the first word of its ``Verdict:`` line is one 

37#: of these, read case-insensitively by :func:`review_verdict_token`. The set is what 

38#: keel and its hosts actually write, not a guess: 

39#: 

40#: * ``LGTM`` — the default of :func:`keel.artifacts.render_review_verdict`, what 

41#: ``contracts.py``'s ``review_verdict_template`` renders for an unblocked change, and 

42#: what :func:`keel.jury.map_verdict` folds ai-jury's ``APPROVE`` / ``READY`` into 

43#: before ``keel review --from-jury`` posts a ballot; 

44#: * ``APPROVE`` — what the host reviewers post today: every verdict on the keel PRs 

45#: from #1405 onward, 631 of the 1,355 verdict comments on the repository; 

46#: * ``PASS`` — 168 verdicts posted on #918–#992, written by hosts as ``Verdict: pass``. 

47#: 

48#: Everything else holds the merge: ``REQUEST_CHANGES`` (keel's own blocked-change 

49#: template, and the jury's ``NEEDS_INFO``), ``COMMENT``, ``ABSTAIN``, a token keel does 

50#: not know, and a verdict with no ``Verdict:`` line at all. Inventing an approval for a 

51#: stance keel does not recognise is the mapping error that cannot be undone — the same 

52#: rule :data:`keel.jury._VERDICT` states for ballots. 

53APPROVING_VERDICTS = frozenset({"APPROVE", "LGTM", "PASS"}) 

54#: The non-approving tokens a refusal names as *requesting changes*; any other 

55#: non-approving token is named by itself (``alice does not approve at <head> (verdict 

56#: COMMENT)``). 

57REQUEST_CHANGES_VERDICTS = frozenset({"REQUEST_CHANGES", "CHANGES_REQUESTED", "NEEDS_INFO"}) 

58#: The finding a non-approving review verdict at the current head raises (#1426). 

59VERDICT_NOT_APPROVED_FINDING = "review-verdict-not-approved" 

60#: The finding a non-approving jury verdict at the current head raises (#1429). 

61JURY_NOT_APPROVED_FINDING = "jury-verdict-not-approved" 

62JURY_VERDICT_MARKER = "keel.jury-verdict.v1" 

63#: The comment a live ship run posts on its PR right after creating it (#1013). It is 

64#: the *primary* arming signal for the evidence gate: unlike the branch name it is 

65#: written by the run itself, so a run that named its branch something else — or whose 

66#: ledger lives in a per-run worktree CI cannot read — still identifies as a keel ship 

67#: run. See :func:`gate_decision` for the full arming order. 

68SHIP_PROVENANCE_MARKER = "keel.ship-provenance.v1" 

69#: The marker the ship adapter tells a host to post when a finding is deferred rather 

70#: than fixed. keel core never counts a deferral as evidence; it is listed among the 

71#: classification markers so a deferral comment reads as *one* artifact instead of 

72#: being classified by whichever marker its prose happens to name. 

73DEFERRAL_MARKER = "keel.deferral.v1" 

74SHIP_ASSESSMENT_HEADING = "### \U0001f6a2 keel ship" 

75#: The banner the ``keel ship`` CLI prints above its own summary. An assessment pasted 

76#: raw — without the workflow's Markdown heading — leads with this line instead, so both 

77#: forms identify the comment. See :func:`_is_ship_assessment`. 

78SHIP_ASSESSMENT_BANNER = "keel ship \u2014" 

79DEFAULT_WAIVER_LABEL = "keel:evidence-waived" 

80TRUSTED_AUTHOR_ASSOCIATIONS = frozenset({"OWNER", "MEMBER", "COLLABORATOR"}) 

81TRUSTED_SHIP_ASSESSMENT_BOTS = frozenset({"github-actions", "github-actions[bot]"}) 

82 

83#: Every marker that *classifies* an evidence comment. Order is the order findings 

84#: report them in, so a malformed-header message is byte-stable. 

85CLASSIFICATION_MARKERS: tuple[str, ...] = ( 

86 REVIEW_VERDICT_MARKER, 

87 JURY_VERDICT_MARKER, 

88 SHIP_PROVENANCE_MARKER, 

89 closure.CLOSURE_SCHEMA_VERSION, 

90 DEFERRAL_MARKER, 

91) 

92 

93#: Membership view of :data:`CLASSIFICATION_MARKERS`, which is a tuple because its **order** 

94#: is the order markers are rendered in. Asking a tuple `x in ...` is a linear scan, and the 

95#: two places below only ask whether every token is a known marker — a set answers that 

96#: question by its type rather than by walking. The tuple stays where order matters. 

97_CLASSIFICATION_MARKERS_SET = frozenset(CLASSIFICATION_MARKERS) 

98 

99#: The finding raised for a comment whose header names more than one marker. 

100MALFORMED_MARKER_FINDING = "malformed-evidence-comment" 

101 

102#: The header fields keel reads off an evidence comment. Closed by design: an 

103#: unlisted ``key: value`` line is prose, and prose ends the header block (#932). 

104#: ``panelists`` joins ``vendors`` as a jury-verdict field, because the size of a 

105#: panel that *is* the review sets the required verdict count (#1015). 

106_FIELD_RE = re.compile( 

107 r"^\s*(?P<key>reviewer|head|vendor|model|vendors|panelists)\s*:\s*(?P<value>\S+)\s*$", 

108 re.IGNORECASE, 

109) 

110_HEADER_LINE_RE = re.compile(r"^[A-Za-z0-9_-]+\s*:") 

111#: A ``Verdict:`` line, as :func:`keel.artifacts.render_review_verdict` writes it. Anchored 

112#: at the line start, so ``AI Jury verdict:`` and a quoted ``> Verdict:`` are not one. 

113_VERDICT_LINE_RE = re.compile(r"^\s*verdict\s*:(?P<value>.*)$", re.IGNORECASE) 

114#: The first word of a verdict line's value: wrapper punctuation (``**``, a backtick, an 

115#: emoji) is skipped, and the word ends at the first character that is not a letter, 

116#: ``_`` or ``-`` — so ``APPROVE — minor nits`` and ``APPROVE, minor nits`` both read 

117#: ``APPROVE``. 

118_VERDICT_TOKEN_RE = re.compile(r"^[\W_]*(?P<token>[A-Za-z][A-Za-z_-]*)") 

119#: A jury verdict's consensus line, as :func:`keel.artifacts.render_jury_verdict` writes it: 

120#: ``AI Jury verdict: REQUEST_CHANGES.`` — the trailing full stop ends the token like any 

121#: other punctuation does (#1429). 

122_JURY_VERDICT_LINE_RE = re.compile(r"^\s*AI\s+Jury\s+verdict\s*:(?P<value>.*)$", re.IGNORECASE) 

123_SHIP_BRANCH_RE = re.compile(r"^(feature|fix|chore|docs|test)/issue-\d+(?:-|$)") 

124#: The exact wrapper a marker line may wear. Every keel renderer emits its marker as 

125#: the whole first line, in one of exactly two shapes: bare 

126#: (``artifacts.render_review_verdict`` / ``render_jury_verdict`` / 

127#: ``render_ship_provenance``) or wrapped in an HTML comment so it renders invisibly 

128#: (``closure.render_closure_comment``). 

129#: 

130#: These are matched literally, never with a regex. A pattern that treats ``-->`` as 

131#: *the* comment terminator is wrong about HTML — a browser also ends a comment at 

132#: ``--!>`` — and a classifier that disagrees with the renderer about where a comment 

133#: ends is exactly the confusion this module exists to remove (CodeQL 

134#: ``py/bad-tag-filter``). keel does not need to parse HTML: it needs to recognise the 

135#: one shape it writes, and refuse everything else. 

136_HTML_COMMENT_OPEN = "<!--" 

137_HTML_COMMENT_CLOSE = "-->" 

138 

139STATUS_PASS = "pass" 

140#: The finding severity that fails verification on its own, with nothing missing. 

141BLOCKING_FINDING_SEVERITY = "major" 

142STATUS_WAITING = "waiting" 

143STATUS_FAIL = "fail" 

144STATUSES = (STATUS_PASS, STATUS_WAITING, STATUS_FAIL) 

145 

146# Evidence phases. An artifact is required in the phase that produces it, mirroring 

147# the step mapping stepverifier already applies (review -> s7, jury -> s8, closure -> 

148# s12). The merge gate at s10 asks for PHASE_PRE_MERGE, because the closure comment 

149# is a post-merge record and requiring it at s10 makes the backbone unsatisfiable. 

150PHASE_PRE_MERGE = "pre-merge" 

151PHASE_POST_MERGE = "post-merge" 

152PHASE_ALL = "all" 

153PHASES = (PHASE_PRE_MERGE, PHASE_POST_MERGE, PHASE_ALL) 

154 

155 

156@dataclass(frozen=True) 

157class EvidenceItem: 

158 id: str 

159 kind: str 

160 required: bool 

161 description: str 

162 phase: str = PHASE_PRE_MERGE 

163 

164 def as_dict(self) -> dict[str, Any]: 

165 return { 

166 "id": self.id, 

167 "kind": self.kind, 

168 "required": self.required, 

169 "description": self.description, 

170 "phase": self.phase, 

171 } 

172 

173 

174def gate_active(labels: Sequence[str] | None, gate_label: str) -> bool: 

175 """Return whether ``gate_label`` is present in ``labels`` (None/empty -> False). 

176 

177 An empty ``gate_label`` is never active, so a misconfigured (blank) label can 

178 never silently match — the schema also forbids an empty ``evidence_gate_label``. 

179 """ 

180 if not gate_label: 

181 return False 

182 return gate_label in set(labels or ()) 

183 

184 

185def gate_decision( 

186 labels: Sequence[str] | None, 

187 gate_label: str, 

188 *, 

189 waiver_label: str = DEFAULT_WAIVER_LABEL, 

190 head_ref: str | None = None, 

191 pr_comments: list[dict[str, Any]] | None = None, 

192 pr_reviews: list[dict[str, Any]] | None = None, 

193 ledger_records: Sequence[object] | None = None, 

194) -> dict[str, Any]: 

195 """Return the fail-closed evidence-gate arming decision. 

196 

197 Ship provenance arms the gate by default. The only disarm path is an explicit 

198 waiver label applied by an operator; the legacy gate label remains an 

199 additional arming signal for already-installed workflows. 

200 

201 The signals are consulted in this order, and the order is part of the contract 

202 (documented in ``docs/keel/evidence.md``): 

203 

204 1. ``operator-waiver-label`` — the one sanctioned disarm, checked first so an 

205 explicit waiver is never shadowed by an arming signal. 

206 2. ``gate-label`` — the legacy opt-in label. 

207 3. ``ship-provenance-comment`` — a trusted PR comment carrying 

208 :data:`SHIP_PROVENANCE_MARKER`, which a live ship run posts as soon as the PR 

209 exists. **Ahead of the branch regex on purpose** (#1013): the marker is written 

210 by the run, the branch name is written by whoever typed it, and a ship run that 

211 named its branch ``fix/2467-slug`` used to read as a non-keel PR. 

212 4. ``ship-branch`` — the legacy branch-name fallback for runs that predate the 

213 marker. Kept, but it is no longer the signal keel relies on. 

214 5. ``ship-assessment-comment`` / ``review-verdict-marker`` / ``ship-run-ledger`` — 

215 the remaining after-the-fact traces, unchanged. 

216 

217 Nothing was removed: every path that armed the gate before still arms it. 

218 """ 

219 label_set = set(labels or ()) 

220 if waiver_label and waiver_label in label_set: 

221 return _gate_decision(False, "operator-waiver-label", waiver_label, waived=True) 

222 if gate_active(labels, gate_label): 

223 return _gate_decision(True, "gate-label", gate_label) 

224 if _has_trusted_ship_provenance(pr_comments or []): 

225 return _gate_decision(True, "ship-provenance-comment", SHIP_PROVENANCE_MARKER) 

226 if head_ref and _SHIP_BRANCH_RE.search(head_ref): 

227 return _gate_decision(True, "ship-branch", head_ref) 

228 if _has_trusted_ship_assessment(pr_comments or []): 

229 return _gate_decision(True, "ship-assessment-comment", SHIP_ASSESSMENT_HEADING) 

230 if _has_trusted_review_marker([*(pr_comments or []), *(pr_reviews or [])]): 

231 return _gate_decision(True, "review-verdict-marker", REVIEW_VERDICT_MARKER) 

232 if ledger_records: 

233 return _gate_decision(True, "ship-run-ledger", "ship_run") 

234 return _gate_decision(False, "no-ship-provenance", None) 

235 

236 

237def _gate_decision( 

238 enforced: bool, 

239 reason: str, 

240 source: str | None, 

241 *, 

242 waived: bool = False, 

243) -> dict[str, Any]: 

244 return { 

245 "schema_version": SCHEMA_VERSION, 

246 "enforced": enforced, 

247 "waived": waived, 

248 "reason": reason, 

249 "source": source, 

250 } 

251 

252 

253def _has_trusted_ship_provenance(items: list[dict[str, Any]]) -> bool: 

254 """True when a trusted PR comment carries the ship-provenance marker. 

255 

256 Trust is the same fail-closed ``author_association`` check every other evidence 

257 source uses: an anonymous drive-by comment must not be able to arm — or, more to 

258 the point, to *look* like it armed — the gate. 

259 """ 

260 return any( 

261 _is_trusted_source(item, enforced=True) 

262 and marker_in_header(_body(item)) == SHIP_PROVENANCE_MARKER 

263 for item in items 

264 ) 

265 

266 

267def _has_trusted_ship_assessment(items: list[dict[str, Any]]) -> bool: 

268 return any( 

269 _is_ship_assessment_source(item) and _is_ship_assessment(_body(item)) for item in items 

270 ) 

271 

272 

273def _is_ship_assessment_source(item: dict[str, Any]) -> bool: 

274 if _is_trusted_source(item, enforced=True): 

275 return True 

276 user = item.get("user") if isinstance(item.get("user"), dict) else {} 

277 login = user.get("login") if isinstance(user.get("login"), str) else None 

278 return bool(login and login.lower() in TRUSTED_SHIP_ASSESSMENT_BOTS) 

279 

280 

281def contract_as_dict( 

282 review_contract: dict[str, Any], 

283 *, 

284 dry_run: bool = False, 

285 enforced: bool = True, 

286 deferrals: tuple[str, ...] = (), 

287 phase: str = PHASE_ALL, 

288) -> dict[str, Any]: 

289 """Return the required evidence set derived from review/jury flags.""" 

290 return { 

291 "schema_version": SCHEMA_VERSION, 

292 "enforced": enforced, 

293 "phase": phase, 

294 "source": "review_merge_contract + closure_comment", 

295 "dry_run_disables_gating": True, 

296 "fail_closed": True, 

297 "require_distinct_vendors": _require_distinct_vendors(review_contract), 

298 "accepted_sources": { 

299 "closure": ("trusted issue/PR comments carrying keel.closure-comment.v1"), 

300 "review": ( 

301 "trusted PR review/comment carrying keel.review-verdict.v1 and current head, " 

302 "whose reviewer's latest Verdict line approves" 

303 ), 

304 "jury": "trusted PR comment carrying keel.jury-verdict.v1 and current head", 

305 }, 

306 "not_accepted": [ 

307 "pull_request_body", 

308 "chat_summary", 

309 "untrusted_public_comment", 

310 "keel_ship_assessment_comment", 

311 ], 

312 "deferrals": list(deferrals), 

313 "required": [ 

314 item.as_dict() 

315 for item in required_items(review_contract, dry_run=False, enforced=enforced) 

316 ], 

317 "active_required": [ 

318 item.as_dict() 

319 for item in required_items( 

320 review_contract, dry_run=dry_run, enforced=enforced, phase=phase 

321 ) 

322 ], 

323 } 

324 

325 

326def required_items( 

327 review_contract: dict[str, Any], 

328 *, 

329 dry_run: bool = False, 

330 enforced: bool = True, 

331 phase: str = PHASE_ALL, 

332) -> tuple[EvidenceItem, ...]: 

333 """Return the tier/flag-derived evidence requirements for ``phase``. 

334 

335 ``phase`` selects which artifacts are in scope: ``pre-merge`` covers the 

336 review verdicts and a gating jury verdict (everything that must exist before 

337 s10 authorizes a merge), ``post-merge`` covers the closure comments s11 

338 posts, and ``all`` — the default, so existing callers are unchanged — covers 

339 both. An unknown phase raises, so a typo cannot silently drop requirements. 

340 """ 

341 if phase not in PHASES: 

342 raise ValueError(f"unknown evidence phase {phase!r}; expected one of {', '.join(PHASES)}") 

343 if dry_run or not enforced: 

344 return () 

345 reviewers = review_contract.get("reviewers") 

346 reviewer_count = reviewers.get("count") if isinstance(reviewers, dict) else 0 

347 reviewer_count = reviewer_count if isinstance(reviewer_count, int) and reviewer_count > 0 else 0 

348 jury = review_contract.get("jury") 

349 jury_required = ( 

350 isinstance(jury, dict) and bool(jury.get("enabled")) and jury.get("mode") == "gating" 

351 ) 

352 items: list[EvidenceItem] = [ 

353 EvidenceItem( 

354 "closure-comment-pr", 

355 "closure", 

356 True, 

357 "PR conversation comment with keel.closure-comment.v1 marker", 

358 PHASE_POST_MERGE, 

359 ), 

360 EvidenceItem( 

361 "closure-comment-issue", 

362 "closure", 

363 True, 

364 "Linked issue comment with keel.closure-comment.v1 marker", 

365 PHASE_POST_MERGE, 

366 ), 

367 ] 

368 # A jury panel's verdicts are panelist ballots mapped onto s7 evidence by 

369 # `keel review --from-jury` (#1015): same marker, same head binding, same 

370 # requirement. The description names the panel that produced them so a 

371 # missing one sends the operator to the jury run rather than to a host 

372 # reviewer that was never dispatched. 

373 review_description = ( 

374 "Distinct posted ai-jury panelist verdict for the current PR" 

375 if review_panel(review_contract) == JURY_PANEL 

376 else "Distinct posted s7 reviewer verdict for the current PR" 

377 ) 

378 for index in range(1, reviewer_count + 1): 

379 items.append( 

380 EvidenceItem( 

381 f"review-verdict-{index}", 

382 "review", 

383 True, 

384 review_description, 

385 PHASE_PRE_MERGE, 

386 ) 

387 ) 

388 if jury_required: 

389 items.append( 

390 EvidenceItem( 

391 "jury-verdict", 

392 "jury", 

393 True, 

394 "Posted gating jury verdict comment for the current PR", 

395 PHASE_PRE_MERGE, 

396 ) 

397 ) 

398 if phase == PHASE_ALL: 

399 return tuple(items) 

400 return tuple(item for item in items if item.phase == phase) 

401 

402 

403def verify( 

404 review_contract: dict[str, Any], 

405 *, 

406 pr_comments: list[dict[str, Any]] | None = None, 

407 issue_comments: list[dict[str, Any]] | None = None, 

408 pr_reviews: list[dict[str, Any]] | None = None, 

409 pr_body: str | None = None, 

410 pr_title: str = "", 

411 pr_labels: Sequence[str] | None = None, 

412 head_sha: str | None = None, 

413 covered_heads: Collection[str] = (), 

414 ledger_record: dict[str, Any] | None = None, 

415 dry_run: bool = False, 

416 enforced: bool = True, 

417 deferrals: tuple[str, ...] = (), 

418 phase: str = PHASE_ALL, 

419 require_armed: bool = False, 

420 waived: bool = False, 

421) -> dict[str, Any]: 

422 """Verify required evidence artifacts and return a deterministic report. 

423 

424 ``phase`` narrows the requirement set to the artifacts that phase produces; 

425 see :func:`required_items`. The s10 merge gate passes ``pre-merge`` so it 

426 does not demand the closure comments s11 has not written yet. 

427 

428 ``require_armed`` closes the vacuous-pass hole: with the gate unarmed there 

429 are no requirements, so the report would otherwise pass without having 

430 checked anything, and a green result could not be told apart from "could not 

431 tell whether this was a ship run". When set, an unarmed gate is a blocking 

432 finding instead. A deliberately non-ship PR still goes green through the 

433 operator waiver label, which disarms explicitly rather than by accident. 

434 

435 When ``ledger_record`` is the ship_run record for this PR, a closure comment 

436 only counts when its content matches the canonical render of that record 

437 (closure-comment fidelity). Without a record the marker-only behavior holds. 

438 

439 Every comment is classified by :func:`marker_in_header` — the marker on its 

440 header line, never one quoted in its prose. A header naming two markers is 

441 malformed: it is excluded from every count and reported as an advisory 

442 ``malformed-evidence-comment`` finding. 

443 

444 When the gate is active, ``pr_labels`` are additionally checked for the 

445 mandatory ``agent:<vendor>`` attribution label (and cross-checked against the 

446 ledger implementer vendor when a record is present); see 

447 :func:`attribution_check`. The labels are separately checked against keel's own 

448 vocabulary — what :func:`keel.agents.attribution` produces from the ledger's 

449 ``actors.implementer`` — by :func:`attribution_vocabulary_check`, which catches a 

450 hand-composed label that happens to agree with a hand-written ledger value. 

451 """ 

452 del pr_body # Explicitly not accepted as evidence. 

453 items = required_items(review_contract, dry_run=dry_run, enforced=enforced, phase=phase) 

454 deferred = set(deferrals) 

455 counts = _evidence_counts( 

456 pr_comments=pr_comments or [], 

457 issue_comments=issue_comments or [], 

458 pr_reviews=pr_reviews or [], 

459 head_sha=head_sha, 

460 covered_heads=covered_heads, 

461 enforced=enforced, 

462 ledger_record=ledger_record, 

463 ) 

464 findings = _run_context_findings( 

465 pr_comments=pr_comments or [], 

466 issue_comments=issue_comments or [], 

467 enforced=enforced, 

468 ledger_record=ledger_record, 

469 ) 

470 mismatch = _closure_mismatch_scopes( 

471 pr_comments=pr_comments or [], 

472 issue_comments=issue_comments or [], 

473 enforced=enforced, 

474 ledger_record=ledger_record, 

475 ) 

476 # A jury whose standing consensus does not approve (#1429) leaves `jury-verdict` 

477 # unsatisfied *and* says why, by name: unsatisfied alone would read as "missing" and 

478 # hide a comment that is on the pull request. It blocks only where the jury verdict is 

479 # a requirement — an advisory panel's rejection is reported, never gated on, which is 

480 # what advisory means. 

481 jury_required = any( 

482 item.id == "jury-verdict" 

483 and not (item.id in deferred or item.kind in deferred or "all" in deferred) 

484 for item in items 

485 ) 

486 jury_refusal = _jury_not_approved_finding( 

487 pr_comments or [], 

488 head_sha=head_sha, 

489 covered_heads=covered_heads, 

490 enforced=enforced, 

491 blocking=jury_required, 

492 ) 

493 results = [] 

494 for item in items: 

495 present = _is_present(item, counts) 

496 is_deferred = item.id in deferred or item.kind in deferred or "all" in deferred 

497 ok = present or is_deferred 

498 results.append( 

499 { 

500 "id": item.id, 

501 "kind": item.kind, 

502 "required": item.required, 

503 "present": present, 

504 "deferred": is_deferred, 

505 "ok": ok, 

506 "reason": None 

507 if ok 

508 else ( 

509 jury_refusal["message"] 

510 if item.id == "jury-verdict" and jury_refusal is not None 

511 else _result_reason(item, mismatch) 

512 ), 

513 } 

514 ) 

515 missing = [result["id"] for result in results if not result["ok"]] 

516 distinct = _distinct_vendor_finding( 

517 review_contract, 

518 items=items, 

519 deferred=deferred, 

520 pr_comments=pr_comments or [], 

521 pr_reviews=pr_reviews or [], 

522 head_sha=head_sha, 

523 covered_heads=covered_heads, 

524 enforced=enforced, 

525 ) 

526 if distinct is not None: 

527 findings = [*findings, distinct] 

528 substance = _verdict_substance_findings( 

529 [*(pr_comments or []), *(pr_reviews or [])], 

530 head_sha=head_sha, 

531 covered_heads=covered_heads, 

532 enforced=enforced, 

533 pr_title=pr_title, 

534 ) 

535 findings = [*findings, *substance] 

536 # A reviewer whose standing verdict does not approve holds the merge by name (#1426): 

537 # the counts above leave them out, and this says why instead of letting another 

538 # reviewer's approval make up the number. 

539 findings = [ 

540 *findings, 

541 *_verdict_not_approved_findings( 

542 [*(pr_comments or []), *(pr_reviews or [])], 

543 head_sha=head_sha, 

544 covered_heads=covered_heads, 

545 enforced=enforced, 

546 ), 

547 ] 

548 if jury_refusal is not None: 

549 findings = [*findings, jury_refusal] 

550 findings = [ 

551 *findings, 

552 *_malformed_marker_findings( 

553 [*(pr_comments or []), *(issue_comments or []), *(pr_reviews or [])], 

554 enforced=enforced, 

555 ), 

556 ] 

557 attribution = _attribution_finding( 

558 pr_labels=pr_labels, 

559 enforced=enforced and not dry_run, 

560 ledger_record=ledger_record, 

561 ) 

562 if attribution is not None: 

563 findings = [*findings, attribution] 

564 vocabulary = _attribution_vocabulary_finding( 

565 pr_labels=pr_labels, 

566 enforced=enforced and not dry_run, 

567 ledger_record=ledger_record, 

568 ) 

569 if vocabulary is not None: 

570 findings = [*findings, vocabulary] 

571 unarmed = _unarmed_finding( 

572 enforced=enforced, 

573 dry_run=dry_run, 

574 require_armed=require_armed, 

575 waived=waived, 

576 ) 

577 if unarmed is not None: 

578 findings = [*findings, unarmed] 

579 blocking_findings = [f for f in findings if f["severity"] == BLOCKING_FINDING_SEVERITY] 

580 has_mismatch = bool(mismatch) 

581 if not missing and not blocking_findings: 

582 status = STATUS_PASS 

583 elif blocking_findings or has_mismatch: 

584 status = STATUS_FAIL 

585 else: 

586 status = STATUS_WAITING 

587 return { 

588 "schema_version": SCHEMA_VERSION, 

589 "status": status, 

590 "dry_run": dry_run, 

591 "enforced": enforced, 

592 "phase": phase, 

593 "required_count": len(items), 

594 "missing": missing, 

595 "results": results, 

596 "counts": counts, 

597 "findings": findings, 

598 } 

599 

600 

601def refusal_reason(verification: dict[str, Any]) -> str: 

602 """Why a verification that is not a pass refuses the merge, naming everything (#1420). 

603 

604 Every missing item, then every blocking finding's message. A finding can fail the 

605 verification with nothing missing — ``attribution-label`` on a pull request whose 

606 review verdicts are all there — and a reason built from ``missing`` alone then read 

607 ``missing evidence: `` with nothing after it, which names no cause at all. Never empty: 

608 a verification that names neither says its status. 

609 """ 

610 missing = [str(item) for item in verification.get("missing") or ()] 

611 blocking = [ 

612 f"{finding.get('id')}: {finding.get('message')}" 

613 for finding in verification.get("findings") or () 

614 if isinstance(finding, dict) and finding.get("severity") == BLOCKING_FINDING_SEVERITY 

615 ] 

616 parts: list[str] = [] 

617 if missing: 

618 parts.append(f"missing evidence: {', '.join(missing)}") 

619 if blocking: 

620 parts.append(f"blocking finding(s): {'; '.join(blocking)}") 

621 return "; ".join(parts) or f"evidence verification is {verification.get('status')}" 

622 

623 

624def _require_distinct_vendors(review_contract: dict[str, Any]) -> bool: 

625 reviewers = review_contract.get("reviewers") 

626 return bool(reviewers.get("require_distinct_vendors")) if isinstance(reviewers, dict) else False 

627 

628 

629def review_panel(review_contract: dict[str, Any]) -> str: 

630 """Who the reviewers are on this contract: ``reviewers`` or ``jury`` (#1015). 

631 

632 A missing or malformed ``reviewers`` block reads as the host bench, which is 

633 the stricter of the two answers everywhere this is asked. 

634 """ 

635 reviewers = review_contract.get("reviewers") 

636 panel = reviewers.get("panel") if isinstance(reviewers, dict) else None 

637 return panel if isinstance(panel, str) and panel else "reviewers" 

638 

639 

640def _minimum_jury_vendors(review_contract: dict[str, Any]) -> int: 

641 """``jury.minimum_vendors`` from the contract, with the schema floor as fallback.""" 

642 jury = review_contract.get("jury") 

643 minimum = jury.get("minimum_vendors") if isinstance(jury, dict) else None 

644 if isinstance(minimum, int) and not isinstance(minimum, bool) and minimum > 0: 

645 return minimum 

646 return DEFAULT_MINIMUM_JURY_VENDORS 

647 

648 

649def _verdict_substance_findings( 

650 items: list[dict[str, Any]], 

651 *, 

652 head_sha: str | None, 

653 covered_heads: Collection[str] = (), 

654 enforced: bool, 

655 pr_title: str, 

656) -> list[dict[str, Any]]: 

657 """One finding per verdict refused for naming nothing concrete (#926). 

658 

659 Reported rather than dropped. A verdict silently not counted surfaces as 

660 "missing required evidence: review-verdict-2", which sends the operator 

661 looking for a comment that is sitting right there — the reason has to say 

662 the verdict was read and found to be a receipt. 

663 

664 ``minor`` when the gate is not enforced, mirroring the other content 

665 findings: an advisory run should say what it saw without failing. 

666 """ 

667 _, rejected = _review_evidence_keys_and_rejections( 

668 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced, pr_title=pr_title 

669 ) 

670 return [ 

671 { 

672 "id": "review-verdict-insubstantial", 

673 "severity": "major" if enforced else "minor", 

674 "kind": "review", 

675 "message": f"{key}: {reason}.", 

676 } 

677 for key, reason in sorted(rejected) 

678 ] 

679 

680 

681def _malformed_marker_findings( 

682 items: list[dict[str, Any]], 

683 *, 

684 enforced: bool, 

685) -> list[dict[str, Any]]: 

686 """One finding per trusted comment whose header names more than one marker. 

687 

688 Such a header does not say which artifact the comment is, so 

689 :func:`marker_in_header` refuses to classify it and the comment counts toward 

690 nothing. Excluding it silently would reproduce the failure #926 named — a 

691 comment sitting right there on the PR, reported as missing evidence — so the 

692 exclusion is stated instead of inferred. 

693 

694 ``minor``, never blocking: the comment is malformed, not fraudulent, and the 

695 requirement it failed to satisfy already fails on its own. 

696 """ 

697 findings: list[dict[str, Any]] = [] 

698 for item in items: 

699 if not _is_trusted_source(item, enforced=enforced): 

700 continue 

701 markers = header_markers(_body(item)) 

702 if len(markers) < 2: 

703 continue 

704 findings.append( 

705 { 

706 "id": MALFORMED_MARKER_FINDING, 

707 "severity": "minor", 

708 "kind": "evidence", 

709 "message": ( 

710 "Comment header carries more than one keel marker " 

711 f"({', '.join(markers)}); it is excluded from evidence." 

712 ), 

713 } 

714 ) 

715 return findings 

716 

717 

718def _distinct_vendor_finding( 

719 review_contract: dict[str, Any], 

720 *, 

721 items: tuple[EvidenceItem, ...], 

722 deferred: set[str], 

723 pr_comments: list[dict[str, Any]], 

724 pr_reviews: list[dict[str, Any]], 

725 head_sha: str | None, 

726 covered_heads: Collection[str] = (), 

727 enforced: bool, 

728) -> dict[str, Any] | None: 

729 """Return a blocking finding when the optional vendor-distinctness check fails. 

730 

731 Off by default: ``None`` unless ``reviewers.require_distinct_vendors`` is set 

732 on the contract. Skipped when review evidence is deferred so the knob never 

733 overrides an explicit deferral. 

734 """ 

735 if not _require_distinct_vendors(review_contract): 

736 return None 

737 if "review" in deferred or "all" in deferred: 

738 return None 

739 required = sum(1 for item in items if item.kind == "review" and item.id not in deferred) 

740 if required <= 0: 

741 return None 

742 provenance = _review_vendor_provenance( 

743 [*pr_comments, *pr_reviews], 

744 head_sha=head_sha, 

745 covered_heads=covered_heads, 

746 enforced=enforced, 

747 ) 

748 if review_panel(review_contract) == JURY_PANEL: 

749 result = panel_vendor_check( 

750 list(provenance.values()), 

751 required_count=required, 

752 minimum_vendors=_minimum_jury_vendors(review_contract), 

753 ) 

754 else: 

755 result = distinct_vendor_check(list(provenance.values()), required_count=required) 

756 if result["ok"]: 

757 return None 

758 return { 

759 "id": "review-vendor-distinctness", 

760 "severity": "major", 

761 "kind": "review", 

762 "message": f"require_distinct_vendors: {result['reason']}.", 

763 } 

764 

765 

766_CLOSURE_MISMATCH_REASON = "closure comment does not match the ship_run ledger record" 

767 

768 

769def _result_reason(item: EvidenceItem, mismatch: set[str]) -> str: 

770 if item.id == "closure-comment-pr" and "pr" in mismatch: 

771 return _CLOSURE_MISMATCH_REASON 

772 if item.id == "closure-comment-issue" and "issue" in mismatch: 

773 return _CLOSURE_MISMATCH_REASON 

774 return f"missing required evidence: {item.id}" 

775 

776 

777def _closure_mismatch_scopes( 

778 *, 

779 pr_comments: list[dict[str, Any]], 

780 issue_comments: list[dict[str, Any]], 

781 enforced: bool, 

782 ledger_record: dict[str, Any] | None, 

783) -> set[str]: 

784 """Return scopes ({"pr"}/{"issue"}) where a marker closure mismatched the ledger. 

785 

786 A scope is reported only when a trusted marker-bearing closure exists but none 

787 of them match the record — so a stale comment alongside a correct re-post does 

788 not produce a misleading mismatch reason. 

789 """ 

790 if ledger_record is None: 

791 return set() 

792 scopes: set[str] = set() 

793 for scope, comments in (("pr", pr_comments), ("issue", issue_comments)): 

794 markered = [ 

795 comment 

796 for comment in comments 

797 if _is_trusted_source(comment, enforced=enforced) 

798 and _has_closure_marker(_body(comment)) 

799 ] 

800 if markered and not any( 

801 closure_body_matches_record(_body(comment), ledger_record) for comment in markered 

802 ): 

803 scopes.add(scope) 

804 return scopes 

805 

806 

807def _evidence_counts( 

808 *, 

809 pr_comments: list[dict[str, Any]], 

810 issue_comments: list[dict[str, Any]], 

811 pr_reviews: list[dict[str, Any]], 

812 head_sha: str | None = None, 

813 covered_heads: Collection[str] = (), 

814 enforced: bool = True, 

815 ledger_record: dict[str, Any] | None = None, 

816) -> dict[str, int]: 

817 review_keys = _review_evidence_keys( 

818 [*pr_comments, *pr_reviews], 

819 head_sha=head_sha, 

820 covered_heads=covered_heads, 

821 enforced=enforced, 

822 ) 

823 return { 

824 "closure_pr": sum( 

825 _is_closure_comment(comment, enforced=enforced, record=ledger_record) 

826 for comment in pr_comments 

827 ), 

828 "closure_issue": sum( 

829 _is_closure_comment(comment, enforced=enforced, record=ledger_record) 

830 for comment in issue_comments 

831 ), 

832 "review_verdict": len(review_keys), 

833 # The jury requirement is met by the *standing* jury verdict at the head, and only 

834 # when its consensus approves (#1429); a rejection is reported by 

835 # `_jury_not_approved_finding` rather than counted. 

836 "jury_verdict": int( 

837 jury_verdict_approves( 

838 _body( 

839 _standing_jury_verdict( 

840 pr_comments, 

841 head_sha=head_sha, 

842 covered_heads=covered_heads, 

843 enforced=enforced, 

844 ) 

845 or {} 

846 ) 

847 ) 

848 ), 

849 } 

850 

851 

852def _is_present(item: EvidenceItem, counts: dict[str, int]) -> bool: 

853 if item.id == "closure-comment-pr": 

854 return counts["closure_pr"] >= 1 

855 if item.id == "closure-comment-issue": 

856 return counts["closure_issue"] >= 1 

857 if item.kind == "review": 

858 index = int(item.id.rsplit("-", 1)[1]) 

859 return counts["review_verdict"] >= index 

860 if item.id == "jury-verdict": 

861 return counts["jury_verdict"] >= 1 

862 return False 

863 

864 

865def _body(item: dict[str, Any]) -> str: 

866 body = item.get("body") 

867 return body if isinstance(body, str) else "" 

868 

869 

870def _header_line(body: str) -> str: 

871 """``body``'s header line: its first *non-empty* line, stripped. 

872 

873 Leading blank lines are skipped rather than treated as the end, for the same 

874 reason :func:`_fields` skips them — a GitHub comment body routinely begins 

875 with a newline. Everything after that line is prose. 

876 """ 

877 for raw_line in (body or "").splitlines(): 

878 line = raw_line.strip() 

879 if not line: 

880 continue 

881 return line 

882 return "" 

883 

884 

885def _unwrap_html_comment(line: str) -> str: 

886 """Strip one literal ``<!-- … -->`` wrapper from ``line``, or return it unchanged. 

887 

888 Deliberately not a regex and deliberately not an HTML parser: the only 

889 wrapper keel has to recognise is the one 

890 :func:`keel.closure.render_closure_comment` writes. Anything else — an 

891 unterminated ``<!--``, a ``--!>`` close, a second wrapper on the same line — 

892 is left intact, so the marker check below sees the delimiters as tokens and 

893 refuses to classify the comment. Failing to recognise a hand-rolled wrapper 

894 costs a comment its classification; guessing at one would let a body render 

895 as an invisible comment while counting as evidence. 

896 """ 

897 if ( 

898 line.startswith(_HTML_COMMENT_OPEN) 

899 and line.endswith(_HTML_COMMENT_CLOSE) 

900 and len(line) >= len(_HTML_COMMENT_OPEN) + len(_HTML_COMMENT_CLOSE) 

901 ): 

902 return line[len(_HTML_COMMENT_OPEN) : -len(_HTML_COMMENT_CLOSE)].strip() 

903 return line 

904 

905 

906def header_markers(body: str) -> tuple[str, ...]: 

907 """Return the distinct :data:`CLASSIFICATION_MARKERS` ``body``'s header carries. 

908 

909 The header line, once unwrapped, must consist of markers and nothing else — 

910 that is exactly what every renderer emits, and it is what separates a marker 

911 line from a sentence that happens to name one. So this is empty for an 

912 ordinary comment (including one whose first line *mentions* a marker in 

913 prose, and one wearing a wrapper keel does not write), one entry for a 

914 well-formed artifact, and two or more for a malformed one, which 

915 :func:`marker_in_header` refuses to classify and 

916 :func:`_malformed_marker_findings` reports. 

917 """ 

918 tokens = _unwrap_html_comment(_header_line(body)).split() 

919 if not tokens or not _CLASSIFICATION_MARKERS_SET.issuperset(tokens): 

920 return () 

921 return tuple(marker for marker in CLASSIFICATION_MARKERS if marker in tokens) 

922 

923 

924def marker_in_header(body: str) -> str | None: 

925 """Return the single keel marker ``body`` is anchored to, or ``None`` (#1026). 

926 

927 **The header block is the only place a marker classifies a comment.** A marker 

928 further down is prose — a reviewer writing "I checked the jury-verdict marker 

929 handling" is quoting a string, not filing a jury verdict. Testing 

930 ``MARKER in body`` could not tell the two apart: two `keel.review-verdict.v1` 

931 comments whose scope mentioned ``keel.jury-verdict.v1`` were counted as 

932 ``jury_verdict: 2, review_verdict: 0``, and the review that happened was 

933 invisible to the gate. 

934 

935 ``None`` for a comment that carries no marker *and* for one whose header 

936 carries several: a header naming two artifacts does not say which one it is, 

937 so it is excluded rather than counted for either. 

938 """ 

939 markers = header_markers(body) 

940 return markers[0] if len(markers) == 1 else None 

941 

942 

943def _has_closure_marker(body: str) -> bool: 

944 return marker_in_header(body) == closure.CLOSURE_SCHEMA_VERSION 

945 

946 

947#: The idempotency marker ``keel post-comment`` appends to a posted body so a re-post 

948#: can find and edit its own comment. It is transport bookkeeping, not content, so it is 

949#: stripped before a closure body is compared to its canonical render. 

950#: 

951#: Matched in the *exact* form the transport emits — a run id, then the close, then end 

952#: of line. A permissive ``.*?`` would let a trusted author smuggle arbitrary text past 

953#: the verbatim comparison: an HTML comment ends at its first ``-->``, so anything after 

954#: that renders visibly on the page while the whole line still normalizes away. 

955RUN_ID_MARKER_RE = re.compile(r"^\s*<!--\s*keel\.run-id:\s*[\w.:@/+-]+\s*-->\s*$") 

956 

957 

958def _normalize_closure_body(body: str) -> str: 

959 """Normalize a closure body for content comparison. 

960 

961 Robust to harmless formatting drift but sensitive to real content changes: 

962 trailing whitespace is stripped per line, runs of blank lines collapse to a 

963 single blank line, and leading/trailing blank lines are dropped. 

964 

965 The transport's ``keel.run-id`` marker line is dropped too. Without that, closure 

966 fidelity and post-comment idempotency were mutually exclusive: the marker is what 

967 lets a re-post edit its own comment instead of duplicating, and its presence made 

968 the body differ from the canonical render. 

969 """ 

970 lines = [line.rstrip() for line in body.splitlines() if not RUN_ID_MARKER_RE.match(line)] 

971 normalized: list[str] = [] 

972 for line in lines: 

973 if not line and (not normalized or not normalized[-1]): 

974 continue 

975 normalized.append(line) 

976 while normalized and not normalized[-1]: 

977 normalized.pop() 

978 return "\n".join(normalized) 

979 

980 

981def closure_body_matches_record(body: str, record: dict[str, Any]) -> bool: 

982 """Return whether ``body`` matches the canonical render of ``record``.""" 

983 expected = closure.render_closure_comment(record) 

984 return _normalize_closure_body(body) == _normalize_closure_body(expected) 

985 

986 

987def _is_closure_comment( 

988 item: dict[str, Any], 

989 *, 

990 enforced: bool = True, 

991 record: dict[str, Any] | None = None, 

992) -> bool: 

993 if not _is_trusted_source(item, enforced=enforced): 

994 return False 

995 if not _has_closure_marker(_body(item)): 

996 return False 

997 if record is None: 

998 return True 

999 return closure_body_matches_record(_body(item), record) 

1000 

1001 

1002def _run_context_findings( 

1003 *, 

1004 pr_comments: list[dict[str, Any]], 

1005 issue_comments: list[dict[str, Any]], 

1006 enforced: bool, 

1007 ledger_record: dict[str, Any] | None = None, 

1008) -> list[dict[str, Any]]: 

1009 comments = [*pr_comments, *issue_comments] 

1010 findings: list[dict[str, Any]] = [] 

1011 for item in comments: 

1012 if not _is_closure_comment(item, enforced=enforced, record=ledger_record): 

1013 continue 

1014 body = _body(item) 

1015 if _has_empty_run_context(body): 

1016 findings.append( 

1017 { 

1018 "id": "run-context-empty", 

1019 "severity": "major" if enforced else "minor", 

1020 "kind": "closure", 

1021 "message": "Closure comment Run context is fully degraded.", 

1022 } 

1023 ) 

1024 return findings 

1025 

1026 

1027def _has_empty_run_context(body: str) -> bool: 

1028 if "### Run context" not in body: 

1029 return False 

1030 fields = _run_context_fields(body) 

1031 return fields == { 

1032 "host agent": "unknown", 

1033 "transport": "unknown", 

1034 "profile": "unknown", 

1035 "jury": "off", 

1036 "consent": "unknown (scopes: none)", 

1037 } 

1038 

1039 

1040def _run_context_fields(body: str) -> dict[str, str]: 

1041 fields: dict[str, str] = {} 

1042 in_block = False 

1043 for line in body.splitlines(): 

1044 if line.strip() == "### Run context": 

1045 in_block = True 

1046 continue 

1047 if in_block and line.startswith("### "): 

1048 break 

1049 if not in_block: 

1050 continue 

1051 match = re.match(r"^-\s+\*\*(?P<key>[^*]+):\*\*\s+(?P<value>.+?)\s*$", line) 

1052 if match: 

1053 fields[match.group("key").strip().lower()] = match.group("value").strip().lower() 

1054 return fields 

1055 

1056 

1057def _is_trusted_source(item: dict[str, Any], *, enforced: bool = True) -> bool: 

1058 """Return whether GitHub marks this evidence source as trusted. 

1059 

1060 Live GitHub comment/review payloads include ``author_association``. Enforced 

1061 evidence fails closed when that field is absent because offline fixtures are 

1062 agent-writable and must not manufacture trust. Untrusted explicit 

1063 associations fail closed even if the author type is ``Bot``. 

1064 """ 

1065 association = item.get("author_association") 

1066 if association is None: 

1067 return not enforced 

1068 if isinstance(association, str) and association.upper() in TRUSTED_AUTHOR_ASSOCIATIONS: 

1069 return True 

1070 return False 

1071 

1072 

1073def _is_ship_assessment(body: str) -> bool: 

1074 """Whether ``body`` is a ship assessment comment, decided by its header (#1035). 

1075 

1076 Header-anchored for the same reason markers are (#1026): this is consulted as an 

1077 *exclusion* by :func:`_is_review_verdict_body` and :func:`_is_jury_verdict`, so a 

1078 whole-body substring test let a reviewer disarm their own verdict by quoting the 

1079 heading while describing what they reviewed ("the ``### \U0001f6a2 keel ship`` 

1080 comment claims the gates passed, but…"). The verdict was then silently uncounted 

1081 and ``evidence-verify`` reported it missing from a PR it was sitting on. 

1082 

1083 The heading is a Markdown heading rather than a versioned ``keel.*.v1`` marker, so 

1084 it cannot join :data:`CLASSIFICATION_MARKERS`; it gets the same anchoring instead. 

1085 A real assessment leads with the heading (the workflow writes it first) or with the 

1086 CLI's own banner, so nothing that armed the gate through a genuine assessment 

1087 comment stops arming it. 

1088 """ 

1089 header = _header_line(body) 

1090 return header.startswith(SHIP_ASSESSMENT_HEADING) or header.startswith(SHIP_ASSESSMENT_BANNER) 

1091 

1092 

1093def count_review_verdicts( 

1094 pr_comments: list[dict[str, Any]] | None = None, 

1095 pr_reviews: list[dict[str, Any]] | None = None, 

1096 *, 

1097 head_sha: str | None = None, 

1098 covered_heads: Collection[str] = (), 

1099 enforced: bool = True, 

1100 pr_title: str = "", 

1101) -> int: 

1102 """Count distinct trusted review-verdict reviewers for a PR. 

1103 

1104 This is the same evidence-side counting the verify report uses for the 

1105 ``review`` items: it collapses idempotent re-posts by the same reviewer to 

1106 one verdict and only counts trusted, head-bound verdicts whose reviewer's 

1107 latest ``Verdict:`` line approves (#1426) — a reviewer who requested changes 

1108 reviewed, but did not pass, the change. Reused by capture reconcile to 

1109 cross-check the ledger's recorded reviewer count. 

1110 """ 

1111 keys = _review_evidence_keys( 

1112 [*(pr_comments or []), *(pr_reviews or [])], 

1113 head_sha=head_sha, 

1114 covered_heads=covered_heads, 

1115 enforced=enforced, 

1116 pr_title=pr_title, 

1117 ) 

1118 return len(keys) 

1119 

1120 

1121def _review_evidence_keys( 

1122 items: list[dict[str, Any]], 

1123 *, 

1124 head_sha: str | None = None, 

1125 covered_heads: Collection[str] = (), 

1126 enforced: bool = True, 

1127 pr_title: str = "", 

1128) -> set[str]: 

1129 keys, _ = _review_evidence_keys_and_rejections( 

1130 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced, pr_title=pr_title 

1131 ) 

1132 return keys 

1133 

1134 

1135def verdict_reviewers( 

1136 items: list[dict[str, Any]], 

1137 *, 

1138 head_sha: str | None = None, 

1139 covered_heads: Collection[str] = (), 

1140 enforced: bool = True, 

1141) -> tuple[str, ...]: 

1142 """The reviewers whose verdicts count for ``head_sha``, by name, sorted (#1422). 

1143 

1144 The verdicts :func:`verify` counts toward ``review-verdict-N`` — trusted, head-pinned, 

1145 with substance, and approving at the reviewer's latest word (#1426), so a reviewer who 

1146 requested changes is not listed as having passed it — named by their ``reviewer:`` 

1147 field, else by the commenter's login. A 

1148 verdict keyed only by its body names nobody and is left out. This is what a closure 

1149 comment written after the merge says reviewed the change, read off the pull request 

1150 rather than recalled. 

1151 """ 

1152 keys = _review_evidence_keys( 

1153 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced 

1154 ) 

1155 names = [] 

1156 for key in keys: 

1157 kind, _, name = key.partition(":") 

1158 if kind in ("reviewer", "user"): 

1159 names.append(name) 

1160 return tuple(sorted(names)) 

1161 

1162 

1163@dataclass(frozen=True) 

1164class _ReviewTally: 

1165 """What the review verdicts at a head add up to, per reviewer (#1426). 

1166 

1167 ``accepted`` are the reviewer keys whose verdict counts toward ``review-verdict-N``; 

1168 ``vendors`` maps each of them to its declared vendor (or ``None``). ``insubstantial`` 

1169 are the ``(key, reason)`` pairs refused for naming nothing (#926), and 

1170 ``not_approved`` the ``(key, token, head)`` triples of reviewers whose standing 

1171 verdict does not approve (``token`` is ``None`` when it has no readable 

1172 ``Verdict:`` line). Both refusals are reported, never dropped. 

1173 """ 

1174 

1175 accepted: frozenset[str] 

1176 vendors: dict[str, str | None] 

1177 insubstantial: tuple[tuple[str, str], ...] 

1178 not_approved: tuple[tuple[str, str | None, str], ...] 

1179 

1180 

1181def _posted_at(item: dict[str, Any]) -> str: 

1182 """When ``item`` was posted, as GitHub's ISO-8601 string, or ``""`` when unknown. 

1183 

1184 An issue comment carries ``created_at`` and a pull-request review ``submitted_at``. 

1185 ``updated_at`` is deliberately not read: editing an old approval (a typo fix, an 

1186 idempotent re-post of the same run's comment) must not make it outrank a rejection 

1187 posted after it. ISO-8601 UTC strings order lexically, so no clock is parsed. 

1188 """ 

1189 for field in ("created_at", "submitted_at"): 

1190 value = item.get(field) 

1191 if isinstance(value, str) and value: 

1192 return value 

1193 return "" 

1194 

1195 

1196def review_verdict_token(body: str) -> str | None: 

1197 """The upper-cased first word of ``body``'s first ``Verdict:`` line, or ``None`` (#1426). 

1198 

1199 The line is the one :func:`keel.artifacts.render_review_verdict` writes 

1200 (``Verdict: <verdict>``); it is found anywhere in the comment, header block included, 

1201 but only at the start of a line. The first word is read case-insensitively, with 

1202 wrapper punctuation skipped and ``-`` spelled ``_``, so ``APPROVE — minor nits``, 

1203 ``**lgtm**`` and ``request-changes`` read ``APPROVE``, ``LGTM`` and 

1204 ``REQUEST_CHANGES``. ``None`` when the comment has no such line or the line has no 

1205 word, which :func:`verdict_approves` treats as not an approval. 

1206 """ 

1207 return _line_token(body, _VERDICT_LINE_RE) 

1208 

1209 

1210def _line_token(body: str, line_re: re.Pattern[str]) -> str | None: 

1211 """The first word of the first line ``line_re`` matches, read as a verdict token.""" 

1212 for line in body.splitlines(): 

1213 match = line_re.match(line) 

1214 if match is None: 

1215 continue 

1216 token = _VERDICT_TOKEN_RE.match(match.group("value").strip()) 

1217 return token.group("token").upper().replace("-", "_") if token else None 

1218 return None 

1219 

1220 

1221def verdict_approves(body: str) -> bool: 

1222 """Whether a review verdict's ``Verdict:`` line approves (:data:`APPROVING_VERDICTS`).""" 

1223 return review_verdict_token(body) in APPROVING_VERDICTS 

1224 

1225 

1226def jury_verdict_token(body: str) -> str | None: 

1227 """The consensus token on a jury verdict's ``AI Jury verdict:`` line, or ``None`` (#1429). 

1228 

1229 Read exactly as :func:`review_verdict_token` reads a review's ``Verdict:`` line — first 

1230 word, any case, wrapper punctuation skipped, ``-`` spelled ``_`` — off the line 

1231 :func:`keel.artifacts.render_jury_verdict` writes, ``AI Jury verdict: <verdict>.``, so 

1232 its trailing full stop is not part of the token. ``None`` when there is no such line or 

1233 it carries no word. 

1234 """ 

1235 return _line_token(body, _JURY_VERDICT_LINE_RE) 

1236 

1237 

1238def jury_verdict_approves(body: str) -> bool: 

1239 """Whether a jury verdict's consensus approves (:data:`APPROVING_VERDICTS`, #1429). 

1240 

1241 The same set a review verdict is read against: the chair's ``APPROVE`` / ``READY`` reach 

1242 the comment as ``LGTM`` through :func:`keel.jury.map_verdict`. ``REQUEST_CHANGES`` (and 

1243 ``NEEDS_INFO``, which maps onto it), ``COMMENT``, ``ABSTAIN`` — the token 

1244 :func:`keel.jury.jury_verdict` writes when the panel produced no chair synthesis — and 

1245 ``NO_QUORUM`` are not approvals, and neither is an unknown word or a missing line. 

1246 """ 

1247 return jury_verdict_token(body) in APPROVING_VERDICTS 

1248 

1249 

1250def _not_approved_message(key: str, token: str | None, head: str) -> str: 

1251 """How a refusal names a reviewer whose standing verdict does not approve (#1426).""" 

1252 name = _reviewer_name(key) 

1253 if token in REQUEST_CHANGES_VERDICTS: 

1254 return f"{name} requests changes at {head}." 

1255 why = f"verdict {token}" if token else "no readable Verdict line" 

1256 return f"{name} does not approve at {head} ({why})." 

1257 

1258 

1259def _verdict_head(item: dict[str, Any], body: str) -> str: 

1260 """The head a verdict answers for: its ``head:`` field, else the review's commit.""" 

1261 recorded = _fields(body).get("head") 

1262 if recorded: 

1263 return recorded 

1264 commit_id = item.get("commit_id") 

1265 return commit_id if isinstance(commit_id, str) and commit_id else "an unrecorded head" 

1266 

1267 

1268def _reviewer_name(key: str) -> str: 

1269 """A reviewer key as a refusal names it: the reviewer, or the verdict's digest.""" 

1270 kind, _, name = key.partition(":") 

1271 return f"an unnamed reviewer (verdict {name[:12]})" if kind == "body" else name 

1272 

1273 

1274def _review_tally( 

1275 items: list[dict[str, Any]], 

1276 *, 

1277 head_sha: str | None = None, 

1278 covered_heads: Collection[str] = (), 

1279 enforced: bool = True, 

1280 pr_title: str = "", 

1281) -> _ReviewTally: 

1282 """Tally the trusted review verdicts that answer for ``head_sha`` (#1426). 

1283 

1284 **A reviewer's latest verdict is their review.** Verdicts are ordered by when they 

1285 were posted (:func:`_posted_at`, then their position), grouped by reviewer key, and 

1286 each reviewer is judged by the last one — at the head or at a head it descends from 

1287 by capture commits alone (``covered_heads``, #1203): 

1288 

1289 * **It does not approve** (:func:`verdict_approves`): the reviewer is refused, with 

1290 their stance and the head it was given at. A rejection is evidence; dropping it 

1291 silently would let another reviewer's approval outvote it. So a reviewer who 

1292 approved and then requested changes holds the merge. 

1293 * **It approves**: the reviewer counts when any verdict in their run of approvals 

1294 since their last non-approving one has substance (:func:`verdict_substance`) — 

1295 a thin verdict followed by a real one is accepted, because the later comment is 

1296 the review (#926). A reviewer who requested changes and then approved counts; a 

1297 thin approval cannot overturn their own substantive rejection. 

1298 

1299 A verdict pinned to any other head is not read at all: a rejection of an older head 

1300 is answered by the commits that moved it, and the reviewer's next verdict. 

1301 """ 

1302 ordered = sorted(enumerate(items), key=lambda pair: (_posted_at(pair[1]), pair[0])) 

1303 by_key: dict[str, list[tuple[dict[str, Any], str]]] = {} 

1304 for _, item in ordered: 

1305 if not _is_trusted_source(item, enforced=enforced): 

1306 continue 

1307 body = _body(item) 

1308 if not _is_review_verdict_body(body): 

1309 continue 

1310 if not _matches_head(item, body, head_sha, covered_heads): 

1311 continue 

1312 by_key.setdefault(_reviewer_key(item, body), []).append((item, body)) 

1313 accepted: set[str] = set() 

1314 vendors: dict[str, str | None] = {} 

1315 insubstantial: list[tuple[str, str]] = [] 

1316 not_approved: list[tuple[str, str | None, str]] = [] 

1317 for key, verdicts in by_key.items(): 

1318 latest_item, latest_body = verdicts[-1] 

1319 token = review_verdict_token(latest_body) 

1320 if token not in APPROVING_VERDICTS: 

1321 not_approved.append((key, token, _verdict_head(latest_item, latest_body))) 

1322 continue 

1323 approvals: list[str] = [] 

1324 for _, body in reversed(verdicts): 

1325 if not verdict_approves(body): 

1326 break 

1327 approvals.append(body) 

1328 substance = [verdict_substance(body, pr_title=pr_title) for body in approvals] 

1329 if not any(ok for ok, _ in substance): 

1330 insubstantial.extend((key, reason) for _, reason in reversed(substance)) 

1331 continue 

1332 accepted.add(key) 

1333 vendor = _fields(verdicts[0][1]).get("vendor") 

1334 vendors[key] = vendor.lower() if vendor else None 

1335 return _ReviewTally( 

1336 accepted=frozenset(accepted), 

1337 vendors=vendors, 

1338 insubstantial=tuple(insubstantial), 

1339 not_approved=tuple(not_approved), 

1340 ) 

1341 

1342 

1343def _review_evidence_keys_and_rejections( 

1344 items: list[dict[str, Any]], 

1345 *, 

1346 head_sha: str | None = None, 

1347 covered_heads: Collection[str] = (), 

1348 enforced: bool = True, 

1349 pr_title: str = "", 

1350) -> tuple[set[str], list[tuple[str, str]]]: 

1351 """Accepted reviewer keys, and the (key, reason) pairs refused for substance. 

1352 

1353 Rejections are returned rather than dropped so the gate can *hold with a 

1354 reason* — a verdict silently not counted would surface as "missing 

1355 review-verdict-2", which sends the operator looking for a comment that is 

1356 right there (#926). A reviewer whose standing verdict does not approve is in 

1357 neither: :func:`_verdict_not_approved_findings` reports them (#1426). 

1358 """ 

1359 tally = _review_tally( 

1360 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced, pr_title=pr_title 

1361 ) 

1362 return set(tally.accepted), list(tally.insubstantial) 

1363 

1364 

1365def _review_vendor_provenance( 

1366 items: list[dict[str, Any]], 

1367 *, 

1368 head_sha: str | None = None, 

1369 covered_heads: Collection[str] = (), 

1370 enforced: bool = True, 

1371) -> dict[str, str | None]: 

1372 """Map each accepted review-verdict reviewer-key to its declared vendor. 

1373 

1374 The value is the lower-cased ``vendor:`` provenance for that verdict, or 

1375 ``None`` when the verdict carries no vendor field. Keys are exactly 

1376 :func:`_review_evidence_keys`, so duplicate reviewer-keys collapse to one 

1377 entry (idempotent re-posts do not inflate the vendor set), and a reviewer 

1378 whose verdict does not count lends the panel no vendor either (#1426). 

1379 """ 

1380 return dict( 

1381 _review_tally( 

1382 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced 

1383 ).vendors 

1384 ) 

1385 

1386 

1387def _verdict_not_approved_findings( 

1388 items: list[dict[str, Any]], 

1389 *, 

1390 head_sha: str | None, 

1391 covered_heads: Collection[str] = (), 

1392 enforced: bool, 

1393) -> list[dict[str, Any]]: 

1394 """One finding per reviewer whose standing verdict at the head does not approve (#1426). 

1395 

1396 ``major`` when the gate is enforced, so a rejection **holds** the merge with its 

1397 reviewer and head named — ``review-verdict-not-approved: alice requests changes at 

1398 <head>`` in :func:`refusal_reason` — instead of merely not counting, where another 

1399 reviewer's approval could make up the number. ``minor`` otherwise, mirroring the 

1400 other content findings. 

1401 """ 

1402 tally = _review_tally(items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced) 

1403 return [ 

1404 { 

1405 "id": VERDICT_NOT_APPROVED_FINDING, 

1406 "severity": "major" if enforced else "minor", 

1407 "kind": "review", 

1408 "message": _not_approved_message(key, token, head), 

1409 } 

1410 for key, token, head in sorted(tally.not_approved, key=lambda entry: entry[0]) 

1411 ] 

1412 

1413 

1414def distinct_vendor_check( 

1415 vendors: Sequence[str | None], 

1416 *, 

1417 required_count: int, 

1418) -> dict[str, Any]: 

1419 """Pure vendor-distinctness check over review-verdict provenance. 

1420 

1421 ``vendors`` is one entry per accepted review verdict: the declared vendor, or 

1422 ``None`` when the verdict carries no vendor provenance. The check passes only 

1423 when at least ``required_count`` verdicts each declare a vendor and those 

1424 vendors are all distinct. It fails when a required verdict is missing vendor 

1425 provenance, or when two required verdicts share a vendor. 

1426 

1427 Returns ``{ok, reason, duplicated, missing_provenance}``. No I/O — fully 

1428 unit-testable. A non-positive ``required_count`` always passes (nothing to 

1429 require). 

1430 """ 

1431 if required_count <= 0: 

1432 return {"ok": True, "reason": None, "duplicated": [], "missing_provenance": 0} 

1433 present = [vendor for vendor in vendors if vendor] 

1434 missing = len(vendors) - len(present) 

1435 seen: set[str] = set() 

1436 duplicated: list[str] = [] 

1437 for vendor in present: 

1438 if vendor in seen and vendor not in duplicated: 

1439 duplicated.append(vendor) 

1440 seen.add(vendor) 

1441 if len(present) < required_count: 

1442 return { 

1443 "ok": False, 

1444 "reason": "missing vendor provenance on required review verdict(s)", 

1445 "duplicated": duplicated, 

1446 "missing_provenance": missing, 

1447 } 

1448 if duplicated: 

1449 return { 

1450 "ok": False, 

1451 "reason": f"review verdicts share a vendor: {', '.join(sorted(duplicated))}", 

1452 "duplicated": sorted(duplicated), 

1453 "missing_provenance": missing, 

1454 } 

1455 return {"ok": True, "reason": None, "duplicated": [], "missing_provenance": missing} 

1456 

1457 

1458def panel_vendor_check( 

1459 vendors: Sequence[str | None], 

1460 *, 

1461 required_count: int, 

1462 minimum_vendors: int, 

1463) -> dict[str, Any]: 

1464 """Cross-vendor check for a **jury panel**, whose size the panel sets (#1015). 

1465 

1466 :func:`distinct_vendor_check` asks for one distinct vendor per required 

1467 verdict, which is the right question for a bench keel staffs: keel chose the 

1468 seats, so keel can insist each one is a different vendor. It is the wrong 

1469 question for a panel, where the *panel* chose the seats and three ballots from 

1470 two vendors is a legitimate cross-vendor review — the same shape 

1471 :data:`keel.ship.MINIMUM_JURY_VENDORS` already accepts as a gating jury. 

1472 

1473 So the panel is held to the jury's own rule instead: every required ballot 

1474 must declare a vendor, and the ballots together must span at least 

1475 ``minimum_vendors`` distinct ones. A whole panel from one vendor is one 

1476 opinion N times and fails, which is precisely what the strict check would 

1477 have caught — the relaxation is only in *how many* distinct vendors are 

1478 demanded, never in whether provenance is required at all. 

1479 

1480 Returns the same ``{ok, reason, duplicated, missing_provenance}`` shape as 

1481 :func:`distinct_vendor_check`, so a caller renders one finding either way. 

1482 """ 

1483 if required_count <= 0: 

1484 return {"ok": True, "reason": None, "duplicated": [], "missing_provenance": 0} 

1485 present = [vendor for vendor in vendors if vendor] 

1486 missing = len(vendors) - len(present) 

1487 distinct = sorted(set(present)) 

1488 duplicated = sorted({vendor for vendor in present if present.count(vendor) > 1}) 

1489 if len(present) < required_count: 

1490 return { 

1491 "ok": False, 

1492 "reason": "missing vendor provenance on required review verdict(s)", 

1493 "duplicated": duplicated, 

1494 "missing_provenance": missing, 

1495 } 

1496 if len(distinct) < minimum_vendors: 

1497 return { 

1498 "ok": False, 

1499 "reason": ( 

1500 f"jury panel of {len(present)} ballot(s) spans " 

1501 f"{len(distinct)} distinct vendor(s), below the minimum of {minimum_vendors}" 

1502 ), 

1503 "duplicated": duplicated, 

1504 "missing_provenance": missing, 

1505 } 

1506 return {"ok": True, "reason": None, "duplicated": duplicated, "missing_provenance": missing} 

1507 

1508 

1509def _label_values(labels: Sequence[str] | None, prefix: str) -> list[str]: 

1510 """Lower-cased values of every ``<prefix><value>`` label, blanks dropped.""" 

1511 values: list[str] = [] 

1512 for label in labels or (): 

1513 if not isinstance(label, str) or not label.startswith(prefix): 

1514 continue 

1515 value = label[len(prefix) :].strip().lower() 

1516 if value: 

1517 values.append(value) 

1518 return values 

1519 

1520 

1521def agent_label_vendors(labels: Sequence[str] | None) -> list[str]: 

1522 """Return the lower-cased vendor slugs from every ``agent:<vendor>`` label. 

1523 

1524 A blank vendor (a bare ``agent:`` label) is ignored. Order is preserved and 

1525 duplicates are kept so callers can reason about the raw label set; this is a 

1526 pure helper with no I/O. 

1527 """ 

1528 return _label_values(labels, AGENT_LABEL_PREFIX) 

1529 

1530 

1531def model_label_bases(labels: Sequence[str] | None) -> list[str]: 

1532 """Return the lower-cased base slugs from every ``model:<base>`` label. 

1533 

1534 The mirror of :func:`agent_label_vendors` for the second half of keel's 

1535 attribution vocabulary. Same conventions: blanks dropped, order and duplicates 

1536 preserved, no I/O. 

1537 """ 

1538 return _label_values(labels, MODEL_LABEL_PREFIX) 

1539 

1540 

1541def ledger_implementer(ledger_record: dict[str, Any] | None) -> str | None: 

1542 """Return the raw ``actors.implementer`` string from a ship_run record, or ``None``. 

1543 

1544 The full ``vendor`` / ``vendor:model`` value, not just the vendor half — the 

1545 vocabulary check needs the model too. Blank/absent reads as ``None``. Pure. 

1546 """ 

1547 if not isinstance(ledger_record, dict): 

1548 return None 

1549 actors = ledger_record.get("actors") 

1550 implementer = actors.get("implementer") if isinstance(actors, dict) else None 

1551 if not isinstance(implementer, str) or not implementer.strip(): 

1552 return None 

1553 return implementer.strip() 

1554 

1555 

1556def ledger_implementer_vendor(ledger_record: dict[str, Any] | None) -> str | None: 

1557 """Return the implementer's vendor slug from a ship_run ``ledger_record``. 

1558 

1559 The ledger stores the effective implementer as a codename or ``vendor:model`` 

1560 string under ``actors.implementer``; the vendor is the part before the first 

1561 ``:``. Returns ``None`` when no record, no implementer, or a blank implementer 

1562 is recorded so the cross-check can degrade to presence-only. Pure — no I/O. 

1563 """ 

1564 implementer = ledger_implementer(ledger_record) 

1565 if implementer is None: 

1566 return None 

1567 vendor, _ = agents.split_delegate(implementer) 

1568 vendor = vendor.strip().lower() 

1569 return vendor or None 

1570 

1571 

1572def attribution_check( 

1573 labels: Sequence[str] | None, 

1574 *, 

1575 implementer_vendor: str | None = None, 

1576) -> dict[str, Any]: 

1577 """Pure attribution-label check over a PR's labels and the ledger implementer. 

1578 

1579 Two layers, both fail-closed only on a real contradiction: 

1580 

1581 * **Presence** — at least one non-blank ``agent:<vendor>`` label must exist. 

1582 Missing one is a ``missing-label`` finding. 

1583 * **Cross-check** — when ``implementer_vendor`` is known (a ship_run record 

1584 recorded an implementer), one of the PR's ``agent:*`` vendors must match it. 

1585 A mismatch is a ``vendor-mismatch`` finding. When ``implementer_vendor`` is 

1586 ``None`` (no record / no implementer) only the presence layer runs, so PRs 

1587 that predate attribution recording are not broken. 

1588 

1589 Returns ``{ok, reason, label_vendors, implementer_vendor}``. No I/O. 

1590 """ 

1591 label_vendors = agent_label_vendors(labels) 

1592 implementer = implementer_vendor.strip().lower() if implementer_vendor else None 

1593 if not label_vendors: 

1594 return { 

1595 "ok": False, 

1596 "reason": "missing-label", 

1597 "label_vendors": label_vendors, 

1598 "implementer_vendor": implementer, 

1599 } 

1600 if implementer is not None and implementer not in label_vendors: 

1601 return { 

1602 "ok": False, 

1603 "reason": "vendor-mismatch", 

1604 "label_vendors": label_vendors, 

1605 "implementer_vendor": implementer, 

1606 } 

1607 return { 

1608 "ok": True, 

1609 "reason": None, 

1610 "label_vendors": label_vendors, 

1611 "implementer_vendor": implementer, 

1612 } 

1613 

1614 

1615def attribution_vocabulary_check( 

1616 labels: Sequence[str] | None, 

1617 *, 

1618 implementer: str | None, 

1619) -> dict[str, Any]: 

1620 """Check a PR's attribution labels against keel's own vocabulary (#1013). 

1621 

1622 :func:`attribution_check` compares the label's *vendor* with the ledger's 

1623 *vendor*. That is a comparison of two hand-written strings: when the host wrote 

1624 ``agent:gemini`` on the PR **and** ``gemini:gemini-3.8-flash-high`` into the 

1625 ledger, the two agreed and the gate passed — while keel's own vocabulary for that 

1626 run is ``agent:agy`` / ``model:gemini-3``. This check closes that hole by deriving 

1627 the expected labels from :func:`keel.agents.attribution` instead of comparing the 

1628 prose to itself. 

1629 

1630 Only labels the PR actually carries are judged: a missing ``agent:`` label is 

1631 :func:`attribution_check`'s ``missing-label`` finding and is not repeated here, 

1632 and a ledger implementer with no model (``claude``) yields no expected 

1633 ``model:`` label, so ``model:`` labels are left alone in that case. 

1634 

1635 Returns ``{ok, checked, reason, expected, actual, implementer}``. ``checked`` is 

1636 ``False`` when there was nothing to compare against — no ledger record, no 

1637 recorded implementer — so a caller can tell "agrees" from "could not tell". 

1638 Pure — no I/O. 

1639 """ 

1640 expected = agents.attribution_from_implementer(implementer) 

1641 actual = { 

1642 "agent_labels": agent_label_vendors(labels), 

1643 "model_labels": model_label_bases(labels), 

1644 } 

1645 if expected is None: 

1646 return { 

1647 "ok": True, 

1648 "checked": False, 

1649 "reason": "no-implementer", 

1650 "expected": None, 

1651 "actual": actual, 

1652 "implementer": None, 

1653 } 

1654 recorded = implementer.strip() if isinstance(implementer, str) else None 

1655 result = { 

1656 "ok": True, 

1657 "checked": True, 

1658 "reason": None, 

1659 "expected": dict(expected), 

1660 "actual": actual, 

1661 "implementer": recorded, 

1662 } 

1663 expected_agent = expected["agent_label"][len(AGENT_LABEL_PREFIX) :] 

1664 expected_model = expected["model_label"] 

1665 if actual["agent_labels"] and expected_agent not in actual["agent_labels"]: 

1666 result["ok"] = False 

1667 result["reason"] = "agent-label" 

1668 return result 

1669 if expected_model is not None: 

1670 base = expected_model[len(MODEL_LABEL_PREFIX) :] 

1671 if actual["model_labels"] and base not in actual["model_labels"]: 

1672 result["ok"] = False 

1673 result["reason"] = "model-label" 

1674 return result 

1675 

1676 

1677def _attribution_vocabulary_finding( 

1678 *, 

1679 pr_labels: Sequence[str] | None, 

1680 enforced: bool, 

1681 ledger_record: dict[str, Any] | None, 

1682) -> dict[str, Any] | None: 

1683 """Return the blocking ``attribution-vocabulary`` finding, or ``None``. 

1684 

1685 Skips — never fails — when the gate is inactive, when labels were not fetched, or 

1686 when no ledger record named an implementer: the check needs a recorded implementer 

1687 to derive the expected labels from, and refusing a PR because keel could not read 

1688 its own ledger would be a fail-closed rule with nothing behind it. 

1689 """ 

1690 if not enforced or pr_labels is None: 

1691 return None 

1692 result = attribution_vocabulary_check( 

1693 pr_labels, 

1694 implementer=ledger_implementer(ledger_record), 

1695 ) 

1696 if result["ok"]: 

1697 return None 

1698 expected = result["expected"] 

1699 labels = ", ".join( 

1700 label for label in (expected["agent_label"], expected["model_label"]) if label 

1701 ) 

1702 observed = ", ".join( 

1703 [ 

1704 *(f"{AGENT_LABEL_PREFIX}{value}" for value in result["actual"]["agent_labels"]), 

1705 *(f"{MODEL_LABEL_PREFIX}{value}" for value in result["actual"]["model_labels"]), 

1706 ] 

1707 ) 

1708 return { 

1709 "id": "attribution-vocabulary", 

1710 "severity": "major", 

1711 "kind": "attribution", 

1712 "message": ( 

1713 f"PR attribution labels ({observed}) are not keel's vocabulary for ledger " 

1714 f"implementer {result['implementer']!r}. Expected: {labels}. " 

1715 "Obtain labels from `keel attribution` instead of composing them by hand." 

1716 ), 

1717 } 

1718 

1719 

1720def _unarmed_finding( 

1721 *, 

1722 enforced: bool, 

1723 dry_run: bool, 

1724 require_armed: bool, 

1725 waived: bool, 

1726) -> dict[str, Any] | None: 

1727 """Return a blocking finding when the gate was never armed, else ``None``. 

1728 

1729 Opt-in via ``require_armed`` so existing callers keep today's behavior. An 

1730 unarmed gate derives no requirements, so without this the report passes 

1731 having verified nothing — indistinguishable from a genuine pass. Skipped 

1732 under ``dry_run``, where producing no evidence is the expected outcome, and 

1733 when ``waived``: an operator disarming the gate on purpose is the sanctioned 

1734 way out, and the whole point is to separate that from arming by accident. 

1735 """ 

1736 if not require_armed or dry_run or enforced or waived: 

1737 return None 

1738 return { 

1739 "id": "gate-unarmed", 

1740 "severity": "major", 

1741 "kind": "arming", 

1742 "message": ( 

1743 "Evidence gate is not armed, so no requirements were checked. Arm it via ship " 

1744 "provenance (the keel.ship-provenance.v1 comment a live run posts on its PR, a " 

1745 "ship branch, a posted review verdict, the ship-run ledger, or the gate label), " 

1746 "or disarm deliberately with the operator waiver label." 

1747 ), 

1748 } 

1749 

1750 

1751def _attribution_finding( 

1752 *, 

1753 pr_labels: Sequence[str] | None, 

1754 enforced: bool, 

1755 ledger_record: dict[str, Any] | None, 

1756) -> dict[str, Any] | None: 

1757 """Return a blocking attribution finding when the gate is active, else ``None``. 

1758 

1759 Only runs when the evidence gate is active (``enforced``) *and* PR labels were 

1760 actually fetched (``pr_labels is not None``): the presence check is cheap and 

1761 default-on, while the vendor cross-check engages only when the ledger recorded 

1762 an implementer vendor. Degrades gracefully — labels not available skips the 

1763 check entirely, no record means presence-only, and a gate-inactive run skips 

1764 the check (back-compat with callers that never pass labels). 

1765 """ 

1766 if not enforced or pr_labels is None: 

1767 return None 

1768 implementer_vendor = ledger_implementer_vendor(ledger_record) 

1769 result = attribution_check(pr_labels, implementer_vendor=implementer_vendor) 

1770 if result["ok"]: 

1771 return None 

1772 if result["reason"] == "missing-label": 

1773 message = "PR is missing a mandatory agent:<vendor> attribution label." 

1774 else: 

1775 message = ( 

1776 "PR agent:<vendor> attribution " 

1777 f"({', '.join(result['label_vendors'])}) does not match the ship_run " 

1778 f"ledger implementer vendor ({result['implementer_vendor']})." 

1779 ) 

1780 return { 

1781 "id": "attribution-label", 

1782 "severity": "major", 

1783 "kind": "attribution", 

1784 "message": message, 

1785 } 

1786 

1787 

1788#: A file named without its directory — ``evidence.py``, ``CHANGELOG.md``. 

1789#: 

1790#: The path anchors read *any* one-to-five-character extension, because a 

1791#: directory has already proved the token is a path. With no directory nothing 

1792#: is left to carry that proof: ``Node.js``, ``Next.js``, ``Vue.js`` and 

1793#: ``D3.js`` are spelled exactly like ``evidence.py``, and in prose a product 

1794#: is named far more often than a bare file is. Listing the extension cannot 

1795#: separate them — ``.js`` is a real source extension, and dropping it would 

1796#: refuse real reviews to refuse four product names — so this form does not 

1797#: anchor a verdict by itself. It *corroborates*: 

1798#: see :data:`_VERDICT_CORROBORATORS`. 

1799_VERDICT_SOURCE_FILE = re.compile( 

1800 r"\b[\w-]+\.(?:py|pyi|md|rst|txt|ya?ml|toml|json|cfg|ini|lock" 

1801 r"|sh|js|mjs|cjs|ts|tsx|jsx|rb|go|rs|css|html?|svg|sql)\b" 

1802) 

1803 

1804#: ``module.symbol`` / ``Class.method`` — the dotted identifier a review writes 

1805#: for something it is not calling (#1106), read only when the token carries a 

1806#: mark prose does not use: an underscore, an internal capital, a run of 

1807#: capitals, or a capitalised segment. 

1808#: 

1809#: **A dot is not evidence, and neither is a capital.** Requiring the mark 

1810#: stops ``github.com`` and ``pypi.org``. It does not stop ``GitHub.com``, 

1811#: ``GitLab.com``, ``OpenAI.com`` or ``SourceForge.net``, because a capitalised 

1812#: first segment is precisely the mark ``Config.parse`` carries — the two are 

1813#: one shape, and no sixth character class tells them apart. That is why this 

1814#: form corroborates rather than anchors (:data:`_VERDICT_CORROBORATORS`): the 

1815#: alternative was a list of hostnames to refuse, which cannot be finished, 

1816#: because anyone can register the next one. 

1817#: 

1818#: Both sides of the dot still need two characters, so "e.g." and "i.e." are not 

1819#: identifiers whatever else they carry. 

1820_VERDICT_DOTTED_SYMBOL = re.compile( 

1821 r"\b(?=[\w.]*(?:_|[a-z0-9][A-Z]|[A-Z]{2}|[A-Z][a-z]))" 

1822 r"[A-Za-z_]\w+(?:\.[A-Za-z_]\w+)+" 

1823) 

1824 

1825#: A **bare** identifier: ``cache_key``, ``_prompt_mode``, ``__post_init__``. 

1826#: 

1827#: **The joining underscore is the whole rule** — two lowercase alphanumeric 

1828#: segments with an underscore between them. Lowercase deliberately: `My_Thing` 

1829#: is prose with a connector, and CamelCase is read only when dotted or 

1830#: backticked, for the reason :data:`_VERDICT_DOTTED_SYMBOL` gives. 

1831#: 

1832#: Underscores that merely wrap a name are 

1833#: decoration, so a lone ``_private`` or ``__dunder__`` is *not* read; it is 

1834#: ``post_init`` inside ``__post_init__`` that matches. 

1835#: 

1836#: The joining underscore is a mark English prose and product names do not use, 

1837#: and it is the only lexical difference between a symbol and a capitalised 

1838#: noun: ``GitHub`` and ``GitLab`` are CamelCase in exactly the way 

1839#: ``JuryConfig`` is, so reading CamelCase let "The GitHub and GitLab side of 

1840#: this is unchanged" clear a floor of two while naming nothing in the change. 

1841#: No pattern separates those two, and a list of product names to refuse would 

1842#: need a new entry every time a reviewer mentions a new product. CamelCase 

1843#: symbols are still read everywhere they are written with a dot or in 

1844#: backticks, which is how a class is normally named. 

1845_VERDICT_BARE_IDENTIFIER = re.compile(r"\b_*[a-z0-9]+(?:_[a-z0-9]+)+_*\b") 

1846 

1847#: A concrete thing a review can point at, on its own — a ``path:line``, a 

1848#: path, a backticked symbol, or a called identifier. Presence of *structure*, 

1849#: never a judgement about whether the review was good — the same line 

1850#: ai-jury's ``emitted_findings_block()`` draws. 

1851#: 

1852#: What these four have and :data:`_VERDICT_CORROBORATORS` do not is that the 

1853#: **punctuation is the author pointing**. Backticks, a directory separator, a 

1854#: ``:42`` and a ``()`` are marks a reviewer types deliberately at a thing; 

1855#: none of them appear around a product name in ordinary prose. A bare dotted 

1856#: token is just a token with a dot in it. 

1857_VERDICT_ANCHORS = re.compile( 

1858 r"[\w./-]+\.[A-Za-z0-9]{1,5}:\d+|" # path/to/file.py:42 

1859 r"[\w-]+/[\w./-]+\.[A-Za-z0-9]{1,5}\b|" # src/keel/thing.py 

1860 r"`[^`\n]{2,}`|" # `a_symbol`, `--a-flag` 

1861 r"\b\w+\.\w+\(\)" # module.function() 

1862) 

1863 

1864#: The unbackticked forms — a bare filename, a bare dotted token, a bare 

1865#: identifier — each of which is *also* how something that is not a symbol gets 

1866#: written. Two of them anchor a verdict; one does not (#1106). 

1867#: 

1868#: **This is a shape, not a lexicon, and that is the point.** Three rounds of 

1869#: widening the anchor set were each undone by a token that is not a symbol: 

1870#: ``claude.ai`` in a tool footer, ``GitHub``/``GitLab`` as bare CamelCase, 

1871#: then ``Node.js`` and ``GitHub.com``. Every one was answered by refining a 

1872#: character class, and every refinement admitted the next such token, because 

1873#: ``Node.js`` and ``evidence.py`` are the same shape and so are ``GitHub.com`` 

1874#: and ``Config.parse``. No character class ends that sequence and no blocklist 

1875#: of products or hostnames can be finished. What separates a review from a 

1876#: mention is not the spelling of one token but how many the verdict has: prose 

1877#: mentions a product in passing; a review that walked the change names more 

1878#: than one thing. A clause that says an act of review happened 

1879#: (:data:`_VERDICT_CHECKED_CLAUSE`) is the one exception, and only for 

1880#: "checked": see there for why the other verbs could not be given the same 

1881#: latitude, and why judging their object by this same test made the branch 

1882#: inert. 

1883#: 

1884#: Measured over every verdict posted across keel and ai-jury: corroboration 

1885#: refuses all four ``*.js`` product names and all four capitalised hostnames, 

1886#: costs 30 of the 141 verdicts #1106 recovered — 22 of those 30 being the 

1887#: default template's "Scope reviewed:" line with a single token dropped into 

1888#: it — and regresses nothing that passed before #1106. Dropping the two bare 

1889#: forms outright instead, the fallback, refuses the same eight strings but 

1890#: keeps only 98 of the 141 and loses a review naming ``evidence.py`` and 

1891#: ``contracts.py``, which is a review. 

1892_VERDICT_CORROBORATORS = ( 

1893 _VERDICT_SOURCE_FILE, # evidence.py, CHANGELOG.md 

1894 _VERDICT_DOTTED_SYMBOL, # cache.cache_key, JuryConfig.__post_init__ 

1895 _VERDICT_BARE_IDENTIFIER, # collect_static_hints, _prompt_mode 

1896) 

1897 

1898#: How many distinct corroborating tokens stand in for an anchor. At one, the 

1899#: corpus says the rule readmits 22 of the 75 ``Reviewed <title>: <affirmation>`` 

1900#: rubber stamps #926 is named for; at two it admits none of them. 

1901_VERDICT_CORROBORATION_FLOOR = 2 

1902 

1903#: What may sit between an act-of-review verb and its object: an optional colon, 

1904#: and an optional break onto a bullet, because "Checked:\n- …" is the same 

1905#: clause with a list under it and refusing it was punctuation pedantry (#1106). 

1906#: Layout only — it says nothing about what the object has to be. 

1907_VERDICT_CLAUSE_TAIL = r"[ \t]*:?[ \t]*(?:\r?\n[ \t]*[-*][ \t]*)?" 

1908 

1909#: One sentence's worth of characters. A sentence ends at `.!?;` or an ellipsis 

1910#: **followed by space or line end** — a period inside `evidence.py` is not a 

1911#: sentence end, and the filename has to survive inside an object, so "Checked 

1912#: evidence.py and contracts.py" is no longer truncated at the first dot — 

1913#: the punctuation pedantry this change is about. Only 

1914#: :data:`_VERDICT_CHECKED_CLAUSE` reads this; the second clause that did was 

1915#: removed with the verb widening. 

1916_VERDICT_SENTENCE = r"(?:[^.!?;\u2026\n]|[.!?;\u2026](?!\s|$))" 

1917 

1918#: The escape hatch the issue insists on: a genuinely clean review must stay 

1919#: expressible. "Checked X, Y and Z; found nothing" is a real review outcome and 

1920#: must not be forced to invent an anchor. Its object stays free-form, as it has 

1921#: been since #926: 35 verdicts in the corpus pass on this clause and nothing 

1922#: else, saying things like "Checked the formula syntax, the version URL and the 

1923#: checksum placeholder" — English objects, naming no symbol. 

1924#: 

1925#: The object now ends at a *sentence* rather than at any period, which is a 

1926#: change from `[^.\n]{8,}`: it buys a filename, since `evidence.py` no longer 

1927#: truncates to `evidence`, and it costs a trailing `;`, `!` or `?` clause — 

1928#: "Checked config; found nothing of concern in the rest" passes on `main` and 

1929#: is refused here. No verdict in the 1,421 measured is written that way, which 

1930#: is why the corpus shows no regression; that is a weaker claim than "takes 

1931#: nothing away" and is the one the evidence supports. 

1932#: 

1933#: **#1106 tried to widen this to traced/read/ran/inspected/verified and the 

1934#: widening turned out inert.** Those verbs could not keep a free-form object — 

1935#: "Read the whole diff and everything looks correct" is the #926 receipt with a 

1936#: synonym at the front — so their object was made to name something. But 

1937#: "names something" is the same test the whole prose already takes, and the 

1938#: object is part of the prose, so a verdict that satisfied the clause had 

1939#: always satisfied the anchor check first: across 1,421 verdicts the branch 

1940#: decided **zero** of them. Dead code with a docstring explaining what it did 

1941#: is worse than neither, so it is gone. Widening the vocabulary needs the 

1942#: object to be judged by something other than the prose test, which #1106 did 

1943#: not find. 

1944_VERDICT_CHECKED_CLAUSE = re.compile( 

1945 rf"\bchecked\b{_VERDICT_CLAUSE_TAIL}{_VERDICT_SENTENCE}{{8,}}", re.IGNORECASE 

1946) 

1947 

1948#: Below this share of novel words, the prose is the PR title said again. The 

1949#: observed shape was `Reviewed <title>: <generic affirmation>` — 75 of 75 

1950#: verdicts across 25 PRs (#926). 

1951_VERDICT_NOVELTY_FLOOR = 0.35 

1952 

1953_WORD = re.compile(r"[a-z0-9]+") 

1954 

1955 

1956def _verdict_prose(body: str) -> str: 

1957 """The verdict's own words: header block, marker line and HTML comments removed. 

1958 

1959 The marker is matched as a *whole line*, never as a substring. It is rendered 

1960 on its own bare line, so the header slice above has already dropped it; a 

1961 substring test therefore only ever reached prose that *quotes* the marker, 

1962 and deleted it. That cost the review of #1119 its entire scope — a 

1963 1,700-character line naming four files was dropped because it named the 

1964 marker it was documenting, and the verdict was then refused for naming 

1965 nothing (#1120). 

1966 

1967 This is the rule :func:`marker_in_header` already states earlier in this 

1968 module: a marker below the header is prose, and ``MARKER in body`` cannot 

1969 tell the two apart (#1026). This function was the one place that substring 

1970 test survived. 

1971 """ 

1972 lines = body.splitlines() 

1973 start = 0 

1974 for index, raw in enumerate(lines): 

1975 if not raw.strip(): 

1976 start = index + 1 

1977 break 

1978 kept = [ 

1979 line 

1980 for line in lines[start:] 

1981 if line.strip() 

1982 and not line.strip().startswith("<!--") 

1983 and line.strip() != REVIEW_VERDICT_MARKER 

1984 ] 

1985 return "\n".join(kept) 

1986 

1987 

1988def _verdict_corroborators(text: str) -> set[str]: 

1989 """The distinct unbackticked tokens in ``text``, counting each one once. 

1990 

1991 A written token is one piece of evidence however many patterns read it: 

1992 ``cache.cache_key`` is matched by :data:`_VERDICT_DOTTED_SYMBOL` whole and by 

1993 :data:`_VERDICT_BARE_IDENTIFIER` as its second half, and counting it twice 

1994 would let one token clear a floor that exists to require two. Matches 

1995 contained inside a longer match are therefore dropped, and what survives is 

1996 deduplicated by text, so a reviewer who names ``evidence.py`` three times has 

1997 still named one file. 

1998 """ 

1999 spans = sorted( 

2000 ( 

2001 (match.start(), match.end(), match.group(0)) 

2002 for pattern in _VERDICT_CORROBORATORS 

2003 for match in pattern.finditer(text) 

2004 ), 

2005 key=lambda span: (span[0], -span[1]), 

2006 ) 

2007 kept: list[tuple[int, int, str]] = [] 

2008 for span in spans: 

2009 if any(span[0] >= start and span[1] <= end for start, end, _ in kept): 

2010 continue 

2011 kept.append(span) 

2012 return {token for _, _, token in kept} 

2013 

2014 

2015def _review_act_clause(prose: str) -> bool: 

2016 """Whether ``prose`` says an act of review was performed on something.""" 

2017 return bool(_VERDICT_CHECKED_CLAUSE.search(prose)) 

2018 

2019 

2020def verdict_substance(body: str, *, pr_title: str = "") -> tuple[bool, str]: 

2021 """Whether a verdict engages with the diff at all. ``(ok, reason)``. 

2022 

2023 The evidence gate verified that verdicts *exist* with the right marker, head 

2024 SHA and distinct reviewer ids — never that any of them looked at anything. A 

2025 verdict engaging with nothing was indistinguishable from one that caught a 

2026 blocker, and the record showed what that permits: 75 of 75 verdicts `pass`, 

2027 all opening `Reviewed <PR title>: <affirmation>`, across 25 PRs that produced 

2028 no review-driven commit between them (#926). 

2029 

2030 Two mechanical requirements, both content-agnostic beyond structure: 

2031 

2032 * **An anchor.** A ``path:line``, a path, a backticked symbol or a called 

2033 identifier, whose punctuation is the author pointing — or, lacking one, 

2034 two distinct unbackticked tokens, because ``Node.js`` and ``evidence.py`` 

2035 are one shape and only the count separates a mention from a review — or 

2036 an act-of-review clause naming what was looked at, because a genuinely 

2037 clean review must stay expressible and forcing it to invent a file 

2038 reference would make the check worse than nothing. 

2039 * **Novelty against the title.** Prose that is substantially the PR title 

2040 restated is the observed shape, and it survives the anchor test whenever 

2041 the title happens to contain a path. 

2042 

2043 The two are independent on purpose, and that is what lets the anchor set be 

2044 generous (#1106). An anchor asks whether the reviewer pointed at anything; 

2045 the novelty floor asks whether the prose is the title said again. A verdict 

2046 that names a symbol *and* is otherwise the title restated fails the second 

2047 check, so widening the first cannot readmit the #926 shape by itself. 

2048 

2049 This says nothing about whether a review was *good*. It cannot, and trying 

2050 would make the gate a critic. It distinguishes a review from a receipt. 

2051 """ 

2052 prose = _verdict_prose(body) 

2053 if not prose.strip(): 

2054 return False, "verdict has no prose beyond its header" 

2055 

2056 # ⚡ Bolt Optimization: Use combined compiled regex instead of generator overhead 

2057 anchored = bool(_VERDICT_ANCHORS.search(prose)) or ( 

2058 len(_verdict_corroborators(prose)) >= _VERDICT_CORROBORATION_FLOOR 

2059 ) 

2060 if not anchored and not _review_act_clause(prose): 

2061 return False, ( 

2062 "verdict names nothing concrete — no file, line, symbol, no two " 

2063 "of a filename/dotted name/identifier, and no 'checked …' clause, " 

2064 "so it cannot be told apart from a receipt" 

2065 ) 

2066 

2067 title_words = set(_WORD.findall(pr_title.lower())) 

2068 prose_words = _WORD.findall(prose.lower()) 

2069 if title_words and prose_words: 

2070 novel = [word for word in prose_words if word not in title_words] 

2071 if len(novel) / len(prose_words) < _VERDICT_NOVELTY_FLOOR: 

2072 return False, "verdict is substantially the pull request title restated" 

2073 return True, "" 

2074 

2075 

2076def _reviewer_key(item: dict[str, Any], body: str) -> str: 

2077 fields = _fields(body) 

2078 reviewer = fields.get("reviewer") 

2079 if reviewer: 

2080 return f"reviewer:{reviewer.lower()}" 

2081 user = item.get("user") 

2082 if isinstance(user, dict) and isinstance(user.get("login"), str) and user["login"]: 

2083 return f"user:{user['login'].lower()}" 

2084 digest = hashlib.sha256(body.encode("utf-8")).hexdigest() 

2085 return f"body:{digest}" 

2086 

2087 

2088def _matches_head( 

2089 item: dict[str, Any], 

2090 body: str, 

2091 head_sha: str | None, 

2092 covered_heads: Collection[str] = (), 

2093) -> bool: 

2094 """Does this comment answer for ``head_sha``? A blank head means *do not filter*. 

2095 

2096 **Deliberately not :func:`keel.juryavail.is_pinnable_head`'s rule, and the difference 

2097 is worth stating** (#1068). This one filters *evidence items* inside a gate that, with 

2098 no head resolved, is head-agnostic from end to end — every review verdict counts, so 

2099 holding jury verdicts alone to a head nobody knows would refuse a gate the rest of 

2100 which is already unfiltered. Nothing reached through here removes a requirement: 

2101 :func:`_review_evidence_keys` and :func:`_review_vendor_provenance` count verdicts 

2102 towards one, and :func:`jury_panel_size` feeds ``max(declared, minimum_vendors)``, so a 

2103 stale ``panelists:`` can only ever raise the bar. 

2104 

2105 :func:`jury_participating_vendors` was the exception, and #1069 closed it rather than 

2106 changing this predicate: its count *can* downgrade a gating jury to advisory (#1015), so 

2107 it now asks :func:`keel.juryavail.is_pinnable_head` for itself before it reads anything. 

2108 Its head rule is its own, for the reason a pin's is — it removes a requirement — and the 

2109 two readers that do not remove one keep this reading. That is the whole distinction 

2110 between the three panel-shaped readers layered here, and it is a property of what each 

2111 one's answer can *do*, not of where it is read from. 

2112 

2113 A *pin* is the case that cannot use this reading, because it does remove requirements — 

2114 it takes ``review-verdict-1..3`` off the required set entirely. So the pin 

2115 (:func:`keel.juryavail.pin`, read by :func:`keel.cli._shipped_jury_availability`) 

2116 refuses a blank head before :func:`panel_verdict_posted` is asked at all, rather than 

2117 this predicate changing under the surfaces that need the permissive one. 

2118 """ 

2119 if not head_sha: 

2120 return True 

2121 # **A head the current one descends from by capture commits alone** answers for it 

2122 # too (#1203). The learning is the pull request's last commit, written after review 

2123 # and before the merge, so every verdict would otherwise be pinned to a head the 

2124 # branch has already moved past. `covered_heads` is never filled from a comment or an 

2125 # argument an agent supplies: its only producer is `capture.capture_only_descent`, 

2126 # which holds each commit in between to one parent, the landing marker, and exactly 

2127 # one path inside the configured sink — so what it admits is a lesson, never code. 

2128 accepted = {head_sha, *covered_heads} 

2129 fields = _fields(body) 

2130 recorded = fields.get("head") 

2131 if recorded: 

2132 return recorded in accepted 

2133 commit_id = item.get("commit_id") 

2134 return isinstance(commit_id, str) and commit_id in accepted 

2135 

2136 

2137def _fields(body: str) -> dict[str, str]: 

2138 """Parse the header block at the top of a verdict comment, and only that. 

2139 

2140 Scanning stops at the first line that is not part of a header block — 

2141 whether or not a header has been seen yet. #868's second requirement said so 

2142 and only the first shipped: the earlier version `continue`d past prose and 

2143 blank lines until it found something header-shaped, so fields could be 

2144 harvested from anywhere in a comment (#932): 

2145 

2146 "Some prose line here.\\n\\nhead: 0000000\\nvendor: spoofed\\n" 

2147 -> {'head': '0000000', 'vendor': 'spoofed'} 

2148 

2149 Reachable through :func:`_reviewer_key`, which calls this with no marker 

2150 requirement, so a comment whose *prose* contains ``reviewer: someone`` was 

2151 keyed to that reviewer. Narrow — it needs a trusted author — and not a live 

2152 hole, but it is the residual of the class #868 named, and stopping costs the 

2153 real comment format nothing: a verdict's header is its first block. 

2154 """ 

2155 fields: dict[str, str] = {} 

2156 started = False 

2157 

2158 for raw_line in (body or "").splitlines(): 

2159 line = raw_line.strip() 

2160 if not line: 

2161 # Leading blank lines are skipped, not treated as the end: a comment 

2162 # body routinely begins with a newline, and breaking there would 

2163 # reject legitimate verdicts. Once the block has started, a blank 

2164 # line ends it — that is the boundary #868 asked for. 

2165 if started: 

2166 break 

2167 continue 

2168 started = True 

2169 # A marker-only line is the artifact's own header, not a field: skip it and 

2170 # keep reading. A line that merely *mentions* a marker is prose, and prose 

2171 # ends the block — the #932 boundary this parser exists to hold. 

2172 if ( 

2173 line.startswith("<!--") and line.endswith("-->") 

2174 ) or _CLASSIFICATION_MARKERS_SET.issuperset(line.split()): 

2175 continue 

2176 match = _FIELD_RE.match(line) 

2177 if match: 

2178 key = match.group("key").lower() 

2179 if key not in fields: 

2180 fields[key] = match.group("value") 

2181 elif not _HEADER_LINE_RE.match(line): 

2182 break 

2183 return fields 

2184 

2185 

2186def _is_review_verdict_body(body: str) -> bool: 

2187 """Whether ``body`` is a review verdict, decided by its header alone (#1026). 

2188 

2189 The jury and closure exclusions are no longer separate substring tests: 

2190 :func:`marker_in_header` yields at most one marker, so a body anchored to the 

2191 jury or closure marker simply is not a review verdict, and a review verdict 

2192 that *mentions* either one in its prose still is. 

2193 """ 

2194 if _is_ship_assessment(body): 

2195 return False 

2196 return marker_in_header(body) == REVIEW_VERDICT_MARKER 

2197 

2198 

2199def _has_trusted_review_marker(items: list[dict[str, Any]]) -> bool: 

2200 return any( 

2201 _is_trusted_source(item, enforced=True) 

2202 and marker_in_header(_body(item)) == REVIEW_VERDICT_MARKER 

2203 for item in items 

2204 ) 

2205 

2206 

2207def jury_participating_vendors( 

2208 pr_comments: list[dict[str, Any]] | None = None, 

2209 pr_reviews: list[dict[str, Any]] | None = None, 

2210 *, 

2211 head_sha: str | None = None, 

2212 covered_heads: Collection[str] = (), 

2213 enforced: bool = True, 

2214) -> int | None: 

2215 """Return the panel size declared by a posted jury verdict, or ``None``. 

2216 

2217 Reads the ``vendors: <N>`` field from a trusted, head-bound 

2218 ``keel.jury-verdict.v1`` comment. This is how the participating-vendor count 

2219 reaches a CI evidence check: the run ledger and the jury artifact both live 

2220 under the gitignored ``.keel/state/``, so a hosted runner cannot read either, 

2221 but PR comments are always visible. 

2222 

2223 ``None`` means "not declared" — no jury verdict posted, or one that predates 

2224 the field — and leaves the jury mode untouched rather than assuming a short 

2225 panel. Only a verdict that actually states the count may relax the gate. 

2226 

2227 When several verdicts qualify, the largest declared count wins: a re-post 

2228 correcting an earlier partial run should not be capped by the stale one. 

2229 

2230 **This one reader is held to an exact head, and its two siblings are not** (#1069). 

2231 Alone among the three panel-shaped readers here, this count can *remove* a 

2232 requirement: below ``jury.min_vendors`` it downgrades a gating jury to advisory 

2233 (:func:`keel.ship.resolve_jury`), which drops ``jury-verdict`` from the required 

2234 evidence entirely. So it asks :func:`keel.juryavail.is_pinnable_head` — the same 

2235 blank-head predicate the panel pins ask — before it reads a comment at all, and a 

2236 run that resolved no head declares nothing rather than inheriting the last verdict 

2237 on the pull request. Without the guard, `keel evidence-verify` run offline with no 

2238 ``--head-sha`` (its documented default) read a ``vendors: 1`` verdict posted against 

2239 an earlier head as this head's and relaxed the gate: a requirement removed by 

2240 evidence nobody re-checked. A *mismatched* head was already refused, by 

2241 :func:`_matches_head` — that predicate is exact once a head is known, and permissive 

2242 only when none is — so the guard closes the blank-head half and nothing else. 

2243 

2244 :func:`jury_panel_size` and :func:`panel_verdict_posted` deliberately keep the 

2245 permissive reading, because neither can relax anything: the first feeds 

2246 ``max(declared, minimum_vendors)`` and can only raise the bar, and the second is 

2247 already refused on a blank head by its caller, which owns the pin order. 

2248 """ 

2249 if not juryavail.is_pinnable_head(head_sha): 

2250 return None 

2251 counts = [ 

2252 parsed 

2253 for item in [*(pr_comments or []), *(pr_reviews or [])] 

2254 if _is_jury_verdict(item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced) 

2255 if (parsed := _parse_vendor_count(_fields(_body(item)).get("vendors"))) is not None 

2256 ] 

2257 return max(counts) if counts else None 

2258 

2259 

2260def jury_panel_size( 

2261 pr_comments: list[dict[str, Any]] | None = None, 

2262 pr_reviews: list[dict[str, Any]] | None = None, 

2263 *, 

2264 head_sha: str | None = None, 

2265 covered_heads: Collection[str] = (), 

2266 enforced: bool = True, 

2267) -> int | None: 

2268 """Return the panel size declared by a posted jury verdict, or ``None`` (#1015). 

2269 

2270 Reads the ``panelists: <N>`` field off the same comment 

2271 :func:`jury_participating_vendors` reads ``vendors:`` from, and for the same 

2272 reason: when the jury **is** the review panel, the number of ballots is the 

2273 reviewer count the evidence gate must require, and a hosted runner can read 

2274 it from nowhere else. 

2275 

2276 ``None`` means "not declared", which leaves the gate on the contract's floor 

2277 rather than requiring nothing. The largest declared count wins, so a re-post 

2278 that completes a partial panel raises the requirement instead of being capped 

2279 by the stale verdict — the direction that fails closed. 

2280 

2281 **It does not share that sibling's head rule, and the asymmetry is deliberate** 

2282 (#1069). This count reaches :func:`keel.ship._jury_panel_size`, which answers 

2283 ``max(declared, minimum_vendors)`` — so a stale or blank-head ``panelists:`` can 

2284 only ever *raise* what the tier owes, never remove a requirement. Holding it to an 

2285 exact head would refuse a bar-raising reading inside a gate whose other half, with 

2286 no head resolved, is already unfiltered (:func:`_matches_head`). ``vendors:`` is the 

2287 one that can relax, so ``vendors:`` is the one that is pinned. 

2288 """ 

2289 counts = [ 

2290 parsed 

2291 for item in [*(pr_comments or []), *(pr_reviews or [])] 

2292 if _is_jury_verdict(item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced) 

2293 if (parsed := _parse_vendor_count(_fields(_body(item)).get("panelists"))) is not None 

2294 ] 

2295 return max(counts) if counts else None 

2296 

2297 

2298def panel_verdict_posted( 

2299 pr_comments: list[dict[str, Any]] | None = None, 

2300 pr_reviews: list[dict[str, Any]] | None = None, 

2301 *, 

2302 head_sha: str | None = None, 

2303 covered_heads: Collection[str] = (), 

2304 enforced: bool = True, 

2305) -> bool: 

2306 """Is a head-pinned jury verdict already on this pull request? (#1066) 

2307 

2308 Proof that the panel *sat*, from the one place a bare CI runner can read it: the run 

2309 ledger and the jury artifact both live under the gitignored ``.keel/state/``, while PR 

2310 comments are always visible. A verification surface uses it to pin the contract to what 

2311 the ship measured rather than re-measuring the panel on its own machine. It is the 

2312 *weaker* of the two pins and speaks only when the run left no ledger record for this 

2313 head; :func:`keel.juryavail.pin` owns that order and is the one place it is written. 

2314 

2315 Distinct from :func:`jury_panel_size`, which answers *how many* ballots and is ``None`` 

2316 for a verdict predating the ``panelists:`` field. Presence is the weaker question, and 

2317 the one that must not depend on an optional field. 

2318 

2319 **Call this only with a head you actually resolved.** Like every reader here it goes 

2320 through :func:`_matches_head`, which reads a blank ``head_sha`` as "do not filter" — 

2321 right for counting evidence, wrong for a pin, which is why the caller refuses a blank 

2322 head first (:func:`keel.juryavail.is_pinnable_head`). 

2323 

2324 :func:`jury_participating_vendors` asks that same predicate *itself* rather than 

2325 leaving it to a caller (#1069), and the difference is about who the callers are: this 

2326 one is read from exactly one place, :func:`keel.juryavail.pin`, which owns the pin 

2327 order and so is the right place for the rule; the vendor count is read straight off 

2328 ``keel evidence-verify``'s argv-derived head, where there is no single owner to put it 

2329 in front of. 

2330 """ 

2331 return any( 

2332 _is_jury_verdict(item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced) 

2333 for item in [*(pr_comments or []), *(pr_reviews or [])] 

2334 ) 

2335 

2336 

2337def shipped_panel_decision( 

2338 pr_comments: list[dict[str, Any]] | None = None, 

2339 *, 

2340 head_sha: str | None = None, 

2341 enforced: bool = True, 

2342) -> str | None: 

2343 """The panel decision **this run** recorded, read back off its closure comment (#1068). 

2344 

2345 The middle pin, and the one that makes the strongest pin work anywhere. The run's own 

2346 ``ship_run`` ledger record outranks a posted jury verdict — a comment records what 

2347 somebody put on the pull request, the ledger records what the run *did* — but the 

2348 ledger lives under the gitignored ``.keel/state/``, so on a hosted ``evidence-verify`` 

2349 or ``merge`` there is no record to read and that precedence held on the shipping 

2350 workstation and nowhere else. A leftover or collaborator-posted ``keel.jury-verdict.v1`` 

2351 then answered for a run that had fallen back, and took ``review-verdict-1..3`` off the 

2352 required set. 

2353 

2354 The run's decision is already on the pull request: s11 posts the closure comment keel 

2355 renders from that same ledger record, and since #1068 round 6 it carries 

2356 :data:`keel.closure.JURY_PANEL_MARKER` beside the human ``Jury panel:`` line. So this 

2357 reads the run's own statement from the one place that travels with the pull request. 

2358 

2359 Three conditions, and each is the same rule its siblings hold to: 

2360 

2361 * **Trusted author only** (:func:`_is_trusted_source`). keel posts the closure comment 

2362 on the operator's behalf, which is exactly the authority a posted jury verdict has — 

2363 no more. An untrusted author must not be able to relax the contract *in either 

2364 direction*: neither to claim a fallback that drops the panel item, nor to claim the 

2365 panel sat. 

2366 * **An actual closure comment** (:func:`_has_closure_marker`), so the marker counts only 

2367 inside the artifact that renders it — a reviewer quoting the marker while describing 

2368 this change is prose, the #1026 rule every marker here is read under. 

2369 * **Pinned to this head.** The marker names the head its record was written for and it 

2370 must be the head under verification, because a pull request outlives its heads and a 

2371 pin removes requirements. 

2372 

2373 **The latest such comment is the answer, not the first** (#1068 round 7). One head can 

2374 be shipped more than once — a re-run, a force-push back onto the same commit, a second 

2375 ship on a different machine — and each ship posts its own closure comment. ``pr_comments`` 

2376 arrives in GitHub's order, oldest first, so this walks the whole list and keeps the last 

2377 match: the newest statement wins, which is the same direction 

2378 :func:`keel.ledger.latest_ship_run_for_pr` selects the ledger record in, and 

2379 :func:`keel.juryavail.pin` ranks the two sources on the premise that they agree about it. 

2380 Returning the first match meant an older ``decision=fallback`` outranked the panel-sat 

2381 ship that followed it — and round 6 emitted no marker at all for a panel that sat, so the 

2382 later run had nothing to outrank the older one *with*. :func:`keel.closure._jury_panel` 

2383 now renders ``decision=available`` too, which is what makes last-wins well defined here. 

2384 

2385 ``None`` for everything else — no comment, an older head, a marker keel did not write — 

2386 and ``None`` means *this source is silent*, never a waiver: :func:`keel.juryavail.pin` 

2387 then goes on to the posted verdict exactly as it did before. 

2388 """ 

2389 if not head_sha: 

2390 return None 

2391 latest: str | None = None 

2392 for item in pr_comments or []: 

2393 if not _is_trusted_source(item, enforced=enforced): 

2394 continue 

2395 body = _body(item) 

2396 if not _has_closure_marker(body): 

2397 continue 

2398 decision = _jury_panel_decision(body, head_sha) 

2399 if decision is not None: 

2400 latest = decision 

2401 return latest 

2402 

2403 

2404def _jury_panel_decision(body: str, head_sha: str) -> str | None: 

2405 """The decision ``body``'s panel marker records for ``head_sha``, or ``None``. 

2406 

2407 A token parser over one HTML-comment line, not a regex over Markdown: the line is 

2408 :func:`keel.closure._jury_panel_marker`'s exact render, so it unwraps with the same 

2409 :func:`_unwrap_html_comment` a header marker does and splits into 

2410 ``<marker> head=<sha> decision=<value>``. A line that is not that shape is prose and is 

2411 skipped, which is why the human sentence above it — which *names* neither field — can 

2412 never be mistaken for the record. 

2413 """ 

2414 for raw_line in (body or "").splitlines(): 

2415 tokens = _unwrap_html_comment(raw_line.strip()).split() 

2416 if not tokens or tokens[0] != closure.JURY_PANEL_MARKER: 

2417 continue 

2418 fields: dict[str, str] = {} 

2419 for token in tokens[1:]: 

2420 key, _, value = token.partition("=") 

2421 # First wins, the convention `_fields` already reads headers under, and a 

2422 # token carrying no `=` becomes a valueless key that matches neither field. 

2423 fields.setdefault(key, value) 

2424 if fields.get("head") == head_sha: 

2425 return fields.get("decision") 

2426 return None 

2427 

2428 

2429def _parse_vendor_count(raw: str | None) -> int | None: 

2430 """Parse a declared vendor count, rejecting anything not a plain non-negative int.""" 

2431 if raw is None: 

2432 return None 

2433 try: 

2434 parsed = int(raw) 

2435 except ValueError: 

2436 return None 

2437 return parsed if parsed >= 0 else None 

2438 

2439 

2440def _is_jury_verdict( 

2441 item: dict[str, Any], 

2442 *, 

2443 head_sha: str | None = None, 

2444 covered_heads: Collection[str] = (), 

2445 enforced: bool = True, 

2446) -> bool: 

2447 """Whether ``item`` is a trusted jury verdict pinned to ``head_sha`` — presence only. 

2448 

2449 **Deliberately blind to the consensus** (#1429). Its readers here are the panel-shape 

2450 checks — :func:`jury_participating_vendors`, :func:`jury_panel_size` and 

2451 :func:`panel_verdict_posted` — which size the panel and pin whether it sat. A panel 

2452 that sat and rejected the change still sat, with that many vendors and ballots, so a 

2453 rejecting verdict must keep answering them: reading the consensus there would turn a 

2454 rejection into "no panel", and drop the ballots it owes. Whether the panel *approves* 

2455 is a separate question, answered by :func:`_standing_jury_verdict` and 

2456 :func:`jury_verdict_approves` for the ``jury-verdict`` requirement. 

2457 """ 

2458 if not _is_trusted_source(item, enforced=enforced): 

2459 return False 

2460 body = _body(item) 

2461 if _is_ship_assessment(body): 

2462 return False 

2463 return marker_in_header(body) == JURY_VERDICT_MARKER and _matches_head( 

2464 item, body, head_sha, covered_heads 

2465 ) 

2466 

2467 

2468def _standing_jury_verdict( 

2469 items: list[dict[str, Any]], 

2470 *, 

2471 head_sha: str | None = None, 

2472 covered_heads: Collection[str] = (), 

2473 enforced: bool = True, 

2474) -> dict[str, Any] | None: 

2475 """The latest trusted jury verdict answering for ``head_sha``, or ``None`` (#1429). 

2476 

2477 **The latest jury verdict is the panel's word**, ordered as :func:`_review_tally` 

2478 orders review verdicts: by when it was posted (:func:`_posted_at` — ``created_at``, 

2479 never ``updated_at``), then by position. A panel re-run on the same head that now 

2480 approves supersedes the rejection before it, and one that now rejects supersedes the 

2481 approval. A jury verdict pinned to an older head is not read: the head that moved 

2482 answers it. 

2483 """ 

2484 standing: dict[str, Any] | None = None 

2485 for _, item in sorted(enumerate(items), key=lambda pair: (_posted_at(pair[1]), pair[0])): 

2486 if _is_jury_verdict( 

2487 item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced 

2488 ): 

2489 standing = item 

2490 return standing 

2491 

2492 

2493def standing_jury_verdict( 

2494 pr_comments: list[dict[str, Any]] | None, 

2495 *, 

2496 head_sha: str | None, 

2497 covered_heads: Collection[str] = (), 

2498 enforced: bool = True, 

2499) -> dict[str, Any] | None: 

2500 """The jury verdict comment that stands for ``head_sha``, or ``None`` (#1437). 

2501 

2502 The merge gate's own reading (:func:`_standing_jury_verdict`): trusted author, the 

2503 jury marker in the header, pinned to the head or a head it covers, latest by when it 

2504 was posted. ``keel ship`` reuses the panel this names instead of convening another, 

2505 so the two can never pick different panels. A blank head is refused here, unlike 

2506 the evidence counters: reuse *replaces* a run, and a panel nobody pinned to a head 

2507 must not stand in for one. 

2508 """ 

2509 if not head_sha: 

2510 return None 

2511 return _standing_jury_verdict( 

2512 pr_comments or [], head_sha=head_sha, covered_heads=covered_heads, enforced=enforced 

2513 ) 

2514 

2515 

2516def verdict_head(item: dict[str, Any]) -> str: 

2517 """The head a verdict comment answers for: its ``head:`` field, else its commit.""" 

2518 return _verdict_head(item, _body(item)) 

2519 

2520 

2521def _jury_not_approved_finding( 

2522 items: list[dict[str, Any]], 

2523 *, 

2524 head_sha: str | None, 

2525 covered_heads: Collection[str] = (), 

2526 enforced: bool, 

2527 blocking: bool, 

2528) -> dict[str, Any] | None: 

2529 """The finding for a standing jury verdict whose consensus does not approve (#1429). 

2530 

2531 ``major`` when ``blocking`` — the ``jury-verdict`` item is required and not deferred — 

2532 so a rejecting panel **holds** the merge with its consensus and head named 

2533 (``jury-verdict-not-approved: the jury's consensus at <head> is REQUEST_CHANGES, not an 

2534 approval.`` in :func:`refusal_reason`). Merely leaving it uncounted would read as 

2535 *missing* and send the operator looking for a comment that is on the pull request. 

2536 ``minor`` otherwise: an advisory panel's rejection is said, never gated on. 

2537 """ 

2538 standing = _standing_jury_verdict( 

2539 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced 

2540 ) 

2541 if standing is None: 

2542 return None 

2543 body = _body(standing) 

2544 token = jury_verdict_token(body) 

2545 if token in APPROVING_VERDICTS: 

2546 return None 

2547 head = _verdict_head(standing, body) 

2548 message = ( 

2549 f"the jury's consensus at {head} is {token}, not an approval." 

2550 if token 

2551 else f"the jury verdict at {head} has no readable AI Jury verdict line." 

2552 ) 

2553 return { 

2554 "id": JURY_NOT_APPROVED_FINDING, 

2555 "severity": "major" if blocking else "minor", 

2556 "kind": "jury", 

2557 "message": message, 

2558 }