Coverage for src/keel/evidence.py: 100%
689 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""Deterministic pre-merge evidence verification.
3The ship adapter is agentic, but the artifacts it must leave behind are not:
4reviewer verdict comments/reviews, the jury verdict a gating jury requires, and the stable
5closure comment marker. This module keeps the check pure so CI can enforce it
6without trusting prose in an agent prompt.
8Classification is **header-anchored**: :func:`marker_in_header` decides what a
9comment is from its first non-empty line and nothing else, so a marker a reviewer
10quotes in prose is content rather than a classification signal (#1026). The ship
11assessment heading is anchored the same way, from its own header line (#1035).
12"""
14from __future__ import annotations
16import hashlib
17import re
18from collections.abc import Collection, Sequence
19from dataclasses import dataclass
20from typing import Any
22from . import agents, closure, juryavail
23from . import team as team_policy
25SCHEMA_VERSION = "keel.evidence.v1"
26#: ``reviewers.panel`` when the cross-vendor jury *is* the review for this tier
27#: (``knobs.team``'s ``review.by_tier.<n>: jury``), and the minimum distinct
28#: vendors such a panel must span. Both come from :mod:`keel.team`, the leaf
29#: module that owns the team vocabulary, so the gate and the policy cannot drift.
30JURY_PANEL = team_policy.JURY_PANEL
31DEFAULT_MINIMUM_JURY_VENDORS = team_policy.DEFAULT_MIN_VENDORS
32AGENT_LABEL_PREFIX = "agent:"
33MODEL_LABEL_PREFIX = "model:"
34REVIEW_VERDICT_MARKER = "keel.review-verdict.v1"
35#: The ``Verdict:`` tokens that **approve** a change (#1426). A review verdict counts
36#: toward ``review-verdict-N`` only when the first word of its ``Verdict:`` line is one
37#: of these, read case-insensitively by :func:`review_verdict_token`. The set is what
38#: keel and its hosts actually write, not a guess:
39#:
40#: * ``LGTM`` — the default of :func:`keel.artifacts.render_review_verdict`, what
41#: ``contracts.py``'s ``review_verdict_template`` renders for an unblocked change, and
42#: what :func:`keel.jury.map_verdict` folds ai-jury's ``APPROVE`` / ``READY`` into
43#: before ``keel review --from-jury`` posts a ballot;
44#: * ``APPROVE`` — what the host reviewers post today: every verdict on the keel PRs
45#: from #1405 onward, 631 of the 1,355 verdict comments on the repository;
46#: * ``PASS`` — 168 verdicts posted on #918–#992, written by hosts as ``Verdict: pass``.
47#:
48#: Everything else holds the merge: ``REQUEST_CHANGES`` (keel's own blocked-change
49#: template, and the jury's ``NEEDS_INFO``), ``COMMENT``, ``ABSTAIN``, a token keel does
50#: not know, and a verdict with no ``Verdict:`` line at all. Inventing an approval for a
51#: stance keel does not recognise is the mapping error that cannot be undone — the same
52#: rule :data:`keel.jury._VERDICT` states for ballots.
53APPROVING_VERDICTS = frozenset({"APPROVE", "LGTM", "PASS"})
54#: The non-approving tokens a refusal names as *requesting changes*; any other
55#: non-approving token is named by itself (``alice does not approve at <head> (verdict
56#: COMMENT)``).
57REQUEST_CHANGES_VERDICTS = frozenset({"REQUEST_CHANGES", "CHANGES_REQUESTED", "NEEDS_INFO"})
58#: The finding a non-approving review verdict at the current head raises (#1426).
59VERDICT_NOT_APPROVED_FINDING = "review-verdict-not-approved"
60#: The finding a non-approving jury verdict at the current head raises (#1429).
61JURY_NOT_APPROVED_FINDING = "jury-verdict-not-approved"
62JURY_VERDICT_MARKER = "keel.jury-verdict.v1"
63#: The comment a live ship run posts on its PR right after creating it (#1013). It is
64#: the *primary* arming signal for the evidence gate: unlike the branch name it is
65#: written by the run itself, so a run that named its branch something else — or whose
66#: ledger lives in a per-run worktree CI cannot read — still identifies as a keel ship
67#: run. See :func:`gate_decision` for the full arming order.
68SHIP_PROVENANCE_MARKER = "keel.ship-provenance.v1"
69#: The marker the ship adapter tells a host to post when a finding is deferred rather
70#: than fixed. keel core never counts a deferral as evidence; it is listed among the
71#: classification markers so a deferral comment reads as *one* artifact instead of
72#: being classified by whichever marker its prose happens to name.
73DEFERRAL_MARKER = "keel.deferral.v1"
74SHIP_ASSESSMENT_HEADING = "### \U0001f6a2 keel ship"
75#: The banner the ``keel ship`` CLI prints above its own summary. An assessment pasted
76#: raw — without the workflow's Markdown heading — leads with this line instead, so both
77#: forms identify the comment. See :func:`_is_ship_assessment`.
78SHIP_ASSESSMENT_BANNER = "keel ship \u2014"
79DEFAULT_WAIVER_LABEL = "keel:evidence-waived"
80TRUSTED_AUTHOR_ASSOCIATIONS = frozenset({"OWNER", "MEMBER", "COLLABORATOR"})
81TRUSTED_SHIP_ASSESSMENT_BOTS = frozenset({"github-actions", "github-actions[bot]"})
83#: Every marker that *classifies* an evidence comment. Order is the order findings
84#: report them in, so a malformed-header message is byte-stable.
85CLASSIFICATION_MARKERS: tuple[str, ...] = (
86 REVIEW_VERDICT_MARKER,
87 JURY_VERDICT_MARKER,
88 SHIP_PROVENANCE_MARKER,
89 closure.CLOSURE_SCHEMA_VERSION,
90 DEFERRAL_MARKER,
91)
93#: Membership view of :data:`CLASSIFICATION_MARKERS`, which is a tuple because its **order**
94#: is the order markers are rendered in. Asking a tuple `x in ...` is a linear scan, and the
95#: two places below only ask whether every token is a known marker — a set answers that
96#: question by its type rather than by walking. The tuple stays where order matters.
97_CLASSIFICATION_MARKERS_SET = frozenset(CLASSIFICATION_MARKERS)
99#: The finding raised for a comment whose header names more than one marker.
100MALFORMED_MARKER_FINDING = "malformed-evidence-comment"
102#: The header fields keel reads off an evidence comment. Closed by design: an
103#: unlisted ``key: value`` line is prose, and prose ends the header block (#932).
104#: ``panelists`` joins ``vendors`` as a jury-verdict field, because the size of a
105#: panel that *is* the review sets the required verdict count (#1015).
106_FIELD_RE = re.compile(
107 r"^\s*(?P<key>reviewer|head|vendor|model|vendors|panelists)\s*:\s*(?P<value>\S+)\s*$",
108 re.IGNORECASE,
109)
110_HEADER_LINE_RE = re.compile(r"^[A-Za-z0-9_-]+\s*:")
111#: A ``Verdict:`` line, as :func:`keel.artifacts.render_review_verdict` writes it. Anchored
112#: at the line start, so ``AI Jury verdict:`` and a quoted ``> Verdict:`` are not one.
113_VERDICT_LINE_RE = re.compile(r"^\s*verdict\s*:(?P<value>.*)$", re.IGNORECASE)
114#: The first word of a verdict line's value: wrapper punctuation (``**``, a backtick, an
115#: emoji) is skipped, and the word ends at the first character that is not a letter,
116#: ``_`` or ``-`` — so ``APPROVE — minor nits`` and ``APPROVE, minor nits`` both read
117#: ``APPROVE``.
118_VERDICT_TOKEN_RE = re.compile(r"^[\W_]*(?P<token>[A-Za-z][A-Za-z_-]*)")
119#: A jury verdict's consensus line, as :func:`keel.artifacts.render_jury_verdict` writes it:
120#: ``AI Jury verdict: REQUEST_CHANGES.`` — the trailing full stop ends the token like any
121#: other punctuation does (#1429).
122_JURY_VERDICT_LINE_RE = re.compile(r"^\s*AI\s+Jury\s+verdict\s*:(?P<value>.*)$", re.IGNORECASE)
123_SHIP_BRANCH_RE = re.compile(r"^(feature|fix|chore|docs|test)/issue-\d+(?:-|$)")
124#: The exact wrapper a marker line may wear. Every keel renderer emits its marker as
125#: the whole first line, in one of exactly two shapes: bare
126#: (``artifacts.render_review_verdict`` / ``render_jury_verdict`` /
127#: ``render_ship_provenance``) or wrapped in an HTML comment so it renders invisibly
128#: (``closure.render_closure_comment``).
129#:
130#: These are matched literally, never with a regex. A pattern that treats ``-->`` as
131#: *the* comment terminator is wrong about HTML — a browser also ends a comment at
132#: ``--!>`` — and a classifier that disagrees with the renderer about where a comment
133#: ends is exactly the confusion this module exists to remove (CodeQL
134#: ``py/bad-tag-filter``). keel does not need to parse HTML: it needs to recognise the
135#: one shape it writes, and refuse everything else.
136_HTML_COMMENT_OPEN = "<!--"
137_HTML_COMMENT_CLOSE = "-->"
139STATUS_PASS = "pass"
140#: The finding severity that fails verification on its own, with nothing missing.
141BLOCKING_FINDING_SEVERITY = "major"
142STATUS_WAITING = "waiting"
143STATUS_FAIL = "fail"
144STATUSES = (STATUS_PASS, STATUS_WAITING, STATUS_FAIL)
146# Evidence phases. An artifact is required in the phase that produces it, mirroring
147# the step mapping stepverifier already applies (review -> s7, jury -> s8, closure ->
148# s12). The merge gate at s10 asks for PHASE_PRE_MERGE, because the closure comment
149# is a post-merge record and requiring it at s10 makes the backbone unsatisfiable.
150PHASE_PRE_MERGE = "pre-merge"
151PHASE_POST_MERGE = "post-merge"
152PHASE_ALL = "all"
153PHASES = (PHASE_PRE_MERGE, PHASE_POST_MERGE, PHASE_ALL)
156@dataclass(frozen=True)
157class EvidenceItem:
158 id: str
159 kind: str
160 required: bool
161 description: str
162 phase: str = PHASE_PRE_MERGE
164 def as_dict(self) -> dict[str, Any]:
165 return {
166 "id": self.id,
167 "kind": self.kind,
168 "required": self.required,
169 "description": self.description,
170 "phase": self.phase,
171 }
174def gate_active(labels: Sequence[str] | None, gate_label: str) -> bool:
175 """Return whether ``gate_label`` is present in ``labels`` (None/empty -> False).
177 An empty ``gate_label`` is never active, so a misconfigured (blank) label can
178 never silently match — the schema also forbids an empty ``evidence_gate_label``.
179 """
180 if not gate_label:
181 return False
182 return gate_label in set(labels or ())
185def gate_decision(
186 labels: Sequence[str] | None,
187 gate_label: str,
188 *,
189 waiver_label: str = DEFAULT_WAIVER_LABEL,
190 head_ref: str | None = None,
191 pr_comments: list[dict[str, Any]] | None = None,
192 pr_reviews: list[dict[str, Any]] | None = None,
193 ledger_records: Sequence[object] | None = None,
194) -> dict[str, Any]:
195 """Return the fail-closed evidence-gate arming decision.
197 Ship provenance arms the gate by default. The only disarm path is an explicit
198 waiver label applied by an operator; the legacy gate label remains an
199 additional arming signal for already-installed workflows.
201 The signals are consulted in this order, and the order is part of the contract
202 (documented in ``docs/keel/evidence.md``):
204 1. ``operator-waiver-label`` — the one sanctioned disarm, checked first so an
205 explicit waiver is never shadowed by an arming signal.
206 2. ``gate-label`` — the legacy opt-in label.
207 3. ``ship-provenance-comment`` — a trusted PR comment carrying
208 :data:`SHIP_PROVENANCE_MARKER`, which a live ship run posts as soon as the PR
209 exists. **Ahead of the branch regex on purpose** (#1013): the marker is written
210 by the run, the branch name is written by whoever typed it, and a ship run that
211 named its branch ``fix/2467-slug`` used to read as a non-keel PR.
212 4. ``ship-branch`` — the legacy branch-name fallback for runs that predate the
213 marker. Kept, but it is no longer the signal keel relies on.
214 5. ``ship-assessment-comment`` / ``review-verdict-marker`` / ``ship-run-ledger`` —
215 the remaining after-the-fact traces, unchanged.
217 Nothing was removed: every path that armed the gate before still arms it.
218 """
219 label_set = set(labels or ())
220 if waiver_label and waiver_label in label_set:
221 return _gate_decision(False, "operator-waiver-label", waiver_label, waived=True)
222 if gate_active(labels, gate_label):
223 return _gate_decision(True, "gate-label", gate_label)
224 if _has_trusted_ship_provenance(pr_comments or []):
225 return _gate_decision(True, "ship-provenance-comment", SHIP_PROVENANCE_MARKER)
226 if head_ref and _SHIP_BRANCH_RE.search(head_ref):
227 return _gate_decision(True, "ship-branch", head_ref)
228 if _has_trusted_ship_assessment(pr_comments or []):
229 return _gate_decision(True, "ship-assessment-comment", SHIP_ASSESSMENT_HEADING)
230 if _has_trusted_review_marker([*(pr_comments or []), *(pr_reviews or [])]):
231 return _gate_decision(True, "review-verdict-marker", REVIEW_VERDICT_MARKER)
232 if ledger_records:
233 return _gate_decision(True, "ship-run-ledger", "ship_run")
234 return _gate_decision(False, "no-ship-provenance", None)
237def _gate_decision(
238 enforced: bool,
239 reason: str,
240 source: str | None,
241 *,
242 waived: bool = False,
243) -> dict[str, Any]:
244 return {
245 "schema_version": SCHEMA_VERSION,
246 "enforced": enforced,
247 "waived": waived,
248 "reason": reason,
249 "source": source,
250 }
253def _has_trusted_ship_provenance(items: list[dict[str, Any]]) -> bool:
254 """True when a trusted PR comment carries the ship-provenance marker.
256 Trust is the same fail-closed ``author_association`` check every other evidence
257 source uses: an anonymous drive-by comment must not be able to arm — or, more to
258 the point, to *look* like it armed — the gate.
259 """
260 return any(
261 _is_trusted_source(item, enforced=True)
262 and marker_in_header(_body(item)) == SHIP_PROVENANCE_MARKER
263 for item in items
264 )
267def _has_trusted_ship_assessment(items: list[dict[str, Any]]) -> bool:
268 return any(
269 _is_ship_assessment_source(item) and _is_ship_assessment(_body(item)) for item in items
270 )
273def _is_ship_assessment_source(item: dict[str, Any]) -> bool:
274 if _is_trusted_source(item, enforced=True):
275 return True
276 user = item.get("user") if isinstance(item.get("user"), dict) else {}
277 login = user.get("login") if isinstance(user.get("login"), str) else None
278 return bool(login and login.lower() in TRUSTED_SHIP_ASSESSMENT_BOTS)
281def contract_as_dict(
282 review_contract: dict[str, Any],
283 *,
284 dry_run: bool = False,
285 enforced: bool = True,
286 deferrals: tuple[str, ...] = (),
287 phase: str = PHASE_ALL,
288) -> dict[str, Any]:
289 """Return the required evidence set derived from review/jury flags."""
290 return {
291 "schema_version": SCHEMA_VERSION,
292 "enforced": enforced,
293 "phase": phase,
294 "source": "review_merge_contract + closure_comment",
295 "dry_run_disables_gating": True,
296 "fail_closed": True,
297 "require_distinct_vendors": _require_distinct_vendors(review_contract),
298 "accepted_sources": {
299 "closure": ("trusted issue/PR comments carrying keel.closure-comment.v1"),
300 "review": (
301 "trusted PR review/comment carrying keel.review-verdict.v1 and current head, "
302 "whose reviewer's latest Verdict line approves"
303 ),
304 "jury": "trusted PR comment carrying keel.jury-verdict.v1 and current head",
305 },
306 "not_accepted": [
307 "pull_request_body",
308 "chat_summary",
309 "untrusted_public_comment",
310 "keel_ship_assessment_comment",
311 ],
312 "deferrals": list(deferrals),
313 "required": [
314 item.as_dict()
315 for item in required_items(review_contract, dry_run=False, enforced=enforced)
316 ],
317 "active_required": [
318 item.as_dict()
319 for item in required_items(
320 review_contract, dry_run=dry_run, enforced=enforced, phase=phase
321 )
322 ],
323 }
326def required_items(
327 review_contract: dict[str, Any],
328 *,
329 dry_run: bool = False,
330 enforced: bool = True,
331 phase: str = PHASE_ALL,
332) -> tuple[EvidenceItem, ...]:
333 """Return the tier/flag-derived evidence requirements for ``phase``.
335 ``phase`` selects which artifacts are in scope: ``pre-merge`` covers the
336 review verdicts and a gating jury verdict (everything that must exist before
337 s10 authorizes a merge), ``post-merge`` covers the closure comments s11
338 posts, and ``all`` — the default, so existing callers are unchanged — covers
339 both. An unknown phase raises, so a typo cannot silently drop requirements.
340 """
341 if phase not in PHASES:
342 raise ValueError(f"unknown evidence phase {phase!r}; expected one of {', '.join(PHASES)}")
343 if dry_run or not enforced:
344 return ()
345 reviewers = review_contract.get("reviewers")
346 reviewer_count = reviewers.get("count") if isinstance(reviewers, dict) else 0
347 reviewer_count = reviewer_count if isinstance(reviewer_count, int) and reviewer_count > 0 else 0
348 jury = review_contract.get("jury")
349 jury_required = (
350 isinstance(jury, dict) and bool(jury.get("enabled")) and jury.get("mode") == "gating"
351 )
352 items: list[EvidenceItem] = [
353 EvidenceItem(
354 "closure-comment-pr",
355 "closure",
356 True,
357 "PR conversation comment with keel.closure-comment.v1 marker",
358 PHASE_POST_MERGE,
359 ),
360 EvidenceItem(
361 "closure-comment-issue",
362 "closure",
363 True,
364 "Linked issue comment with keel.closure-comment.v1 marker",
365 PHASE_POST_MERGE,
366 ),
367 ]
368 # A jury panel's verdicts are panelist ballots mapped onto s7 evidence by
369 # `keel review --from-jury` (#1015): same marker, same head binding, same
370 # requirement. The description names the panel that produced them so a
371 # missing one sends the operator to the jury run rather than to a host
372 # reviewer that was never dispatched.
373 review_description = (
374 "Distinct posted ai-jury panelist verdict for the current PR"
375 if review_panel(review_contract) == JURY_PANEL
376 else "Distinct posted s7 reviewer verdict for the current PR"
377 )
378 for index in range(1, reviewer_count + 1):
379 items.append(
380 EvidenceItem(
381 f"review-verdict-{index}",
382 "review",
383 True,
384 review_description,
385 PHASE_PRE_MERGE,
386 )
387 )
388 if jury_required:
389 items.append(
390 EvidenceItem(
391 "jury-verdict",
392 "jury",
393 True,
394 "Posted gating jury verdict comment for the current PR",
395 PHASE_PRE_MERGE,
396 )
397 )
398 if phase == PHASE_ALL:
399 return tuple(items)
400 return tuple(item for item in items if item.phase == phase)
403def verify(
404 review_contract: dict[str, Any],
405 *,
406 pr_comments: list[dict[str, Any]] | None = None,
407 issue_comments: list[dict[str, Any]] | None = None,
408 pr_reviews: list[dict[str, Any]] | None = None,
409 pr_body: str | None = None,
410 pr_title: str = "",
411 pr_labels: Sequence[str] | None = None,
412 head_sha: str | None = None,
413 covered_heads: Collection[str] = (),
414 ledger_record: dict[str, Any] | None = None,
415 dry_run: bool = False,
416 enforced: bool = True,
417 deferrals: tuple[str, ...] = (),
418 phase: str = PHASE_ALL,
419 require_armed: bool = False,
420 waived: bool = False,
421) -> dict[str, Any]:
422 """Verify required evidence artifacts and return a deterministic report.
424 ``phase`` narrows the requirement set to the artifacts that phase produces;
425 see :func:`required_items`. The s10 merge gate passes ``pre-merge`` so it
426 does not demand the closure comments s11 has not written yet.
428 ``require_armed`` closes the vacuous-pass hole: with the gate unarmed there
429 are no requirements, so the report would otherwise pass without having
430 checked anything, and a green result could not be told apart from "could not
431 tell whether this was a ship run". When set, an unarmed gate is a blocking
432 finding instead. A deliberately non-ship PR still goes green through the
433 operator waiver label, which disarms explicitly rather than by accident.
435 When ``ledger_record`` is the ship_run record for this PR, a closure comment
436 only counts when its content matches the canonical render of that record
437 (closure-comment fidelity). Without a record the marker-only behavior holds.
439 Every comment is classified by :func:`marker_in_header` — the marker on its
440 header line, never one quoted in its prose. A header naming two markers is
441 malformed: it is excluded from every count and reported as an advisory
442 ``malformed-evidence-comment`` finding.
444 When the gate is active, ``pr_labels`` are additionally checked for the
445 mandatory ``agent:<vendor>`` attribution label (and cross-checked against the
446 ledger implementer vendor when a record is present); see
447 :func:`attribution_check`. The labels are separately checked against keel's own
448 vocabulary — what :func:`keel.agents.attribution` produces from the ledger's
449 ``actors.implementer`` — by :func:`attribution_vocabulary_check`, which catches a
450 hand-composed label that happens to agree with a hand-written ledger value.
451 """
452 del pr_body # Explicitly not accepted as evidence.
453 items = required_items(review_contract, dry_run=dry_run, enforced=enforced, phase=phase)
454 deferred = set(deferrals)
455 counts = _evidence_counts(
456 pr_comments=pr_comments or [],
457 issue_comments=issue_comments or [],
458 pr_reviews=pr_reviews or [],
459 head_sha=head_sha,
460 covered_heads=covered_heads,
461 enforced=enforced,
462 ledger_record=ledger_record,
463 )
464 findings = _run_context_findings(
465 pr_comments=pr_comments or [],
466 issue_comments=issue_comments or [],
467 enforced=enforced,
468 ledger_record=ledger_record,
469 )
470 mismatch = _closure_mismatch_scopes(
471 pr_comments=pr_comments or [],
472 issue_comments=issue_comments or [],
473 enforced=enforced,
474 ledger_record=ledger_record,
475 )
476 # A jury whose standing consensus does not approve (#1429) leaves `jury-verdict`
477 # unsatisfied *and* says why, by name: unsatisfied alone would read as "missing" and
478 # hide a comment that is on the pull request. It blocks only where the jury verdict is
479 # a requirement — an advisory panel's rejection is reported, never gated on, which is
480 # what advisory means.
481 jury_required = any(
482 item.id == "jury-verdict"
483 and not (item.id in deferred or item.kind in deferred or "all" in deferred)
484 for item in items
485 )
486 jury_refusal = _jury_not_approved_finding(
487 pr_comments or [],
488 head_sha=head_sha,
489 covered_heads=covered_heads,
490 enforced=enforced,
491 blocking=jury_required,
492 )
493 results = []
494 for item in items:
495 present = _is_present(item, counts)
496 is_deferred = item.id in deferred or item.kind in deferred or "all" in deferred
497 ok = present or is_deferred
498 results.append(
499 {
500 "id": item.id,
501 "kind": item.kind,
502 "required": item.required,
503 "present": present,
504 "deferred": is_deferred,
505 "ok": ok,
506 "reason": None
507 if ok
508 else (
509 jury_refusal["message"]
510 if item.id == "jury-verdict" and jury_refusal is not None
511 else _result_reason(item, mismatch)
512 ),
513 }
514 )
515 missing = [result["id"] for result in results if not result["ok"]]
516 distinct = _distinct_vendor_finding(
517 review_contract,
518 items=items,
519 deferred=deferred,
520 pr_comments=pr_comments or [],
521 pr_reviews=pr_reviews or [],
522 head_sha=head_sha,
523 covered_heads=covered_heads,
524 enforced=enforced,
525 )
526 if distinct is not None:
527 findings = [*findings, distinct]
528 substance = _verdict_substance_findings(
529 [*(pr_comments or []), *(pr_reviews or [])],
530 head_sha=head_sha,
531 covered_heads=covered_heads,
532 enforced=enforced,
533 pr_title=pr_title,
534 )
535 findings = [*findings, *substance]
536 # A reviewer whose standing verdict does not approve holds the merge by name (#1426):
537 # the counts above leave them out, and this says why instead of letting another
538 # reviewer's approval make up the number.
539 findings = [
540 *findings,
541 *_verdict_not_approved_findings(
542 [*(pr_comments or []), *(pr_reviews or [])],
543 head_sha=head_sha,
544 covered_heads=covered_heads,
545 enforced=enforced,
546 ),
547 ]
548 if jury_refusal is not None:
549 findings = [*findings, jury_refusal]
550 findings = [
551 *findings,
552 *_malformed_marker_findings(
553 [*(pr_comments or []), *(issue_comments or []), *(pr_reviews or [])],
554 enforced=enforced,
555 ),
556 ]
557 attribution = _attribution_finding(
558 pr_labels=pr_labels,
559 enforced=enforced and not dry_run,
560 ledger_record=ledger_record,
561 )
562 if attribution is not None:
563 findings = [*findings, attribution]
564 vocabulary = _attribution_vocabulary_finding(
565 pr_labels=pr_labels,
566 enforced=enforced and not dry_run,
567 ledger_record=ledger_record,
568 )
569 if vocabulary is not None:
570 findings = [*findings, vocabulary]
571 unarmed = _unarmed_finding(
572 enforced=enforced,
573 dry_run=dry_run,
574 require_armed=require_armed,
575 waived=waived,
576 )
577 if unarmed is not None:
578 findings = [*findings, unarmed]
579 blocking_findings = [f for f in findings if f["severity"] == BLOCKING_FINDING_SEVERITY]
580 has_mismatch = bool(mismatch)
581 if not missing and not blocking_findings:
582 status = STATUS_PASS
583 elif blocking_findings or has_mismatch:
584 status = STATUS_FAIL
585 else:
586 status = STATUS_WAITING
587 return {
588 "schema_version": SCHEMA_VERSION,
589 "status": status,
590 "dry_run": dry_run,
591 "enforced": enforced,
592 "phase": phase,
593 "required_count": len(items),
594 "missing": missing,
595 "results": results,
596 "counts": counts,
597 "findings": findings,
598 }
601def refusal_reason(verification: dict[str, Any]) -> str:
602 """Why a verification that is not a pass refuses the merge, naming everything (#1420).
604 Every missing item, then every blocking finding's message. A finding can fail the
605 verification with nothing missing — ``attribution-label`` on a pull request whose
606 review verdicts are all there — and a reason built from ``missing`` alone then read
607 ``missing evidence: `` with nothing after it, which names no cause at all. Never empty:
608 a verification that names neither says its status.
609 """
610 missing = [str(item) for item in verification.get("missing") or ()]
611 blocking = [
612 f"{finding.get('id')}: {finding.get('message')}"
613 for finding in verification.get("findings") or ()
614 if isinstance(finding, dict) and finding.get("severity") == BLOCKING_FINDING_SEVERITY
615 ]
616 parts: list[str] = []
617 if missing:
618 parts.append(f"missing evidence: {', '.join(missing)}")
619 if blocking:
620 parts.append(f"blocking finding(s): {'; '.join(blocking)}")
621 return "; ".join(parts) or f"evidence verification is {verification.get('status')}"
624def _require_distinct_vendors(review_contract: dict[str, Any]) -> bool:
625 reviewers = review_contract.get("reviewers")
626 return bool(reviewers.get("require_distinct_vendors")) if isinstance(reviewers, dict) else False
629def review_panel(review_contract: dict[str, Any]) -> str:
630 """Who the reviewers are on this contract: ``reviewers`` or ``jury`` (#1015).
632 A missing or malformed ``reviewers`` block reads as the host bench, which is
633 the stricter of the two answers everywhere this is asked.
634 """
635 reviewers = review_contract.get("reviewers")
636 panel = reviewers.get("panel") if isinstance(reviewers, dict) else None
637 return panel if isinstance(panel, str) and panel else "reviewers"
640def _minimum_jury_vendors(review_contract: dict[str, Any]) -> int:
641 """``jury.minimum_vendors`` from the contract, with the schema floor as fallback."""
642 jury = review_contract.get("jury")
643 minimum = jury.get("minimum_vendors") if isinstance(jury, dict) else None
644 if isinstance(minimum, int) and not isinstance(minimum, bool) and minimum > 0:
645 return minimum
646 return DEFAULT_MINIMUM_JURY_VENDORS
649def _verdict_substance_findings(
650 items: list[dict[str, Any]],
651 *,
652 head_sha: str | None,
653 covered_heads: Collection[str] = (),
654 enforced: bool,
655 pr_title: str,
656) -> list[dict[str, Any]]:
657 """One finding per verdict refused for naming nothing concrete (#926).
659 Reported rather than dropped. A verdict silently not counted surfaces as
660 "missing required evidence: review-verdict-2", which sends the operator
661 looking for a comment that is sitting right there — the reason has to say
662 the verdict was read and found to be a receipt.
664 ``minor`` when the gate is not enforced, mirroring the other content
665 findings: an advisory run should say what it saw without failing.
666 """
667 _, rejected = _review_evidence_keys_and_rejections(
668 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced, pr_title=pr_title
669 )
670 return [
671 {
672 "id": "review-verdict-insubstantial",
673 "severity": "major" if enforced else "minor",
674 "kind": "review",
675 "message": f"{key}: {reason}.",
676 }
677 for key, reason in sorted(rejected)
678 ]
681def _malformed_marker_findings(
682 items: list[dict[str, Any]],
683 *,
684 enforced: bool,
685) -> list[dict[str, Any]]:
686 """One finding per trusted comment whose header names more than one marker.
688 Such a header does not say which artifact the comment is, so
689 :func:`marker_in_header` refuses to classify it and the comment counts toward
690 nothing. Excluding it silently would reproduce the failure #926 named — a
691 comment sitting right there on the PR, reported as missing evidence — so the
692 exclusion is stated instead of inferred.
694 ``minor``, never blocking: the comment is malformed, not fraudulent, and the
695 requirement it failed to satisfy already fails on its own.
696 """
697 findings: list[dict[str, Any]] = []
698 for item in items:
699 if not _is_trusted_source(item, enforced=enforced):
700 continue
701 markers = header_markers(_body(item))
702 if len(markers) < 2:
703 continue
704 findings.append(
705 {
706 "id": MALFORMED_MARKER_FINDING,
707 "severity": "minor",
708 "kind": "evidence",
709 "message": (
710 "Comment header carries more than one keel marker "
711 f"({', '.join(markers)}); it is excluded from evidence."
712 ),
713 }
714 )
715 return findings
718def _distinct_vendor_finding(
719 review_contract: dict[str, Any],
720 *,
721 items: tuple[EvidenceItem, ...],
722 deferred: set[str],
723 pr_comments: list[dict[str, Any]],
724 pr_reviews: list[dict[str, Any]],
725 head_sha: str | None,
726 covered_heads: Collection[str] = (),
727 enforced: bool,
728) -> dict[str, Any] | None:
729 """Return a blocking finding when the optional vendor-distinctness check fails.
731 Off by default: ``None`` unless ``reviewers.require_distinct_vendors`` is set
732 on the contract. Skipped when review evidence is deferred so the knob never
733 overrides an explicit deferral.
734 """
735 if not _require_distinct_vendors(review_contract):
736 return None
737 if "review" in deferred or "all" in deferred:
738 return None
739 required = sum(1 for item in items if item.kind == "review" and item.id not in deferred)
740 if required <= 0:
741 return None
742 provenance = _review_vendor_provenance(
743 [*pr_comments, *pr_reviews],
744 head_sha=head_sha,
745 covered_heads=covered_heads,
746 enforced=enforced,
747 )
748 if review_panel(review_contract) == JURY_PANEL:
749 result = panel_vendor_check(
750 list(provenance.values()),
751 required_count=required,
752 minimum_vendors=_minimum_jury_vendors(review_contract),
753 )
754 else:
755 result = distinct_vendor_check(list(provenance.values()), required_count=required)
756 if result["ok"]:
757 return None
758 return {
759 "id": "review-vendor-distinctness",
760 "severity": "major",
761 "kind": "review",
762 "message": f"require_distinct_vendors: {result['reason']}.",
763 }
766_CLOSURE_MISMATCH_REASON = "closure comment does not match the ship_run ledger record"
769def _result_reason(item: EvidenceItem, mismatch: set[str]) -> str:
770 if item.id == "closure-comment-pr" and "pr" in mismatch:
771 return _CLOSURE_MISMATCH_REASON
772 if item.id == "closure-comment-issue" and "issue" in mismatch:
773 return _CLOSURE_MISMATCH_REASON
774 return f"missing required evidence: {item.id}"
777def _closure_mismatch_scopes(
778 *,
779 pr_comments: list[dict[str, Any]],
780 issue_comments: list[dict[str, Any]],
781 enforced: bool,
782 ledger_record: dict[str, Any] | None,
783) -> set[str]:
784 """Return scopes ({"pr"}/{"issue"}) where a marker closure mismatched the ledger.
786 A scope is reported only when a trusted marker-bearing closure exists but none
787 of them match the record — so a stale comment alongside a correct re-post does
788 not produce a misleading mismatch reason.
789 """
790 if ledger_record is None:
791 return set()
792 scopes: set[str] = set()
793 for scope, comments in (("pr", pr_comments), ("issue", issue_comments)):
794 markered = [
795 comment
796 for comment in comments
797 if _is_trusted_source(comment, enforced=enforced)
798 and _has_closure_marker(_body(comment))
799 ]
800 if markered and not any(
801 closure_body_matches_record(_body(comment), ledger_record) for comment in markered
802 ):
803 scopes.add(scope)
804 return scopes
807def _evidence_counts(
808 *,
809 pr_comments: list[dict[str, Any]],
810 issue_comments: list[dict[str, Any]],
811 pr_reviews: list[dict[str, Any]],
812 head_sha: str | None = None,
813 covered_heads: Collection[str] = (),
814 enforced: bool = True,
815 ledger_record: dict[str, Any] | None = None,
816) -> dict[str, int]:
817 review_keys = _review_evidence_keys(
818 [*pr_comments, *pr_reviews],
819 head_sha=head_sha,
820 covered_heads=covered_heads,
821 enforced=enforced,
822 )
823 return {
824 "closure_pr": sum(
825 _is_closure_comment(comment, enforced=enforced, record=ledger_record)
826 for comment in pr_comments
827 ),
828 "closure_issue": sum(
829 _is_closure_comment(comment, enforced=enforced, record=ledger_record)
830 for comment in issue_comments
831 ),
832 "review_verdict": len(review_keys),
833 # The jury requirement is met by the *standing* jury verdict at the head, and only
834 # when its consensus approves (#1429); a rejection is reported by
835 # `_jury_not_approved_finding` rather than counted.
836 "jury_verdict": int(
837 jury_verdict_approves(
838 _body(
839 _standing_jury_verdict(
840 pr_comments,
841 head_sha=head_sha,
842 covered_heads=covered_heads,
843 enforced=enforced,
844 )
845 or {}
846 )
847 )
848 ),
849 }
852def _is_present(item: EvidenceItem, counts: dict[str, int]) -> bool:
853 if item.id == "closure-comment-pr":
854 return counts["closure_pr"] >= 1
855 if item.id == "closure-comment-issue":
856 return counts["closure_issue"] >= 1
857 if item.kind == "review":
858 index = int(item.id.rsplit("-", 1)[1])
859 return counts["review_verdict"] >= index
860 if item.id == "jury-verdict":
861 return counts["jury_verdict"] >= 1
862 return False
865def _body(item: dict[str, Any]) -> str:
866 body = item.get("body")
867 return body if isinstance(body, str) else ""
870def _header_line(body: str) -> str:
871 """``body``'s header line: its first *non-empty* line, stripped.
873 Leading blank lines are skipped rather than treated as the end, for the same
874 reason :func:`_fields` skips them — a GitHub comment body routinely begins
875 with a newline. Everything after that line is prose.
876 """
877 for raw_line in (body or "").splitlines():
878 line = raw_line.strip()
879 if not line:
880 continue
881 return line
882 return ""
885def _unwrap_html_comment(line: str) -> str:
886 """Strip one literal ``<!-- … -->`` wrapper from ``line``, or return it unchanged.
888 Deliberately not a regex and deliberately not an HTML parser: the only
889 wrapper keel has to recognise is the one
890 :func:`keel.closure.render_closure_comment` writes. Anything else — an
891 unterminated ``<!--``, a ``--!>`` close, a second wrapper on the same line —
892 is left intact, so the marker check below sees the delimiters as tokens and
893 refuses to classify the comment. Failing to recognise a hand-rolled wrapper
894 costs a comment its classification; guessing at one would let a body render
895 as an invisible comment while counting as evidence.
896 """
897 if (
898 line.startswith(_HTML_COMMENT_OPEN)
899 and line.endswith(_HTML_COMMENT_CLOSE)
900 and len(line) >= len(_HTML_COMMENT_OPEN) + len(_HTML_COMMENT_CLOSE)
901 ):
902 return line[len(_HTML_COMMENT_OPEN) : -len(_HTML_COMMENT_CLOSE)].strip()
903 return line
906def header_markers(body: str) -> tuple[str, ...]:
907 """Return the distinct :data:`CLASSIFICATION_MARKERS` ``body``'s header carries.
909 The header line, once unwrapped, must consist of markers and nothing else —
910 that is exactly what every renderer emits, and it is what separates a marker
911 line from a sentence that happens to name one. So this is empty for an
912 ordinary comment (including one whose first line *mentions* a marker in
913 prose, and one wearing a wrapper keel does not write), one entry for a
914 well-formed artifact, and two or more for a malformed one, which
915 :func:`marker_in_header` refuses to classify and
916 :func:`_malformed_marker_findings` reports.
917 """
918 tokens = _unwrap_html_comment(_header_line(body)).split()
919 if not tokens or not _CLASSIFICATION_MARKERS_SET.issuperset(tokens):
920 return ()
921 return tuple(marker for marker in CLASSIFICATION_MARKERS if marker in tokens)
924def marker_in_header(body: str) -> str | None:
925 """Return the single keel marker ``body`` is anchored to, or ``None`` (#1026).
927 **The header block is the only place a marker classifies a comment.** A marker
928 further down is prose — a reviewer writing "I checked the jury-verdict marker
929 handling" is quoting a string, not filing a jury verdict. Testing
930 ``MARKER in body`` could not tell the two apart: two `keel.review-verdict.v1`
931 comments whose scope mentioned ``keel.jury-verdict.v1`` were counted as
932 ``jury_verdict: 2, review_verdict: 0``, and the review that happened was
933 invisible to the gate.
935 ``None`` for a comment that carries no marker *and* for one whose header
936 carries several: a header naming two artifacts does not say which one it is,
937 so it is excluded rather than counted for either.
938 """
939 markers = header_markers(body)
940 return markers[0] if len(markers) == 1 else None
943def _has_closure_marker(body: str) -> bool:
944 return marker_in_header(body) == closure.CLOSURE_SCHEMA_VERSION
947#: The idempotency marker ``keel post-comment`` appends to a posted body so a re-post
948#: can find and edit its own comment. It is transport bookkeeping, not content, so it is
949#: stripped before a closure body is compared to its canonical render.
950#:
951#: Matched in the *exact* form the transport emits — a run id, then the close, then end
952#: of line. A permissive ``.*?`` would let a trusted author smuggle arbitrary text past
953#: the verbatim comparison: an HTML comment ends at its first ``-->``, so anything after
954#: that renders visibly on the page while the whole line still normalizes away.
955RUN_ID_MARKER_RE = re.compile(r"^\s*<!--\s*keel\.run-id:\s*[\w.:@/+-]+\s*-->\s*$")
958def _normalize_closure_body(body: str) -> str:
959 """Normalize a closure body for content comparison.
961 Robust to harmless formatting drift but sensitive to real content changes:
962 trailing whitespace is stripped per line, runs of blank lines collapse to a
963 single blank line, and leading/trailing blank lines are dropped.
965 The transport's ``keel.run-id`` marker line is dropped too. Without that, closure
966 fidelity and post-comment idempotency were mutually exclusive: the marker is what
967 lets a re-post edit its own comment instead of duplicating, and its presence made
968 the body differ from the canonical render.
969 """
970 lines = [line.rstrip() for line in body.splitlines() if not RUN_ID_MARKER_RE.match(line)]
971 normalized: list[str] = []
972 for line in lines:
973 if not line and (not normalized or not normalized[-1]):
974 continue
975 normalized.append(line)
976 while normalized and not normalized[-1]:
977 normalized.pop()
978 return "\n".join(normalized)
981def closure_body_matches_record(body: str, record: dict[str, Any]) -> bool:
982 """Return whether ``body`` matches the canonical render of ``record``."""
983 expected = closure.render_closure_comment(record)
984 return _normalize_closure_body(body) == _normalize_closure_body(expected)
987def _is_closure_comment(
988 item: dict[str, Any],
989 *,
990 enforced: bool = True,
991 record: dict[str, Any] | None = None,
992) -> bool:
993 if not _is_trusted_source(item, enforced=enforced):
994 return False
995 if not _has_closure_marker(_body(item)):
996 return False
997 if record is None:
998 return True
999 return closure_body_matches_record(_body(item), record)
1002def _run_context_findings(
1003 *,
1004 pr_comments: list[dict[str, Any]],
1005 issue_comments: list[dict[str, Any]],
1006 enforced: bool,
1007 ledger_record: dict[str, Any] | None = None,
1008) -> list[dict[str, Any]]:
1009 comments = [*pr_comments, *issue_comments]
1010 findings: list[dict[str, Any]] = []
1011 for item in comments:
1012 if not _is_closure_comment(item, enforced=enforced, record=ledger_record):
1013 continue
1014 body = _body(item)
1015 if _has_empty_run_context(body):
1016 findings.append(
1017 {
1018 "id": "run-context-empty",
1019 "severity": "major" if enforced else "minor",
1020 "kind": "closure",
1021 "message": "Closure comment Run context is fully degraded.",
1022 }
1023 )
1024 return findings
1027def _has_empty_run_context(body: str) -> bool:
1028 if "### Run context" not in body:
1029 return False
1030 fields = _run_context_fields(body)
1031 return fields == {
1032 "host agent": "unknown",
1033 "transport": "unknown",
1034 "profile": "unknown",
1035 "jury": "off",
1036 "consent": "unknown (scopes: none)",
1037 }
1040def _run_context_fields(body: str) -> dict[str, str]:
1041 fields: dict[str, str] = {}
1042 in_block = False
1043 for line in body.splitlines():
1044 if line.strip() == "### Run context":
1045 in_block = True
1046 continue
1047 if in_block and line.startswith("### "):
1048 break
1049 if not in_block:
1050 continue
1051 match = re.match(r"^-\s+\*\*(?P<key>[^*]+):\*\*\s+(?P<value>.+?)\s*$", line)
1052 if match:
1053 fields[match.group("key").strip().lower()] = match.group("value").strip().lower()
1054 return fields
1057def _is_trusted_source(item: dict[str, Any], *, enforced: bool = True) -> bool:
1058 """Return whether GitHub marks this evidence source as trusted.
1060 Live GitHub comment/review payloads include ``author_association``. Enforced
1061 evidence fails closed when that field is absent because offline fixtures are
1062 agent-writable and must not manufacture trust. Untrusted explicit
1063 associations fail closed even if the author type is ``Bot``.
1064 """
1065 association = item.get("author_association")
1066 if association is None:
1067 return not enforced
1068 if isinstance(association, str) and association.upper() in TRUSTED_AUTHOR_ASSOCIATIONS:
1069 return True
1070 return False
1073def _is_ship_assessment(body: str) -> bool:
1074 """Whether ``body`` is a ship assessment comment, decided by its header (#1035).
1076 Header-anchored for the same reason markers are (#1026): this is consulted as an
1077 *exclusion* by :func:`_is_review_verdict_body` and :func:`_is_jury_verdict`, so a
1078 whole-body substring test let a reviewer disarm their own verdict by quoting the
1079 heading while describing what they reviewed ("the ``### \U0001f6a2 keel ship``
1080 comment claims the gates passed, but…"). The verdict was then silently uncounted
1081 and ``evidence-verify`` reported it missing from a PR it was sitting on.
1083 The heading is a Markdown heading rather than a versioned ``keel.*.v1`` marker, so
1084 it cannot join :data:`CLASSIFICATION_MARKERS`; it gets the same anchoring instead.
1085 A real assessment leads with the heading (the workflow writes it first) or with the
1086 CLI's own banner, so nothing that armed the gate through a genuine assessment
1087 comment stops arming it.
1088 """
1089 header = _header_line(body)
1090 return header.startswith(SHIP_ASSESSMENT_HEADING) or header.startswith(SHIP_ASSESSMENT_BANNER)
1093def count_review_verdicts(
1094 pr_comments: list[dict[str, Any]] | None = None,
1095 pr_reviews: list[dict[str, Any]] | None = None,
1096 *,
1097 head_sha: str | None = None,
1098 covered_heads: Collection[str] = (),
1099 enforced: bool = True,
1100 pr_title: str = "",
1101) -> int:
1102 """Count distinct trusted review-verdict reviewers for a PR.
1104 This is the same evidence-side counting the verify report uses for the
1105 ``review`` items: it collapses idempotent re-posts by the same reviewer to
1106 one verdict and only counts trusted, head-bound verdicts whose reviewer's
1107 latest ``Verdict:`` line approves (#1426) — a reviewer who requested changes
1108 reviewed, but did not pass, the change. Reused by capture reconcile to
1109 cross-check the ledger's recorded reviewer count.
1110 """
1111 keys = _review_evidence_keys(
1112 [*(pr_comments or []), *(pr_reviews or [])],
1113 head_sha=head_sha,
1114 covered_heads=covered_heads,
1115 enforced=enforced,
1116 pr_title=pr_title,
1117 )
1118 return len(keys)
1121def _review_evidence_keys(
1122 items: list[dict[str, Any]],
1123 *,
1124 head_sha: str | None = None,
1125 covered_heads: Collection[str] = (),
1126 enforced: bool = True,
1127 pr_title: str = "",
1128) -> set[str]:
1129 keys, _ = _review_evidence_keys_and_rejections(
1130 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced, pr_title=pr_title
1131 )
1132 return keys
1135def verdict_reviewers(
1136 items: list[dict[str, Any]],
1137 *,
1138 head_sha: str | None = None,
1139 covered_heads: Collection[str] = (),
1140 enforced: bool = True,
1141) -> tuple[str, ...]:
1142 """The reviewers whose verdicts count for ``head_sha``, by name, sorted (#1422).
1144 The verdicts :func:`verify` counts toward ``review-verdict-N`` — trusted, head-pinned,
1145 with substance, and approving at the reviewer's latest word (#1426), so a reviewer who
1146 requested changes is not listed as having passed it — named by their ``reviewer:``
1147 field, else by the commenter's login. A
1148 verdict keyed only by its body names nobody and is left out. This is what a closure
1149 comment written after the merge says reviewed the change, read off the pull request
1150 rather than recalled.
1151 """
1152 keys = _review_evidence_keys(
1153 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced
1154 )
1155 names = []
1156 for key in keys:
1157 kind, _, name = key.partition(":")
1158 if kind in ("reviewer", "user"):
1159 names.append(name)
1160 return tuple(sorted(names))
1163@dataclass(frozen=True)
1164class _ReviewTally:
1165 """What the review verdicts at a head add up to, per reviewer (#1426).
1167 ``accepted`` are the reviewer keys whose verdict counts toward ``review-verdict-N``;
1168 ``vendors`` maps each of them to its declared vendor (or ``None``). ``insubstantial``
1169 are the ``(key, reason)`` pairs refused for naming nothing (#926), and
1170 ``not_approved`` the ``(key, token, head)`` triples of reviewers whose standing
1171 verdict does not approve (``token`` is ``None`` when it has no readable
1172 ``Verdict:`` line). Both refusals are reported, never dropped.
1173 """
1175 accepted: frozenset[str]
1176 vendors: dict[str, str | None]
1177 insubstantial: tuple[tuple[str, str], ...]
1178 not_approved: tuple[tuple[str, str | None, str], ...]
1181def _posted_at(item: dict[str, Any]) -> str:
1182 """When ``item`` was posted, as GitHub's ISO-8601 string, or ``""`` when unknown.
1184 An issue comment carries ``created_at`` and a pull-request review ``submitted_at``.
1185 ``updated_at`` is deliberately not read: editing an old approval (a typo fix, an
1186 idempotent re-post of the same run's comment) must not make it outrank a rejection
1187 posted after it. ISO-8601 UTC strings order lexically, so no clock is parsed.
1188 """
1189 for field in ("created_at", "submitted_at"):
1190 value = item.get(field)
1191 if isinstance(value, str) and value:
1192 return value
1193 return ""
1196def review_verdict_token(body: str) -> str | None:
1197 """The upper-cased first word of ``body``'s first ``Verdict:`` line, or ``None`` (#1426).
1199 The line is the one :func:`keel.artifacts.render_review_verdict` writes
1200 (``Verdict: <verdict>``); it is found anywhere in the comment, header block included,
1201 but only at the start of a line. The first word is read case-insensitively, with
1202 wrapper punctuation skipped and ``-`` spelled ``_``, so ``APPROVE — minor nits``,
1203 ``**lgtm**`` and ``request-changes`` read ``APPROVE``, ``LGTM`` and
1204 ``REQUEST_CHANGES``. ``None`` when the comment has no such line or the line has no
1205 word, which :func:`verdict_approves` treats as not an approval.
1206 """
1207 return _line_token(body, _VERDICT_LINE_RE)
1210def _line_token(body: str, line_re: re.Pattern[str]) -> str | None:
1211 """The first word of the first line ``line_re`` matches, read as a verdict token."""
1212 for line in body.splitlines():
1213 match = line_re.match(line)
1214 if match is None:
1215 continue
1216 token = _VERDICT_TOKEN_RE.match(match.group("value").strip())
1217 return token.group("token").upper().replace("-", "_") if token else None
1218 return None
1221def verdict_approves(body: str) -> bool:
1222 """Whether a review verdict's ``Verdict:`` line approves (:data:`APPROVING_VERDICTS`)."""
1223 return review_verdict_token(body) in APPROVING_VERDICTS
1226def jury_verdict_token(body: str) -> str | None:
1227 """The consensus token on a jury verdict's ``AI Jury verdict:`` line, or ``None`` (#1429).
1229 Read exactly as :func:`review_verdict_token` reads a review's ``Verdict:`` line — first
1230 word, any case, wrapper punctuation skipped, ``-`` spelled ``_`` — off the line
1231 :func:`keel.artifacts.render_jury_verdict` writes, ``AI Jury verdict: <verdict>.``, so
1232 its trailing full stop is not part of the token. ``None`` when there is no such line or
1233 it carries no word.
1234 """
1235 return _line_token(body, _JURY_VERDICT_LINE_RE)
1238def jury_verdict_approves(body: str) -> bool:
1239 """Whether a jury verdict's consensus approves (:data:`APPROVING_VERDICTS`, #1429).
1241 The same set a review verdict is read against: the chair's ``APPROVE`` / ``READY`` reach
1242 the comment as ``LGTM`` through :func:`keel.jury.map_verdict`. ``REQUEST_CHANGES`` (and
1243 ``NEEDS_INFO``, which maps onto it), ``COMMENT``, ``ABSTAIN`` — the token
1244 :func:`keel.jury.jury_verdict` writes when the panel produced no chair synthesis — and
1245 ``NO_QUORUM`` are not approvals, and neither is an unknown word or a missing line.
1246 """
1247 return jury_verdict_token(body) in APPROVING_VERDICTS
1250def _not_approved_message(key: str, token: str | None, head: str) -> str:
1251 """How a refusal names a reviewer whose standing verdict does not approve (#1426)."""
1252 name = _reviewer_name(key)
1253 if token in REQUEST_CHANGES_VERDICTS:
1254 return f"{name} requests changes at {head}."
1255 why = f"verdict {token}" if token else "no readable Verdict line"
1256 return f"{name} does not approve at {head} ({why})."
1259def _verdict_head(item: dict[str, Any], body: str) -> str:
1260 """The head a verdict answers for: its ``head:`` field, else the review's commit."""
1261 recorded = _fields(body).get("head")
1262 if recorded:
1263 return recorded
1264 commit_id = item.get("commit_id")
1265 return commit_id if isinstance(commit_id, str) and commit_id else "an unrecorded head"
1268def _reviewer_name(key: str) -> str:
1269 """A reviewer key as a refusal names it: the reviewer, or the verdict's digest."""
1270 kind, _, name = key.partition(":")
1271 return f"an unnamed reviewer (verdict {name[:12]})" if kind == "body" else name
1274def _review_tally(
1275 items: list[dict[str, Any]],
1276 *,
1277 head_sha: str | None = None,
1278 covered_heads: Collection[str] = (),
1279 enforced: bool = True,
1280 pr_title: str = "",
1281) -> _ReviewTally:
1282 """Tally the trusted review verdicts that answer for ``head_sha`` (#1426).
1284 **A reviewer's latest verdict is their review.** Verdicts are ordered by when they
1285 were posted (:func:`_posted_at`, then their position), grouped by reviewer key, and
1286 each reviewer is judged by the last one — at the head or at a head it descends from
1287 by capture commits alone (``covered_heads``, #1203):
1289 * **It does not approve** (:func:`verdict_approves`): the reviewer is refused, with
1290 their stance and the head it was given at. A rejection is evidence; dropping it
1291 silently would let another reviewer's approval outvote it. So a reviewer who
1292 approved and then requested changes holds the merge.
1293 * **It approves**: the reviewer counts when any verdict in their run of approvals
1294 since their last non-approving one has substance (:func:`verdict_substance`) —
1295 a thin verdict followed by a real one is accepted, because the later comment is
1296 the review (#926). A reviewer who requested changes and then approved counts; a
1297 thin approval cannot overturn their own substantive rejection.
1299 A verdict pinned to any other head is not read at all: a rejection of an older head
1300 is answered by the commits that moved it, and the reviewer's next verdict.
1301 """
1302 ordered = sorted(enumerate(items), key=lambda pair: (_posted_at(pair[1]), pair[0]))
1303 by_key: dict[str, list[tuple[dict[str, Any], str]]] = {}
1304 for _, item in ordered:
1305 if not _is_trusted_source(item, enforced=enforced):
1306 continue
1307 body = _body(item)
1308 if not _is_review_verdict_body(body):
1309 continue
1310 if not _matches_head(item, body, head_sha, covered_heads):
1311 continue
1312 by_key.setdefault(_reviewer_key(item, body), []).append((item, body))
1313 accepted: set[str] = set()
1314 vendors: dict[str, str | None] = {}
1315 insubstantial: list[tuple[str, str]] = []
1316 not_approved: list[tuple[str, str | None, str]] = []
1317 for key, verdicts in by_key.items():
1318 latest_item, latest_body = verdicts[-1]
1319 token = review_verdict_token(latest_body)
1320 if token not in APPROVING_VERDICTS:
1321 not_approved.append((key, token, _verdict_head(latest_item, latest_body)))
1322 continue
1323 approvals: list[str] = []
1324 for _, body in reversed(verdicts):
1325 if not verdict_approves(body):
1326 break
1327 approvals.append(body)
1328 substance = [verdict_substance(body, pr_title=pr_title) for body in approvals]
1329 if not any(ok for ok, _ in substance):
1330 insubstantial.extend((key, reason) for _, reason in reversed(substance))
1331 continue
1332 accepted.add(key)
1333 vendor = _fields(verdicts[0][1]).get("vendor")
1334 vendors[key] = vendor.lower() if vendor else None
1335 return _ReviewTally(
1336 accepted=frozenset(accepted),
1337 vendors=vendors,
1338 insubstantial=tuple(insubstantial),
1339 not_approved=tuple(not_approved),
1340 )
1343def _review_evidence_keys_and_rejections(
1344 items: list[dict[str, Any]],
1345 *,
1346 head_sha: str | None = None,
1347 covered_heads: Collection[str] = (),
1348 enforced: bool = True,
1349 pr_title: str = "",
1350) -> tuple[set[str], list[tuple[str, str]]]:
1351 """Accepted reviewer keys, and the (key, reason) pairs refused for substance.
1353 Rejections are returned rather than dropped so the gate can *hold with a
1354 reason* — a verdict silently not counted would surface as "missing
1355 review-verdict-2", which sends the operator looking for a comment that is
1356 right there (#926). A reviewer whose standing verdict does not approve is in
1357 neither: :func:`_verdict_not_approved_findings` reports them (#1426).
1358 """
1359 tally = _review_tally(
1360 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced, pr_title=pr_title
1361 )
1362 return set(tally.accepted), list(tally.insubstantial)
1365def _review_vendor_provenance(
1366 items: list[dict[str, Any]],
1367 *,
1368 head_sha: str | None = None,
1369 covered_heads: Collection[str] = (),
1370 enforced: bool = True,
1371) -> dict[str, str | None]:
1372 """Map each accepted review-verdict reviewer-key to its declared vendor.
1374 The value is the lower-cased ``vendor:`` provenance for that verdict, or
1375 ``None`` when the verdict carries no vendor field. Keys are exactly
1376 :func:`_review_evidence_keys`, so duplicate reviewer-keys collapse to one
1377 entry (idempotent re-posts do not inflate the vendor set), and a reviewer
1378 whose verdict does not count lends the panel no vendor either (#1426).
1379 """
1380 return dict(
1381 _review_tally(
1382 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced
1383 ).vendors
1384 )
1387def _verdict_not_approved_findings(
1388 items: list[dict[str, Any]],
1389 *,
1390 head_sha: str | None,
1391 covered_heads: Collection[str] = (),
1392 enforced: bool,
1393) -> list[dict[str, Any]]:
1394 """One finding per reviewer whose standing verdict at the head does not approve (#1426).
1396 ``major`` when the gate is enforced, so a rejection **holds** the merge with its
1397 reviewer and head named — ``review-verdict-not-approved: alice requests changes at
1398 <head>`` in :func:`refusal_reason` — instead of merely not counting, where another
1399 reviewer's approval could make up the number. ``minor`` otherwise, mirroring the
1400 other content findings.
1401 """
1402 tally = _review_tally(items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced)
1403 return [
1404 {
1405 "id": VERDICT_NOT_APPROVED_FINDING,
1406 "severity": "major" if enforced else "minor",
1407 "kind": "review",
1408 "message": _not_approved_message(key, token, head),
1409 }
1410 for key, token, head in sorted(tally.not_approved, key=lambda entry: entry[0])
1411 ]
1414def distinct_vendor_check(
1415 vendors: Sequence[str | None],
1416 *,
1417 required_count: int,
1418) -> dict[str, Any]:
1419 """Pure vendor-distinctness check over review-verdict provenance.
1421 ``vendors`` is one entry per accepted review verdict: the declared vendor, or
1422 ``None`` when the verdict carries no vendor provenance. The check passes only
1423 when at least ``required_count`` verdicts each declare a vendor and those
1424 vendors are all distinct. It fails when a required verdict is missing vendor
1425 provenance, or when two required verdicts share a vendor.
1427 Returns ``{ok, reason, duplicated, missing_provenance}``. No I/O — fully
1428 unit-testable. A non-positive ``required_count`` always passes (nothing to
1429 require).
1430 """
1431 if required_count <= 0:
1432 return {"ok": True, "reason": None, "duplicated": [], "missing_provenance": 0}
1433 present = [vendor for vendor in vendors if vendor]
1434 missing = len(vendors) - len(present)
1435 seen: set[str] = set()
1436 duplicated: list[str] = []
1437 for vendor in present:
1438 if vendor in seen and vendor not in duplicated:
1439 duplicated.append(vendor)
1440 seen.add(vendor)
1441 if len(present) < required_count:
1442 return {
1443 "ok": False,
1444 "reason": "missing vendor provenance on required review verdict(s)",
1445 "duplicated": duplicated,
1446 "missing_provenance": missing,
1447 }
1448 if duplicated:
1449 return {
1450 "ok": False,
1451 "reason": f"review verdicts share a vendor: {', '.join(sorted(duplicated))}",
1452 "duplicated": sorted(duplicated),
1453 "missing_provenance": missing,
1454 }
1455 return {"ok": True, "reason": None, "duplicated": [], "missing_provenance": missing}
1458def panel_vendor_check(
1459 vendors: Sequence[str | None],
1460 *,
1461 required_count: int,
1462 minimum_vendors: int,
1463) -> dict[str, Any]:
1464 """Cross-vendor check for a **jury panel**, whose size the panel sets (#1015).
1466 :func:`distinct_vendor_check` asks for one distinct vendor per required
1467 verdict, which is the right question for a bench keel staffs: keel chose the
1468 seats, so keel can insist each one is a different vendor. It is the wrong
1469 question for a panel, where the *panel* chose the seats and three ballots from
1470 two vendors is a legitimate cross-vendor review — the same shape
1471 :data:`keel.ship.MINIMUM_JURY_VENDORS` already accepts as a gating jury.
1473 So the panel is held to the jury's own rule instead: every required ballot
1474 must declare a vendor, and the ballots together must span at least
1475 ``minimum_vendors`` distinct ones. A whole panel from one vendor is one
1476 opinion N times and fails, which is precisely what the strict check would
1477 have caught — the relaxation is only in *how many* distinct vendors are
1478 demanded, never in whether provenance is required at all.
1480 Returns the same ``{ok, reason, duplicated, missing_provenance}`` shape as
1481 :func:`distinct_vendor_check`, so a caller renders one finding either way.
1482 """
1483 if required_count <= 0:
1484 return {"ok": True, "reason": None, "duplicated": [], "missing_provenance": 0}
1485 present = [vendor for vendor in vendors if vendor]
1486 missing = len(vendors) - len(present)
1487 distinct = sorted(set(present))
1488 duplicated = sorted({vendor for vendor in present if present.count(vendor) > 1})
1489 if len(present) < required_count:
1490 return {
1491 "ok": False,
1492 "reason": "missing vendor provenance on required review verdict(s)",
1493 "duplicated": duplicated,
1494 "missing_provenance": missing,
1495 }
1496 if len(distinct) < minimum_vendors:
1497 return {
1498 "ok": False,
1499 "reason": (
1500 f"jury panel of {len(present)} ballot(s) spans "
1501 f"{len(distinct)} distinct vendor(s), below the minimum of {minimum_vendors}"
1502 ),
1503 "duplicated": duplicated,
1504 "missing_provenance": missing,
1505 }
1506 return {"ok": True, "reason": None, "duplicated": duplicated, "missing_provenance": missing}
1509def _label_values(labels: Sequence[str] | None, prefix: str) -> list[str]:
1510 """Lower-cased values of every ``<prefix><value>`` label, blanks dropped."""
1511 values: list[str] = []
1512 for label in labels or ():
1513 if not isinstance(label, str) or not label.startswith(prefix):
1514 continue
1515 value = label[len(prefix) :].strip().lower()
1516 if value:
1517 values.append(value)
1518 return values
1521def agent_label_vendors(labels: Sequence[str] | None) -> list[str]:
1522 """Return the lower-cased vendor slugs from every ``agent:<vendor>`` label.
1524 A blank vendor (a bare ``agent:`` label) is ignored. Order is preserved and
1525 duplicates are kept so callers can reason about the raw label set; this is a
1526 pure helper with no I/O.
1527 """
1528 return _label_values(labels, AGENT_LABEL_PREFIX)
1531def model_label_bases(labels: Sequence[str] | None) -> list[str]:
1532 """Return the lower-cased base slugs from every ``model:<base>`` label.
1534 The mirror of :func:`agent_label_vendors` for the second half of keel's
1535 attribution vocabulary. Same conventions: blanks dropped, order and duplicates
1536 preserved, no I/O.
1537 """
1538 return _label_values(labels, MODEL_LABEL_PREFIX)
1541def ledger_implementer(ledger_record: dict[str, Any] | None) -> str | None:
1542 """Return the raw ``actors.implementer`` string from a ship_run record, or ``None``.
1544 The full ``vendor`` / ``vendor:model`` value, not just the vendor half — the
1545 vocabulary check needs the model too. Blank/absent reads as ``None``. Pure.
1546 """
1547 if not isinstance(ledger_record, dict):
1548 return None
1549 actors = ledger_record.get("actors")
1550 implementer = actors.get("implementer") if isinstance(actors, dict) else None
1551 if not isinstance(implementer, str) or not implementer.strip():
1552 return None
1553 return implementer.strip()
1556def ledger_implementer_vendor(ledger_record: dict[str, Any] | None) -> str | None:
1557 """Return the implementer's vendor slug from a ship_run ``ledger_record``.
1559 The ledger stores the effective implementer as a codename or ``vendor:model``
1560 string under ``actors.implementer``; the vendor is the part before the first
1561 ``:``. Returns ``None`` when no record, no implementer, or a blank implementer
1562 is recorded so the cross-check can degrade to presence-only. Pure — no I/O.
1563 """
1564 implementer = ledger_implementer(ledger_record)
1565 if implementer is None:
1566 return None
1567 vendor, _ = agents.split_delegate(implementer)
1568 vendor = vendor.strip().lower()
1569 return vendor or None
1572def attribution_check(
1573 labels: Sequence[str] | None,
1574 *,
1575 implementer_vendor: str | None = None,
1576) -> dict[str, Any]:
1577 """Pure attribution-label check over a PR's labels and the ledger implementer.
1579 Two layers, both fail-closed only on a real contradiction:
1581 * **Presence** — at least one non-blank ``agent:<vendor>`` label must exist.
1582 Missing one is a ``missing-label`` finding.
1583 * **Cross-check** — when ``implementer_vendor`` is known (a ship_run record
1584 recorded an implementer), one of the PR's ``agent:*`` vendors must match it.
1585 A mismatch is a ``vendor-mismatch`` finding. When ``implementer_vendor`` is
1586 ``None`` (no record / no implementer) only the presence layer runs, so PRs
1587 that predate attribution recording are not broken.
1589 Returns ``{ok, reason, label_vendors, implementer_vendor}``. No I/O.
1590 """
1591 label_vendors = agent_label_vendors(labels)
1592 implementer = implementer_vendor.strip().lower() if implementer_vendor else None
1593 if not label_vendors:
1594 return {
1595 "ok": False,
1596 "reason": "missing-label",
1597 "label_vendors": label_vendors,
1598 "implementer_vendor": implementer,
1599 }
1600 if implementer is not None and implementer not in label_vendors:
1601 return {
1602 "ok": False,
1603 "reason": "vendor-mismatch",
1604 "label_vendors": label_vendors,
1605 "implementer_vendor": implementer,
1606 }
1607 return {
1608 "ok": True,
1609 "reason": None,
1610 "label_vendors": label_vendors,
1611 "implementer_vendor": implementer,
1612 }
1615def attribution_vocabulary_check(
1616 labels: Sequence[str] | None,
1617 *,
1618 implementer: str | None,
1619) -> dict[str, Any]:
1620 """Check a PR's attribution labels against keel's own vocabulary (#1013).
1622 :func:`attribution_check` compares the label's *vendor* with the ledger's
1623 *vendor*. That is a comparison of two hand-written strings: when the host wrote
1624 ``agent:gemini`` on the PR **and** ``gemini:gemini-3.8-flash-high`` into the
1625 ledger, the two agreed and the gate passed — while keel's own vocabulary for that
1626 run is ``agent:agy`` / ``model:gemini-3``. This check closes that hole by deriving
1627 the expected labels from :func:`keel.agents.attribution` instead of comparing the
1628 prose to itself.
1630 Only labels the PR actually carries are judged: a missing ``agent:`` label is
1631 :func:`attribution_check`'s ``missing-label`` finding and is not repeated here,
1632 and a ledger implementer with no model (``claude``) yields no expected
1633 ``model:`` label, so ``model:`` labels are left alone in that case.
1635 Returns ``{ok, checked, reason, expected, actual, implementer}``. ``checked`` is
1636 ``False`` when there was nothing to compare against — no ledger record, no
1637 recorded implementer — so a caller can tell "agrees" from "could not tell".
1638 Pure — no I/O.
1639 """
1640 expected = agents.attribution_from_implementer(implementer)
1641 actual = {
1642 "agent_labels": agent_label_vendors(labels),
1643 "model_labels": model_label_bases(labels),
1644 }
1645 if expected is None:
1646 return {
1647 "ok": True,
1648 "checked": False,
1649 "reason": "no-implementer",
1650 "expected": None,
1651 "actual": actual,
1652 "implementer": None,
1653 }
1654 recorded = implementer.strip() if isinstance(implementer, str) else None
1655 result = {
1656 "ok": True,
1657 "checked": True,
1658 "reason": None,
1659 "expected": dict(expected),
1660 "actual": actual,
1661 "implementer": recorded,
1662 }
1663 expected_agent = expected["agent_label"][len(AGENT_LABEL_PREFIX) :]
1664 expected_model = expected["model_label"]
1665 if actual["agent_labels"] and expected_agent not in actual["agent_labels"]:
1666 result["ok"] = False
1667 result["reason"] = "agent-label"
1668 return result
1669 if expected_model is not None:
1670 base = expected_model[len(MODEL_LABEL_PREFIX) :]
1671 if actual["model_labels"] and base not in actual["model_labels"]:
1672 result["ok"] = False
1673 result["reason"] = "model-label"
1674 return result
1677def _attribution_vocabulary_finding(
1678 *,
1679 pr_labels: Sequence[str] | None,
1680 enforced: bool,
1681 ledger_record: dict[str, Any] | None,
1682) -> dict[str, Any] | None:
1683 """Return the blocking ``attribution-vocabulary`` finding, or ``None``.
1685 Skips — never fails — when the gate is inactive, when labels were not fetched, or
1686 when no ledger record named an implementer: the check needs a recorded implementer
1687 to derive the expected labels from, and refusing a PR because keel could not read
1688 its own ledger would be a fail-closed rule with nothing behind it.
1689 """
1690 if not enforced or pr_labels is None:
1691 return None
1692 result = attribution_vocabulary_check(
1693 pr_labels,
1694 implementer=ledger_implementer(ledger_record),
1695 )
1696 if result["ok"]:
1697 return None
1698 expected = result["expected"]
1699 labels = ", ".join(
1700 label for label in (expected["agent_label"], expected["model_label"]) if label
1701 )
1702 observed = ", ".join(
1703 [
1704 *(f"{AGENT_LABEL_PREFIX}{value}" for value in result["actual"]["agent_labels"]),
1705 *(f"{MODEL_LABEL_PREFIX}{value}" for value in result["actual"]["model_labels"]),
1706 ]
1707 )
1708 return {
1709 "id": "attribution-vocabulary",
1710 "severity": "major",
1711 "kind": "attribution",
1712 "message": (
1713 f"PR attribution labels ({observed}) are not keel's vocabulary for ledger "
1714 f"implementer {result['implementer']!r}. Expected: {labels}. "
1715 "Obtain labels from `keel attribution` instead of composing them by hand."
1716 ),
1717 }
1720def _unarmed_finding(
1721 *,
1722 enforced: bool,
1723 dry_run: bool,
1724 require_armed: bool,
1725 waived: bool,
1726) -> dict[str, Any] | None:
1727 """Return a blocking finding when the gate was never armed, else ``None``.
1729 Opt-in via ``require_armed`` so existing callers keep today's behavior. An
1730 unarmed gate derives no requirements, so without this the report passes
1731 having verified nothing — indistinguishable from a genuine pass. Skipped
1732 under ``dry_run``, where producing no evidence is the expected outcome, and
1733 when ``waived``: an operator disarming the gate on purpose is the sanctioned
1734 way out, and the whole point is to separate that from arming by accident.
1735 """
1736 if not require_armed or dry_run or enforced or waived:
1737 return None
1738 return {
1739 "id": "gate-unarmed",
1740 "severity": "major",
1741 "kind": "arming",
1742 "message": (
1743 "Evidence gate is not armed, so no requirements were checked. Arm it via ship "
1744 "provenance (the keel.ship-provenance.v1 comment a live run posts on its PR, a "
1745 "ship branch, a posted review verdict, the ship-run ledger, or the gate label), "
1746 "or disarm deliberately with the operator waiver label."
1747 ),
1748 }
1751def _attribution_finding(
1752 *,
1753 pr_labels: Sequence[str] | None,
1754 enforced: bool,
1755 ledger_record: dict[str, Any] | None,
1756) -> dict[str, Any] | None:
1757 """Return a blocking attribution finding when the gate is active, else ``None``.
1759 Only runs when the evidence gate is active (``enforced``) *and* PR labels were
1760 actually fetched (``pr_labels is not None``): the presence check is cheap and
1761 default-on, while the vendor cross-check engages only when the ledger recorded
1762 an implementer vendor. Degrades gracefully — labels not available skips the
1763 check entirely, no record means presence-only, and a gate-inactive run skips
1764 the check (back-compat with callers that never pass labels).
1765 """
1766 if not enforced or pr_labels is None:
1767 return None
1768 implementer_vendor = ledger_implementer_vendor(ledger_record)
1769 result = attribution_check(pr_labels, implementer_vendor=implementer_vendor)
1770 if result["ok"]:
1771 return None
1772 if result["reason"] == "missing-label":
1773 message = "PR is missing a mandatory agent:<vendor> attribution label."
1774 else:
1775 message = (
1776 "PR agent:<vendor> attribution "
1777 f"({', '.join(result['label_vendors'])}) does not match the ship_run "
1778 f"ledger implementer vendor ({result['implementer_vendor']})."
1779 )
1780 return {
1781 "id": "attribution-label",
1782 "severity": "major",
1783 "kind": "attribution",
1784 "message": message,
1785 }
1788#: A file named without its directory — ``evidence.py``, ``CHANGELOG.md``.
1789#:
1790#: The path anchors read *any* one-to-five-character extension, because a
1791#: directory has already proved the token is a path. With no directory nothing
1792#: is left to carry that proof: ``Node.js``, ``Next.js``, ``Vue.js`` and
1793#: ``D3.js`` are spelled exactly like ``evidence.py``, and in prose a product
1794#: is named far more often than a bare file is. Listing the extension cannot
1795#: separate them — ``.js`` is a real source extension, and dropping it would
1796#: refuse real reviews to refuse four product names — so this form does not
1797#: anchor a verdict by itself. It *corroborates*:
1798#: see :data:`_VERDICT_CORROBORATORS`.
1799_VERDICT_SOURCE_FILE = re.compile(
1800 r"\b[\w-]+\.(?:py|pyi|md|rst|txt|ya?ml|toml|json|cfg|ini|lock"
1801 r"|sh|js|mjs|cjs|ts|tsx|jsx|rb|go|rs|css|html?|svg|sql)\b"
1802)
1804#: ``module.symbol`` / ``Class.method`` — the dotted identifier a review writes
1805#: for something it is not calling (#1106), read only when the token carries a
1806#: mark prose does not use: an underscore, an internal capital, a run of
1807#: capitals, or a capitalised segment.
1808#:
1809#: **A dot is not evidence, and neither is a capital.** Requiring the mark
1810#: stops ``github.com`` and ``pypi.org``. It does not stop ``GitHub.com``,
1811#: ``GitLab.com``, ``OpenAI.com`` or ``SourceForge.net``, because a capitalised
1812#: first segment is precisely the mark ``Config.parse`` carries — the two are
1813#: one shape, and no sixth character class tells them apart. That is why this
1814#: form corroborates rather than anchors (:data:`_VERDICT_CORROBORATORS`): the
1815#: alternative was a list of hostnames to refuse, which cannot be finished,
1816#: because anyone can register the next one.
1817#:
1818#: Both sides of the dot still need two characters, so "e.g." and "i.e." are not
1819#: identifiers whatever else they carry.
1820_VERDICT_DOTTED_SYMBOL = re.compile(
1821 r"\b(?=[\w.]*(?:_|[a-z0-9][A-Z]|[A-Z]{2}|[A-Z][a-z]))"
1822 r"[A-Za-z_]\w+(?:\.[A-Za-z_]\w+)+"
1823)
1825#: A **bare** identifier: ``cache_key``, ``_prompt_mode``, ``__post_init__``.
1826#:
1827#: **The joining underscore is the whole rule** — two lowercase alphanumeric
1828#: segments with an underscore between them. Lowercase deliberately: `My_Thing`
1829#: is prose with a connector, and CamelCase is read only when dotted or
1830#: backticked, for the reason :data:`_VERDICT_DOTTED_SYMBOL` gives.
1831#:
1832#: Underscores that merely wrap a name are
1833#: decoration, so a lone ``_private`` or ``__dunder__`` is *not* read; it is
1834#: ``post_init`` inside ``__post_init__`` that matches.
1835#:
1836#: The joining underscore is a mark English prose and product names do not use,
1837#: and it is the only lexical difference between a symbol and a capitalised
1838#: noun: ``GitHub`` and ``GitLab`` are CamelCase in exactly the way
1839#: ``JuryConfig`` is, so reading CamelCase let "The GitHub and GitLab side of
1840#: this is unchanged" clear a floor of two while naming nothing in the change.
1841#: No pattern separates those two, and a list of product names to refuse would
1842#: need a new entry every time a reviewer mentions a new product. CamelCase
1843#: symbols are still read everywhere they are written with a dot or in
1844#: backticks, which is how a class is normally named.
1845_VERDICT_BARE_IDENTIFIER = re.compile(r"\b_*[a-z0-9]+(?:_[a-z0-9]+)+_*\b")
1847#: A concrete thing a review can point at, on its own — a ``path:line``, a
1848#: path, a backticked symbol, or a called identifier. Presence of *structure*,
1849#: never a judgement about whether the review was good — the same line
1850#: ai-jury's ``emitted_findings_block()`` draws.
1851#:
1852#: What these four have and :data:`_VERDICT_CORROBORATORS` do not is that the
1853#: **punctuation is the author pointing**. Backticks, a directory separator, a
1854#: ``:42`` and a ``()`` are marks a reviewer types deliberately at a thing;
1855#: none of them appear around a product name in ordinary prose. A bare dotted
1856#: token is just a token with a dot in it.
1857_VERDICT_ANCHORS = re.compile(
1858 r"[\w./-]+\.[A-Za-z0-9]{1,5}:\d+|" # path/to/file.py:42
1859 r"[\w-]+/[\w./-]+\.[A-Za-z0-9]{1,5}\b|" # src/keel/thing.py
1860 r"`[^`\n]{2,}`|" # `a_symbol`, `--a-flag`
1861 r"\b\w+\.\w+\(\)" # module.function()
1862)
1864#: The unbackticked forms — a bare filename, a bare dotted token, a bare
1865#: identifier — each of which is *also* how something that is not a symbol gets
1866#: written. Two of them anchor a verdict; one does not (#1106).
1867#:
1868#: **This is a shape, not a lexicon, and that is the point.** Three rounds of
1869#: widening the anchor set were each undone by a token that is not a symbol:
1870#: ``claude.ai`` in a tool footer, ``GitHub``/``GitLab`` as bare CamelCase,
1871#: then ``Node.js`` and ``GitHub.com``. Every one was answered by refining a
1872#: character class, and every refinement admitted the next such token, because
1873#: ``Node.js`` and ``evidence.py`` are the same shape and so are ``GitHub.com``
1874#: and ``Config.parse``. No character class ends that sequence and no blocklist
1875#: of products or hostnames can be finished. What separates a review from a
1876#: mention is not the spelling of one token but how many the verdict has: prose
1877#: mentions a product in passing; a review that walked the change names more
1878#: than one thing. A clause that says an act of review happened
1879#: (:data:`_VERDICT_CHECKED_CLAUSE`) is the one exception, and only for
1880#: "checked": see there for why the other verbs could not be given the same
1881#: latitude, and why judging their object by this same test made the branch
1882#: inert.
1883#:
1884#: Measured over every verdict posted across keel and ai-jury: corroboration
1885#: refuses all four ``*.js`` product names and all four capitalised hostnames,
1886#: costs 30 of the 141 verdicts #1106 recovered — 22 of those 30 being the
1887#: default template's "Scope reviewed:" line with a single token dropped into
1888#: it — and regresses nothing that passed before #1106. Dropping the two bare
1889#: forms outright instead, the fallback, refuses the same eight strings but
1890#: keeps only 98 of the 141 and loses a review naming ``evidence.py`` and
1891#: ``contracts.py``, which is a review.
1892_VERDICT_CORROBORATORS = (
1893 _VERDICT_SOURCE_FILE, # evidence.py, CHANGELOG.md
1894 _VERDICT_DOTTED_SYMBOL, # cache.cache_key, JuryConfig.__post_init__
1895 _VERDICT_BARE_IDENTIFIER, # collect_static_hints, _prompt_mode
1896)
1898#: How many distinct corroborating tokens stand in for an anchor. At one, the
1899#: corpus says the rule readmits 22 of the 75 ``Reviewed <title>: <affirmation>``
1900#: rubber stamps #926 is named for; at two it admits none of them.
1901_VERDICT_CORROBORATION_FLOOR = 2
1903#: What may sit between an act-of-review verb and its object: an optional colon,
1904#: and an optional break onto a bullet, because "Checked:\n- …" is the same
1905#: clause with a list under it and refusing it was punctuation pedantry (#1106).
1906#: Layout only — it says nothing about what the object has to be.
1907_VERDICT_CLAUSE_TAIL = r"[ \t]*:?[ \t]*(?:\r?\n[ \t]*[-*][ \t]*)?"
1909#: One sentence's worth of characters. A sentence ends at `.!?;` or an ellipsis
1910#: **followed by space or line end** — a period inside `evidence.py` is not a
1911#: sentence end, and the filename has to survive inside an object, so "Checked
1912#: evidence.py and contracts.py" is no longer truncated at the first dot —
1913#: the punctuation pedantry this change is about. Only
1914#: :data:`_VERDICT_CHECKED_CLAUSE` reads this; the second clause that did was
1915#: removed with the verb widening.
1916_VERDICT_SENTENCE = r"(?:[^.!?;\u2026\n]|[.!?;\u2026](?!\s|$))"
1918#: The escape hatch the issue insists on: a genuinely clean review must stay
1919#: expressible. "Checked X, Y and Z; found nothing" is a real review outcome and
1920#: must not be forced to invent an anchor. Its object stays free-form, as it has
1921#: been since #926: 35 verdicts in the corpus pass on this clause and nothing
1922#: else, saying things like "Checked the formula syntax, the version URL and the
1923#: checksum placeholder" — English objects, naming no symbol.
1924#:
1925#: The object now ends at a *sentence* rather than at any period, which is a
1926#: change from `[^.\n]{8,}`: it buys a filename, since `evidence.py` no longer
1927#: truncates to `evidence`, and it costs a trailing `;`, `!` or `?` clause —
1928#: "Checked config; found nothing of concern in the rest" passes on `main` and
1929#: is refused here. No verdict in the 1,421 measured is written that way, which
1930#: is why the corpus shows no regression; that is a weaker claim than "takes
1931#: nothing away" and is the one the evidence supports.
1932#:
1933#: **#1106 tried to widen this to traced/read/ran/inspected/verified and the
1934#: widening turned out inert.** Those verbs could not keep a free-form object —
1935#: "Read the whole diff and everything looks correct" is the #926 receipt with a
1936#: synonym at the front — so their object was made to name something. But
1937#: "names something" is the same test the whole prose already takes, and the
1938#: object is part of the prose, so a verdict that satisfied the clause had
1939#: always satisfied the anchor check first: across 1,421 verdicts the branch
1940#: decided **zero** of them. Dead code with a docstring explaining what it did
1941#: is worse than neither, so it is gone. Widening the vocabulary needs the
1942#: object to be judged by something other than the prose test, which #1106 did
1943#: not find.
1944_VERDICT_CHECKED_CLAUSE = re.compile(
1945 rf"\bchecked\b{_VERDICT_CLAUSE_TAIL}{_VERDICT_SENTENCE}{{8,}}", re.IGNORECASE
1946)
1948#: Below this share of novel words, the prose is the PR title said again. The
1949#: observed shape was `Reviewed <title>: <generic affirmation>` — 75 of 75
1950#: verdicts across 25 PRs (#926).
1951_VERDICT_NOVELTY_FLOOR = 0.35
1953_WORD = re.compile(r"[a-z0-9]+")
1956def _verdict_prose(body: str) -> str:
1957 """The verdict's own words: header block, marker line and HTML comments removed.
1959 The marker is matched as a *whole line*, never as a substring. It is rendered
1960 on its own bare line, so the header slice above has already dropped it; a
1961 substring test therefore only ever reached prose that *quotes* the marker,
1962 and deleted it. That cost the review of #1119 its entire scope — a
1963 1,700-character line naming four files was dropped because it named the
1964 marker it was documenting, and the verdict was then refused for naming
1965 nothing (#1120).
1967 This is the rule :func:`marker_in_header` already states earlier in this
1968 module: a marker below the header is prose, and ``MARKER in body`` cannot
1969 tell the two apart (#1026). This function was the one place that substring
1970 test survived.
1971 """
1972 lines = body.splitlines()
1973 start = 0
1974 for index, raw in enumerate(lines):
1975 if not raw.strip():
1976 start = index + 1
1977 break
1978 kept = [
1979 line
1980 for line in lines[start:]
1981 if line.strip()
1982 and not line.strip().startswith("<!--")
1983 and line.strip() != REVIEW_VERDICT_MARKER
1984 ]
1985 return "\n".join(kept)
1988def _verdict_corroborators(text: str) -> set[str]:
1989 """The distinct unbackticked tokens in ``text``, counting each one once.
1991 A written token is one piece of evidence however many patterns read it:
1992 ``cache.cache_key`` is matched by :data:`_VERDICT_DOTTED_SYMBOL` whole and by
1993 :data:`_VERDICT_BARE_IDENTIFIER` as its second half, and counting it twice
1994 would let one token clear a floor that exists to require two. Matches
1995 contained inside a longer match are therefore dropped, and what survives is
1996 deduplicated by text, so a reviewer who names ``evidence.py`` three times has
1997 still named one file.
1998 """
1999 spans = sorted(
2000 (
2001 (match.start(), match.end(), match.group(0))
2002 for pattern in _VERDICT_CORROBORATORS
2003 for match in pattern.finditer(text)
2004 ),
2005 key=lambda span: (span[0], -span[1]),
2006 )
2007 kept: list[tuple[int, int, str]] = []
2008 for span in spans:
2009 if any(span[0] >= start and span[1] <= end for start, end, _ in kept):
2010 continue
2011 kept.append(span)
2012 return {token for _, _, token in kept}
2015def _review_act_clause(prose: str) -> bool:
2016 """Whether ``prose`` says an act of review was performed on something."""
2017 return bool(_VERDICT_CHECKED_CLAUSE.search(prose))
2020def verdict_substance(body: str, *, pr_title: str = "") -> tuple[bool, str]:
2021 """Whether a verdict engages with the diff at all. ``(ok, reason)``.
2023 The evidence gate verified that verdicts *exist* with the right marker, head
2024 SHA and distinct reviewer ids — never that any of them looked at anything. A
2025 verdict engaging with nothing was indistinguishable from one that caught a
2026 blocker, and the record showed what that permits: 75 of 75 verdicts `pass`,
2027 all opening `Reviewed <PR title>: <affirmation>`, across 25 PRs that produced
2028 no review-driven commit between them (#926).
2030 Two mechanical requirements, both content-agnostic beyond structure:
2032 * **An anchor.** A ``path:line``, a path, a backticked symbol or a called
2033 identifier, whose punctuation is the author pointing — or, lacking one,
2034 two distinct unbackticked tokens, because ``Node.js`` and ``evidence.py``
2035 are one shape and only the count separates a mention from a review — or
2036 an act-of-review clause naming what was looked at, because a genuinely
2037 clean review must stay expressible and forcing it to invent a file
2038 reference would make the check worse than nothing.
2039 * **Novelty against the title.** Prose that is substantially the PR title
2040 restated is the observed shape, and it survives the anchor test whenever
2041 the title happens to contain a path.
2043 The two are independent on purpose, and that is what lets the anchor set be
2044 generous (#1106). An anchor asks whether the reviewer pointed at anything;
2045 the novelty floor asks whether the prose is the title said again. A verdict
2046 that names a symbol *and* is otherwise the title restated fails the second
2047 check, so widening the first cannot readmit the #926 shape by itself.
2049 This says nothing about whether a review was *good*. It cannot, and trying
2050 would make the gate a critic. It distinguishes a review from a receipt.
2051 """
2052 prose = _verdict_prose(body)
2053 if not prose.strip():
2054 return False, "verdict has no prose beyond its header"
2056 # ⚡ Bolt Optimization: Use combined compiled regex instead of generator overhead
2057 anchored = bool(_VERDICT_ANCHORS.search(prose)) or (
2058 len(_verdict_corroborators(prose)) >= _VERDICT_CORROBORATION_FLOOR
2059 )
2060 if not anchored and not _review_act_clause(prose):
2061 return False, (
2062 "verdict names nothing concrete — no file, line, symbol, no two "
2063 "of a filename/dotted name/identifier, and no 'checked …' clause, "
2064 "so it cannot be told apart from a receipt"
2065 )
2067 title_words = set(_WORD.findall(pr_title.lower()))
2068 prose_words = _WORD.findall(prose.lower())
2069 if title_words and prose_words:
2070 novel = [word for word in prose_words if word not in title_words]
2071 if len(novel) / len(prose_words) < _VERDICT_NOVELTY_FLOOR:
2072 return False, "verdict is substantially the pull request title restated"
2073 return True, ""
2076def _reviewer_key(item: dict[str, Any], body: str) -> str:
2077 fields = _fields(body)
2078 reviewer = fields.get("reviewer")
2079 if reviewer:
2080 return f"reviewer:{reviewer.lower()}"
2081 user = item.get("user")
2082 if isinstance(user, dict) and isinstance(user.get("login"), str) and user["login"]:
2083 return f"user:{user['login'].lower()}"
2084 digest = hashlib.sha256(body.encode("utf-8")).hexdigest()
2085 return f"body:{digest}"
2088def _matches_head(
2089 item: dict[str, Any],
2090 body: str,
2091 head_sha: str | None,
2092 covered_heads: Collection[str] = (),
2093) -> bool:
2094 """Does this comment answer for ``head_sha``? A blank head means *do not filter*.
2096 **Deliberately not :func:`keel.juryavail.is_pinnable_head`'s rule, and the difference
2097 is worth stating** (#1068). This one filters *evidence items* inside a gate that, with
2098 no head resolved, is head-agnostic from end to end — every review verdict counts, so
2099 holding jury verdicts alone to a head nobody knows would refuse a gate the rest of
2100 which is already unfiltered. Nothing reached through here removes a requirement:
2101 :func:`_review_evidence_keys` and :func:`_review_vendor_provenance` count verdicts
2102 towards one, and :func:`jury_panel_size` feeds ``max(declared, minimum_vendors)``, so a
2103 stale ``panelists:`` can only ever raise the bar.
2105 :func:`jury_participating_vendors` was the exception, and #1069 closed it rather than
2106 changing this predicate: its count *can* downgrade a gating jury to advisory (#1015), so
2107 it now asks :func:`keel.juryavail.is_pinnable_head` for itself before it reads anything.
2108 Its head rule is its own, for the reason a pin's is — it removes a requirement — and the
2109 two readers that do not remove one keep this reading. That is the whole distinction
2110 between the three panel-shaped readers layered here, and it is a property of what each
2111 one's answer can *do*, not of where it is read from.
2113 A *pin* is the case that cannot use this reading, because it does remove requirements —
2114 it takes ``review-verdict-1..3`` off the required set entirely. So the pin
2115 (:func:`keel.juryavail.pin`, read by :func:`keel.cli._shipped_jury_availability`)
2116 refuses a blank head before :func:`panel_verdict_posted` is asked at all, rather than
2117 this predicate changing under the surfaces that need the permissive one.
2118 """
2119 if not head_sha:
2120 return True
2121 # **A head the current one descends from by capture commits alone** answers for it
2122 # too (#1203). The learning is the pull request's last commit, written after review
2123 # and before the merge, so every verdict would otherwise be pinned to a head the
2124 # branch has already moved past. `covered_heads` is never filled from a comment or an
2125 # argument an agent supplies: its only producer is `capture.capture_only_descent`,
2126 # which holds each commit in between to one parent, the landing marker, and exactly
2127 # one path inside the configured sink — so what it admits is a lesson, never code.
2128 accepted = {head_sha, *covered_heads}
2129 fields = _fields(body)
2130 recorded = fields.get("head")
2131 if recorded:
2132 return recorded in accepted
2133 commit_id = item.get("commit_id")
2134 return isinstance(commit_id, str) and commit_id in accepted
2137def _fields(body: str) -> dict[str, str]:
2138 """Parse the header block at the top of a verdict comment, and only that.
2140 Scanning stops at the first line that is not part of a header block —
2141 whether or not a header has been seen yet. #868's second requirement said so
2142 and only the first shipped: the earlier version `continue`d past prose and
2143 blank lines until it found something header-shaped, so fields could be
2144 harvested from anywhere in a comment (#932):
2146 "Some prose line here.\\n\\nhead: 0000000\\nvendor: spoofed\\n"
2147 -> {'head': '0000000', 'vendor': 'spoofed'}
2149 Reachable through :func:`_reviewer_key`, which calls this with no marker
2150 requirement, so a comment whose *prose* contains ``reviewer: someone`` was
2151 keyed to that reviewer. Narrow — it needs a trusted author — and not a live
2152 hole, but it is the residual of the class #868 named, and stopping costs the
2153 real comment format nothing: a verdict's header is its first block.
2154 """
2155 fields: dict[str, str] = {}
2156 started = False
2158 for raw_line in (body or "").splitlines():
2159 line = raw_line.strip()
2160 if not line:
2161 # Leading blank lines are skipped, not treated as the end: a comment
2162 # body routinely begins with a newline, and breaking there would
2163 # reject legitimate verdicts. Once the block has started, a blank
2164 # line ends it — that is the boundary #868 asked for.
2165 if started:
2166 break
2167 continue
2168 started = True
2169 # A marker-only line is the artifact's own header, not a field: skip it and
2170 # keep reading. A line that merely *mentions* a marker is prose, and prose
2171 # ends the block — the #932 boundary this parser exists to hold.
2172 if (
2173 line.startswith("<!--") and line.endswith("-->")
2174 ) or _CLASSIFICATION_MARKERS_SET.issuperset(line.split()):
2175 continue
2176 match = _FIELD_RE.match(line)
2177 if match:
2178 key = match.group("key").lower()
2179 if key not in fields:
2180 fields[key] = match.group("value")
2181 elif not _HEADER_LINE_RE.match(line):
2182 break
2183 return fields
2186def _is_review_verdict_body(body: str) -> bool:
2187 """Whether ``body`` is a review verdict, decided by its header alone (#1026).
2189 The jury and closure exclusions are no longer separate substring tests:
2190 :func:`marker_in_header` yields at most one marker, so a body anchored to the
2191 jury or closure marker simply is not a review verdict, and a review verdict
2192 that *mentions* either one in its prose still is.
2193 """
2194 if _is_ship_assessment(body):
2195 return False
2196 return marker_in_header(body) == REVIEW_VERDICT_MARKER
2199def _has_trusted_review_marker(items: list[dict[str, Any]]) -> bool:
2200 return any(
2201 _is_trusted_source(item, enforced=True)
2202 and marker_in_header(_body(item)) == REVIEW_VERDICT_MARKER
2203 for item in items
2204 )
2207def jury_participating_vendors(
2208 pr_comments: list[dict[str, Any]] | None = None,
2209 pr_reviews: list[dict[str, Any]] | None = None,
2210 *,
2211 head_sha: str | None = None,
2212 covered_heads: Collection[str] = (),
2213 enforced: bool = True,
2214) -> int | None:
2215 """Return the panel size declared by a posted jury verdict, or ``None``.
2217 Reads the ``vendors: <N>`` field from a trusted, head-bound
2218 ``keel.jury-verdict.v1`` comment. This is how the participating-vendor count
2219 reaches a CI evidence check: the run ledger and the jury artifact both live
2220 under the gitignored ``.keel/state/``, so a hosted runner cannot read either,
2221 but PR comments are always visible.
2223 ``None`` means "not declared" — no jury verdict posted, or one that predates
2224 the field — and leaves the jury mode untouched rather than assuming a short
2225 panel. Only a verdict that actually states the count may relax the gate.
2227 When several verdicts qualify, the largest declared count wins: a re-post
2228 correcting an earlier partial run should not be capped by the stale one.
2230 **This one reader is held to an exact head, and its two siblings are not** (#1069).
2231 Alone among the three panel-shaped readers here, this count can *remove* a
2232 requirement: below ``jury.min_vendors`` it downgrades a gating jury to advisory
2233 (:func:`keel.ship.resolve_jury`), which drops ``jury-verdict`` from the required
2234 evidence entirely. So it asks :func:`keel.juryavail.is_pinnable_head` — the same
2235 blank-head predicate the panel pins ask — before it reads a comment at all, and a
2236 run that resolved no head declares nothing rather than inheriting the last verdict
2237 on the pull request. Without the guard, `keel evidence-verify` run offline with no
2238 ``--head-sha`` (its documented default) read a ``vendors: 1`` verdict posted against
2239 an earlier head as this head's and relaxed the gate: a requirement removed by
2240 evidence nobody re-checked. A *mismatched* head was already refused, by
2241 :func:`_matches_head` — that predicate is exact once a head is known, and permissive
2242 only when none is — so the guard closes the blank-head half and nothing else.
2244 :func:`jury_panel_size` and :func:`panel_verdict_posted` deliberately keep the
2245 permissive reading, because neither can relax anything: the first feeds
2246 ``max(declared, minimum_vendors)`` and can only raise the bar, and the second is
2247 already refused on a blank head by its caller, which owns the pin order.
2248 """
2249 if not juryavail.is_pinnable_head(head_sha):
2250 return None
2251 counts = [
2252 parsed
2253 for item in [*(pr_comments or []), *(pr_reviews or [])]
2254 if _is_jury_verdict(item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced)
2255 if (parsed := _parse_vendor_count(_fields(_body(item)).get("vendors"))) is not None
2256 ]
2257 return max(counts) if counts else None
2260def jury_panel_size(
2261 pr_comments: list[dict[str, Any]] | None = None,
2262 pr_reviews: list[dict[str, Any]] | None = None,
2263 *,
2264 head_sha: str | None = None,
2265 covered_heads: Collection[str] = (),
2266 enforced: bool = True,
2267) -> int | None:
2268 """Return the panel size declared by a posted jury verdict, or ``None`` (#1015).
2270 Reads the ``panelists: <N>`` field off the same comment
2271 :func:`jury_participating_vendors` reads ``vendors:`` from, and for the same
2272 reason: when the jury **is** the review panel, the number of ballots is the
2273 reviewer count the evidence gate must require, and a hosted runner can read
2274 it from nowhere else.
2276 ``None`` means "not declared", which leaves the gate on the contract's floor
2277 rather than requiring nothing. The largest declared count wins, so a re-post
2278 that completes a partial panel raises the requirement instead of being capped
2279 by the stale verdict — the direction that fails closed.
2281 **It does not share that sibling's head rule, and the asymmetry is deliberate**
2282 (#1069). This count reaches :func:`keel.ship._jury_panel_size`, which answers
2283 ``max(declared, minimum_vendors)`` — so a stale or blank-head ``panelists:`` can
2284 only ever *raise* what the tier owes, never remove a requirement. Holding it to an
2285 exact head would refuse a bar-raising reading inside a gate whose other half, with
2286 no head resolved, is already unfiltered (:func:`_matches_head`). ``vendors:`` is the
2287 one that can relax, so ``vendors:`` is the one that is pinned.
2288 """
2289 counts = [
2290 parsed
2291 for item in [*(pr_comments or []), *(pr_reviews or [])]
2292 if _is_jury_verdict(item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced)
2293 if (parsed := _parse_vendor_count(_fields(_body(item)).get("panelists"))) is not None
2294 ]
2295 return max(counts) if counts else None
2298def panel_verdict_posted(
2299 pr_comments: list[dict[str, Any]] | None = None,
2300 pr_reviews: list[dict[str, Any]] | None = None,
2301 *,
2302 head_sha: str | None = None,
2303 covered_heads: Collection[str] = (),
2304 enforced: bool = True,
2305) -> bool:
2306 """Is a head-pinned jury verdict already on this pull request? (#1066)
2308 Proof that the panel *sat*, from the one place a bare CI runner can read it: the run
2309 ledger and the jury artifact both live under the gitignored ``.keel/state/``, while PR
2310 comments are always visible. A verification surface uses it to pin the contract to what
2311 the ship measured rather than re-measuring the panel on its own machine. It is the
2312 *weaker* of the two pins and speaks only when the run left no ledger record for this
2313 head; :func:`keel.juryavail.pin` owns that order and is the one place it is written.
2315 Distinct from :func:`jury_panel_size`, which answers *how many* ballots and is ``None``
2316 for a verdict predating the ``panelists:`` field. Presence is the weaker question, and
2317 the one that must not depend on an optional field.
2319 **Call this only with a head you actually resolved.** Like every reader here it goes
2320 through :func:`_matches_head`, which reads a blank ``head_sha`` as "do not filter" —
2321 right for counting evidence, wrong for a pin, which is why the caller refuses a blank
2322 head first (:func:`keel.juryavail.is_pinnable_head`).
2324 :func:`jury_participating_vendors` asks that same predicate *itself* rather than
2325 leaving it to a caller (#1069), and the difference is about who the callers are: this
2326 one is read from exactly one place, :func:`keel.juryavail.pin`, which owns the pin
2327 order and so is the right place for the rule; the vendor count is read straight off
2328 ``keel evidence-verify``'s argv-derived head, where there is no single owner to put it
2329 in front of.
2330 """
2331 return any(
2332 _is_jury_verdict(item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced)
2333 for item in [*(pr_comments or []), *(pr_reviews or [])]
2334 )
2337def shipped_panel_decision(
2338 pr_comments: list[dict[str, Any]] | None = None,
2339 *,
2340 head_sha: str | None = None,
2341 enforced: bool = True,
2342) -> str | None:
2343 """The panel decision **this run** recorded, read back off its closure comment (#1068).
2345 The middle pin, and the one that makes the strongest pin work anywhere. The run's own
2346 ``ship_run`` ledger record outranks a posted jury verdict — a comment records what
2347 somebody put on the pull request, the ledger records what the run *did* — but the
2348 ledger lives under the gitignored ``.keel/state/``, so on a hosted ``evidence-verify``
2349 or ``merge`` there is no record to read and that precedence held on the shipping
2350 workstation and nowhere else. A leftover or collaborator-posted ``keel.jury-verdict.v1``
2351 then answered for a run that had fallen back, and took ``review-verdict-1..3`` off the
2352 required set.
2354 The run's decision is already on the pull request: s11 posts the closure comment keel
2355 renders from that same ledger record, and since #1068 round 6 it carries
2356 :data:`keel.closure.JURY_PANEL_MARKER` beside the human ``Jury panel:`` line. So this
2357 reads the run's own statement from the one place that travels with the pull request.
2359 Three conditions, and each is the same rule its siblings hold to:
2361 * **Trusted author only** (:func:`_is_trusted_source`). keel posts the closure comment
2362 on the operator's behalf, which is exactly the authority a posted jury verdict has —
2363 no more. An untrusted author must not be able to relax the contract *in either
2364 direction*: neither to claim a fallback that drops the panel item, nor to claim the
2365 panel sat.
2366 * **An actual closure comment** (:func:`_has_closure_marker`), so the marker counts only
2367 inside the artifact that renders it — a reviewer quoting the marker while describing
2368 this change is prose, the #1026 rule every marker here is read under.
2369 * **Pinned to this head.** The marker names the head its record was written for and it
2370 must be the head under verification, because a pull request outlives its heads and a
2371 pin removes requirements.
2373 **The latest such comment is the answer, not the first** (#1068 round 7). One head can
2374 be shipped more than once — a re-run, a force-push back onto the same commit, a second
2375 ship on a different machine — and each ship posts its own closure comment. ``pr_comments``
2376 arrives in GitHub's order, oldest first, so this walks the whole list and keeps the last
2377 match: the newest statement wins, which is the same direction
2378 :func:`keel.ledger.latest_ship_run_for_pr` selects the ledger record in, and
2379 :func:`keel.juryavail.pin` ranks the two sources on the premise that they agree about it.
2380 Returning the first match meant an older ``decision=fallback`` outranked the panel-sat
2381 ship that followed it — and round 6 emitted no marker at all for a panel that sat, so the
2382 later run had nothing to outrank the older one *with*. :func:`keel.closure._jury_panel`
2383 now renders ``decision=available`` too, which is what makes last-wins well defined here.
2385 ``None`` for everything else — no comment, an older head, a marker keel did not write —
2386 and ``None`` means *this source is silent*, never a waiver: :func:`keel.juryavail.pin`
2387 then goes on to the posted verdict exactly as it did before.
2388 """
2389 if not head_sha:
2390 return None
2391 latest: str | None = None
2392 for item in pr_comments or []:
2393 if not _is_trusted_source(item, enforced=enforced):
2394 continue
2395 body = _body(item)
2396 if not _has_closure_marker(body):
2397 continue
2398 decision = _jury_panel_decision(body, head_sha)
2399 if decision is not None:
2400 latest = decision
2401 return latest
2404def _jury_panel_decision(body: str, head_sha: str) -> str | None:
2405 """The decision ``body``'s panel marker records for ``head_sha``, or ``None``.
2407 A token parser over one HTML-comment line, not a regex over Markdown: the line is
2408 :func:`keel.closure._jury_panel_marker`'s exact render, so it unwraps with the same
2409 :func:`_unwrap_html_comment` a header marker does and splits into
2410 ``<marker> head=<sha> decision=<value>``. A line that is not that shape is prose and is
2411 skipped, which is why the human sentence above it — which *names* neither field — can
2412 never be mistaken for the record.
2413 """
2414 for raw_line in (body or "").splitlines():
2415 tokens = _unwrap_html_comment(raw_line.strip()).split()
2416 if not tokens or tokens[0] != closure.JURY_PANEL_MARKER:
2417 continue
2418 fields: dict[str, str] = {}
2419 for token in tokens[1:]:
2420 key, _, value = token.partition("=")
2421 # First wins, the convention `_fields` already reads headers under, and a
2422 # token carrying no `=` becomes a valueless key that matches neither field.
2423 fields.setdefault(key, value)
2424 if fields.get("head") == head_sha:
2425 return fields.get("decision")
2426 return None
2429def _parse_vendor_count(raw: str | None) -> int | None:
2430 """Parse a declared vendor count, rejecting anything not a plain non-negative int."""
2431 if raw is None:
2432 return None
2433 try:
2434 parsed = int(raw)
2435 except ValueError:
2436 return None
2437 return parsed if parsed >= 0 else None
2440def _is_jury_verdict(
2441 item: dict[str, Any],
2442 *,
2443 head_sha: str | None = None,
2444 covered_heads: Collection[str] = (),
2445 enforced: bool = True,
2446) -> bool:
2447 """Whether ``item`` is a trusted jury verdict pinned to ``head_sha`` — presence only.
2449 **Deliberately blind to the consensus** (#1429). Its readers here are the panel-shape
2450 checks — :func:`jury_participating_vendors`, :func:`jury_panel_size` and
2451 :func:`panel_verdict_posted` — which size the panel and pin whether it sat. A panel
2452 that sat and rejected the change still sat, with that many vendors and ballots, so a
2453 rejecting verdict must keep answering them: reading the consensus there would turn a
2454 rejection into "no panel", and drop the ballots it owes. Whether the panel *approves*
2455 is a separate question, answered by :func:`_standing_jury_verdict` and
2456 :func:`jury_verdict_approves` for the ``jury-verdict`` requirement.
2457 """
2458 if not _is_trusted_source(item, enforced=enforced):
2459 return False
2460 body = _body(item)
2461 if _is_ship_assessment(body):
2462 return False
2463 return marker_in_header(body) == JURY_VERDICT_MARKER and _matches_head(
2464 item, body, head_sha, covered_heads
2465 )
2468def _standing_jury_verdict(
2469 items: list[dict[str, Any]],
2470 *,
2471 head_sha: str | None = None,
2472 covered_heads: Collection[str] = (),
2473 enforced: bool = True,
2474) -> dict[str, Any] | None:
2475 """The latest trusted jury verdict answering for ``head_sha``, or ``None`` (#1429).
2477 **The latest jury verdict is the panel's word**, ordered as :func:`_review_tally`
2478 orders review verdicts: by when it was posted (:func:`_posted_at` — ``created_at``,
2479 never ``updated_at``), then by position. A panel re-run on the same head that now
2480 approves supersedes the rejection before it, and one that now rejects supersedes the
2481 approval. A jury verdict pinned to an older head is not read: the head that moved
2482 answers it.
2483 """
2484 standing: dict[str, Any] | None = None
2485 for _, item in sorted(enumerate(items), key=lambda pair: (_posted_at(pair[1]), pair[0])):
2486 if _is_jury_verdict(
2487 item, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced
2488 ):
2489 standing = item
2490 return standing
2493def standing_jury_verdict(
2494 pr_comments: list[dict[str, Any]] | None,
2495 *,
2496 head_sha: str | None,
2497 covered_heads: Collection[str] = (),
2498 enforced: bool = True,
2499) -> dict[str, Any] | None:
2500 """The jury verdict comment that stands for ``head_sha``, or ``None`` (#1437).
2502 The merge gate's own reading (:func:`_standing_jury_verdict`): trusted author, the
2503 jury marker in the header, pinned to the head or a head it covers, latest by when it
2504 was posted. ``keel ship`` reuses the panel this names instead of convening another,
2505 so the two can never pick different panels. A blank head is refused here, unlike
2506 the evidence counters: reuse *replaces* a run, and a panel nobody pinned to a head
2507 must not stand in for one.
2508 """
2509 if not head_sha:
2510 return None
2511 return _standing_jury_verdict(
2512 pr_comments or [], head_sha=head_sha, covered_heads=covered_heads, enforced=enforced
2513 )
2516def verdict_head(item: dict[str, Any]) -> str:
2517 """The head a verdict comment answers for: its ``head:`` field, else its commit."""
2518 return _verdict_head(item, _body(item))
2521def _jury_not_approved_finding(
2522 items: list[dict[str, Any]],
2523 *,
2524 head_sha: str | None,
2525 covered_heads: Collection[str] = (),
2526 enforced: bool,
2527 blocking: bool,
2528) -> dict[str, Any] | None:
2529 """The finding for a standing jury verdict whose consensus does not approve (#1429).
2531 ``major`` when ``blocking`` — the ``jury-verdict`` item is required and not deferred —
2532 so a rejecting panel **holds** the merge with its consensus and head named
2533 (``jury-verdict-not-approved: the jury's consensus at <head> is REQUEST_CHANGES, not an
2534 approval.`` in :func:`refusal_reason`). Merely leaving it uncounted would read as
2535 *missing* and send the operator looking for a comment that is on the pull request.
2536 ``minor`` otherwise: an advisory panel's rejection is said, never gated on.
2537 """
2538 standing = _standing_jury_verdict(
2539 items, head_sha=head_sha, covered_heads=covered_heads, enforced=enforced
2540 )
2541 if standing is None:
2542 return None
2543 body = _body(standing)
2544 token = jury_verdict_token(body)
2545 if token in APPROVING_VERDICTS:
2546 return None
2547 head = _verdict_head(standing, body)
2548 message = (
2549 f"the jury's consensus at {head} is {token}, not an approval."
2550 if token
2551 else f"the jury verdict at {head} has no readable AI Jury verdict line."
2552 )
2553 return {
2554 "id": JURY_NOT_APPROVED_FINDING,
2555 "severity": "major" if blocking else "minor",
2556 "kind": "jury",
2557 "message": message,
2558 }