Coverage for src/keel/jury.py: 100%

286 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""The ``jury`` built-in gate — run the ai-jury CLI on the diff at s8. 

2 

3keel does **not** depend on ai-jury. If the ``jury`` CLI is on PATH, this gate runs it on 

4the change's diff and maps its findings into keel :class:`~keel.findings.Finding`s. Without 

5the ``jury`` binary the s8 run is a no-op (reported ``SKIPPED``, and blocking when ``jury`` is 

6the only gate planned), but a tier-3 merge still requires a 

7``jury-verdict`` unless the run passes ``--no-jury``; it relaxes to advisory only when a 

8posted verdict (or ``--jury-vendors``) reports fewer than 2 vendors. That is the default 

9policy: off a jury-panel tier, ``team.jury.mode: advisory`` or ``--jury-advisory`` never 

10requires the verdict and ``team.jury.min_vendors`` may raise the 2; on a tier whose review is 

11the jury panel, no flag or short panel relaxes it, and only a probe that finds the panel 

12unstaffable turns that tier's jury off, under ``team.jury.on_unavailable: fallback`` (the 

13default). That requirement is not decided here: :func:`keel.ship.resolve_jury` resolves the 

14mode, and :func:`keel.evidence.required_items` demands the verdict from it. Parsing is pure and 

15unit-tested; the subprocess is behind the injectable ``_run`` seam. 

16""" 

17 

18from __future__ import annotations 

19 

20import json 

21import os 

22import tempfile 

23from dataclasses import dataclass 

24from typing import Any 

25 

26from . import artifacts, evidence 

27from .findings import Finding 

28from .model import DEFAULT_JURY_TIMEOUT_S 

29from .runner import CommandResult, run_argv 

30 

31#: ai-jury severities → keel severities (unknown ⇒ ``minor``). 

32_SEVERITY = { 

33 "critical": "critical", 

34 "blocker": "critical", 

35 "major": "major", 

36 "minor": "minor", 

37 "nit": "nit", 

38 "info": "nit", 

39 "note": "nit", 

40} 

41 

42MAX_DIFF_BYTES = 1_000_000 

43 

44 

45def map_severity(severity: str) -> str: 

46 """Map an ai-jury severity onto a keel severity (default ``minor``).""" 

47 return _SEVERITY.get((severity or "").strip().lower(), "minor") 

48 

49 

50def parse_report(data: dict | str) -> list[Finding] | None: 

51 """Map an ai-jury JSON report into Findings, or ``None`` if it is not a report. 

52 

53 The ``None`` return is the point: it separates *"the panel reviewed the diff and 

54 found nothing"* from *"this output is not a verdict at all"*, which 

55 :func:`parse_findings` collapses into the same empty list. Only the caller that 

56 decides whether a gate passed needs that distinction — see :func:`run_gate`. 

57 

58 Tolerates trailing non-JSON. :func:`keel.runner.run_argv` hands back 

59 ``stdout + stderr`` concatenated, and ai-jury logs its progress to stderr, so a 

60 real report is followed by ``[jury] …`` lines. A strict ``json.loads`` rejects the 

61 whole thing and silently loses every finding. 

62 """ 

63 if isinstance(data, str): 

64 try: 

65 data, _end = json.JSONDecoder().raw_decode(data.lstrip()) 

66 except json.JSONDecodeError: 

67 return None 

68 if not isinstance(data, dict) or "findings" not in data: 

69 return None 

70 return _findings_from(data) 

71 

72 

73def parse_findings(data: dict | str) -> list[Finding]: 

74 """Map an ai-jury JSON report (dict or raw string) into keel Findings. 

75 

76 Unparseable input yields ``[]``. Use :func:`parse_report` when the difference 

77 between "no findings" and "no report" matters. 

78 """ 

79 return parse_report(data) or [] 

80 

81 

82def _findings_from(data: dict) -> list[Finding]: 

83 out: list[Finding] = [] 

84 for f in data.get("findings") or []: 

85 path = f.get("file") 

86 line = f.get("line") 

87 line = line if isinstance(line, int) else None 

88 out.append( 

89 Finding( 

90 severity=map_severity(f.get("severity", "")), 

91 message=f.get("claim") or "(jury finding)", 

92 source=f"jury:{f.get('reviewer') or 'consensus'}", 

93 path=path, 

94 line=line, 

95 anchorable=bool(path) and line is not None, 

96 ) 

97 ) 

98 return out 

99 

100 

101def _kw(_run): 

102 return {"_run": _run} if _run is not None else {} 

103 

104 

105def available(*, cwd: str | None = None, _run=None) -> bool: 

106 """True if the ``jury`` CLI is callable.""" 

107 return run_argv(["jury", "--version"], cwd=cwd, timeout=30, **_kw(_run)).ok 

108 

109 

110def _incomplete_finding(result: CommandResult, *, timeout: int, severity: str = "nit") -> Finding: 

111 """Record that the jury CLI ran but produced no verdict. 

112 

113 A timeout, or a nonzero exit whose output carries no parseable findings, means the 

114 panel never reached a conclusion. That is emphatically **not** a clean pass — it is 

115 the *absence* of a review — so in gating mode it fails closed exactly as an oversize 

116 diff does. The timeout case is named apart from a crash so the operator can tell a 

117 slow panel from a broken one. 

118 """ 

119 if result.timed_out: 

120 detail = ( 

121 f"timed out after {timeout}s; no verdict was produced. Raise " 

122 "knobs.jury_timeout_s if the panel legitimately needs longer" 

123 ) 

124 else: 

125 detail = f"exited {result.code} without a parseable verdict; the panel did not complete" 

126 return Finding( 

127 severity=severity, 

128 message=f"jury run incomplete: the jury CLI {detail}.", 

129 source="jury:incomplete-run", 

130 path=None, 

131 line=None, 

132 anchorable=False, 

133 ) 

134 

135 

136def _unreadable_diff_finding(*, severity: str = "minor") -> Finding: 

137 """Record that the diff itself could not be read, so no review was possible.""" 

138 return Finding( 

139 severity=severity, 

140 message=( 

141 "jury could not run: the diff could not be read from git (is the base " 

142 "branch fetched locally? a shallow or single-branch clone cannot " 

143 "resolve base...HEAD). No review was performed." 

144 ), 

145 source="jury:unreadable-diff", 

146 path=None, 

147 line=None, 

148 anchorable=False, 

149 ) 

150 

151 

152def _oversize_finding(size: int, *, severity: str = "nit") -> Finding: 

153 """Record that the jury gate skipped an oversize diff. 

154 

155 Advisory jury mode keeps the finding non-blocking (``nit``). Gating jury mode 

156 escalates it to ``major`` so an oversize diff cannot bypass the blocking 

157 cross-vendor review gate. 

158 """ 

159 return Finding( 

160 severity=severity, 

161 message=( 

162 f"jury skipped: diff is {size} bytes, over the {MAX_DIFF_BYTES}-byte " 

163 "limit (ai-jury large-diff chunking not applied)" 

164 ), 

165 source="jury:skipped-oversize", 

166 path=None, 

167 line=None, 

168 anchorable=False, 

169 ) 

170 

171 

172#: The source of the finding a jury gate that **could not run** reports (#1369): no ``jury`` 

173#: CLI on the host, or an empty diff. It judged nothing, so it must never read as ``ok``. 

174#: :func:`could_not_run` reads it back; :func:`keel.gates.lone_jury_cannot_judge` turns it 

175#: into a blocking outcome when the jury is the only gate planned. 

176NOT_RUN_SOURCE = "jury:not-run" 

177 

178#: Why the jury could not run, as the finding says it. 

179NOT_RUN_NO_CLI = "the jury CLI is not available (`jury --version` failed; install ai-jury)" 

180NOT_RUN_EMPTY_DIFF = "the diff against the base branch is empty" 

181 

182 

183def _not_run_finding(reason: str) -> Finding: 

184 """Record that the jury gate judged nothing, and why (#1369). 

185 

186 ``nit``: beside another gate that judges, a jury that could not run stays the 

187 documented s8 no-op and does not hold the merge. It is reported rather than silent, 

188 and it is what marks the outcome ``SKIPPED`` rather than ``ok``. 

189 """ 

190 return Finding( 

191 severity="nit", 

192 message=f"jury did not run: {reason}; nothing was judged.", 

193 source=NOT_RUN_SOURCE, 

194 path=None, 

195 line=None, 

196 anchorable=False, 

197 ) 

198 

199 

200#: The source of the finding a jury gate reports for the panel's **consensus** (#1436). 

201CONSENSUS_SOURCE = "jury:consensus" 

202 

203 

204def panel_consensus(data: dict | str) -> str | None: 

205 """The panel's consensus in an ai-jury report, in keel's vocabulary, or ``None``. 

206 

207 The consensus is the **chair record's** ``verdict`` in the report's ``reviewers`` 

208 array (``role: chair``). ai-jury writes it from the chair's synthesis headline, or 

209 from the panel vote when the run is configured with ``decision: vote`` 

210 (``ai_jury.ballots.chair_verdict``), so one field carries both. It is read through 

211 :func:`map_verdict`, so ``APPROVE`` / ``READY`` arrive as ``LGTM``. 

212 

213 A ballot report with **no chair record** — the synthesis failed — has no consensus, and 

214 reads ``ABSTAIN``, which is what :func:`jury_verdict` posts for it too. ``None`` means 

215 the report carries **no ballots at all**: an ai-jury from before report schema 1.1, 

216 which has no ``reviewers`` array, or one whose ballots are malformed. 

217 """ 

218 try: 

219 panel = parse_panel(data) 

220 except JuryReportError: 

221 return None 

222 if panel is None: 

223 return None 

224 return panel.chair.verdict if panel.chair is not None else "ABSTAIN" 

225 

226 

227def _consensus_finding(data: dict | str, *, gating: bool) -> Finding | None: 

228 """The finding a jury gate reports when the panel's consensus does not approve (#1436). 

229 

230 Read against :data:`keel.evidence.APPROVING_VERDICTS` by the same reader the 

231 evidence gate applies to the posted ``AI Jury verdict:`` line 

232 (:func:`keel.evidence.jury_verdict_approves`), so the gates-pass and the evidence 

233 gate cannot disagree about what the panel said. ``major`` in gating mode — the gate 

234 fails — and ``minor`` in advisory mode, which reports a rejection and never gates on 

235 one, as the evidence gate does. 

236 

237 **A report with no consensus fails closed in gating mode, and is silent in advisory 

238 mode.** A gating jury's verdict comment has to approve before ``keel merge`` will land 

239 the change, and a report that states no consensus cannot truthfully produce one; a 

240 gates-pass for it would certify a review that never concluded. Advisory mode keeps 

241 today's behaviour: the severity rule alone, with nothing added. 

242 """ 

243 consensus = panel_consensus(data) 

244 if consensus is None: 

245 if not gating: 

246 return None 

247 return Finding( 

248 severity="major", 

249 message=( 

250 "jury report states no panel consensus: it carries no readable `reviewers` " 

251 "ballots with a chair record (ai-jury before report schema 1.1, or a " 

252 "malformed report). A gating jury must conclude; upgrade ai-jury." 

253 ), 

254 source=CONSENSUS_SOURCE, 

255 path=None, 

256 line=None, 

257 anchorable=False, 

258 ) 

259 line = f"AI Jury verdict: {consensus}." 

260 if evidence.jury_verdict_approves(line): 

261 return None 

262 token = evidence.jury_verdict_token(line) or consensus 

263 return Finding( 

264 severity="major" if gating else "minor", 

265 message=f"jury consensus is {token}, not an approval.", 

266 source=CONSENSUS_SOURCE, 

267 path=None, 

268 line=None, 

269 anchorable=False, 

270 ) 

271 

272 

273#: The source of the findings a jury gate carries over from a reused panel (#1437). 

274REUSED_SOURCE = "jury:reused" 

275 

276 

277def reuse_posted_verdict(body: str, *, gating: bool) -> tuple[bool, list[Finding]] | None: 

278 """The jury gate's outcome read off the panel already posted for the head (#1437). 

279 

280 ``body`` is the standing ``keel.jury-verdict.v1`` comment for the head 

281 (:func:`keel.evidence.standing_jury_verdict`). The gate is judged by the two rules a 

282 panel keel ran is judged by, so reusing one can never be kinder than running it: 

283 

284 * **the severity rule** — each ``<severity>: <message>`` item of the comment's findings 

285 summary (:func:`keel.artifacts.jury_verdict_summary`, the verified findings 

286 :func:`jury_verdict` posts) becomes a finding, and a critical or major one fails; 

287 * **the consensus rule** (#1436) — an ``AI Jury verdict:`` line that does not approve, 

288 or no readable line at all, is a ``major`` finding in gating mode and a ``minor`` one 

289 in advisory mode, the same as :func:`_consensus_finding` for a report. 

290 

291 ``(ok, findings)``, or ``None`` when the comment's summary cannot be read — a body keel 

292 did not render. ``None`` means *do not reuse*: the caller convenes the panel as it 

293 always did, so an unreadable comment can never stand in for a run. 

294 """ 

295 summary = artifacts.jury_verdict_summary(body) 

296 if summary is None: 

297 return None 

298 findings: list[Finding] = [] 

299 for item in summary: 

300 label, sep, _message = item.partition(":") 

301 findings.append( 

302 Finding( 

303 # An item with no `<severity>:` label is read as unknown, which 

304 # map_severity already maps to `minor`. 

305 severity=map_severity(label if sep else ""), 

306 message=f"posted jury finding — {item}", 

307 source=REUSED_SOURCE, 

308 path=None, 

309 line=None, 

310 anchorable=False, 

311 ) 

312 ) 

313 if not evidence.jury_verdict_approves(body): 

314 token = evidence.jury_verdict_token(body) 

315 findings.append( 

316 Finding( 

317 severity="major" if gating else "minor", 

318 message=( 

319 f"posted jury consensus is {token}, not an approval." 

320 if token 

321 else "posted jury verdict has no readable AI Jury verdict line." 

322 ), 

323 source=CONSENSUS_SOURCE, 

324 path=None, 

325 line=None, 

326 anchorable=False, 

327 ) 

328 ) 

329 blocked = any(f.severity in ("critical", "major") for f in findings) 

330 return (not blocked), findings 

331 

332 

333def could_not_run(findings) -> bool: 

334 """Did this jury gate result come back without running (no CLI, or an empty diff)?""" 

335 return any(f.source == NOT_RUN_SOURCE for f in findings) 

336 

337 

338def run_gate( 

339 diff_text: str, 

340 *, 

341 cwd: str | None = None, 

342 mode: str = "advisory", 

343 timeout: int = DEFAULT_JURY_TIMEOUT_S, 

344 _run=None, 

345) -> tuple[bool, list[Finding], bool]: 

346 """Run ``jury`` on ``diff_text`` and map its findings. 

347 

348 Returns ``(ok, findings, timed_out)``. ``ok`` is False when a finding blocks 

349 (critical/major) or when the run produced no verdict at all in gating mode — and, in 

350 gating mode, when the panel's **consensus** does not approve or the report states 

351 none (#1436, :func:`_consensus_finding`). Both rules hold at once: a verified major 

352 still blocks a panel that approved, and a panel that requested changes over minors 

353 alone, or abstained, no longer passes because no finding was severe. 

354 No-op when there is no diff or the ``jury`` CLI is not installed — keel does not 

355 depend on ai-jury, so an absent CLI is a legitimate no-op *for this run*, distinct 

356 from a run that started and did not finish. A no-op is not a pass: it returns one 

357 ``nit`` finding from :data:`NOT_RUN_SOURCE` saying why nothing was judged, so the 

358 outcome reads ``SKIPPED`` and a plan with no other gate blocks (#1369). It waives 

359 nothing downstream: a gating jury's ``jury-verdict`` is still required at merge (see 

360 the module docstring). 

361 

362 Three ways a run can end without a review, all handled alike — gating fails closed 

363 with a blocking ``major``, advisory surfaces a ``minor``: 

364 

365 * the diff is oversize and was never submitted, 

366 * the CLI was killed by ``timeout``, 

367 * the CLI returned no parseable verdict, whatever its exit code. 

368 

369 The last used to report ``(True, [])``: :func:`parse_findings` yields ``[]`` for 

370 unparseable output, so ``blocked`` came out False and a hung, crashed, or 

371 unreadable panel read as a clean pass. The test is deliberately *"did we parse a 

372 verdict"* rather than *"was the exit code zero"* — ai-jury exits nonzero to signal 

373 "request changes", which is a completed review whose findings must be honoured, 

374 while an exit of zero carrying unreadable output is not a review at all. 

375 """ 

376 if diff_text is None: 

377 # The diff could not be read (git failed). That is not "nothing to review": 

378 # passing here would silently remove the review gate from the merge decision, 

379 # which is the same fail-open the verdict check below exists to prevent. 

380 gating = mode == "gating" 

381 return ( 

382 (not gating), 

383 [_unreadable_diff_finding(severity="major" if gating else "minor")], 

384 False, 

385 ) 

386 if not diff_text: 

387 return True, [_not_run_finding(NOT_RUN_EMPTY_DIFF)], False 

388 size = len(diff_text.encode("utf-8")) 

389 if size > MAX_DIFF_BYTES: 

390 if mode == "gating": 

391 return False, [_oversize_finding(size, severity="major")], False 

392 return True, [_oversize_finding(size)], False 

393 if not available(cwd=cwd, _run=_run): 

394 return True, [_not_run_finding(NOT_RUN_NO_CLI)], False 

395 fd, path = tempfile.mkstemp(suffix=".diff") 

396 try: 

397 with os.fdopen(fd, "w", encoding="utf-8") as fh: 

398 fh.write(diff_text) 

399 result = run_argv( 

400 ["jury", "--format", "json", "--diff-file", path], cwd=cwd, timeout=timeout, **_kw(_run) 

401 ) 

402 finally: 

403 os.unlink(path) 

404 # stdout alone: ai-jury logs its progress (`[jury] …`) to stderr, and reading the 

405 # concatenation is what made every report unparseable (#624). `parse_report` still 

406 # tolerates trailing non-JSON, for a vendor that also chats on stdout. 

407 report = parse_report(result.stdout) 

408 if report is None: 

409 gating = mode == "gating" 

410 incomplete = _incomplete_finding( 

411 result, timeout=timeout, severity="major" if gating else "minor" 

412 ) 

413 # timed_out rides along so the outcome renders as TIMEOUT rather than FAIL, 

414 # the distinction #622 established for command gates. 

415 return (not gating), [incomplete], result.timed_out 

416 consensus = _consensus_finding(result.stdout, gating=mode == "gating") 

417 if consensus is not None: 

418 report = [*report, consensus] 

419 blocked = any(f.severity in ("critical", "major") for f in report) 

420 return (not blocked), report, False 

421 

422 

423# --------------------------------------------------------------------------- # 

424# Per-reviewer ballots (#1015) — the panel *as* the review, not beside it. 

425# --------------------------------------------------------------------------- # 

426 

427#: The ``role`` ai-jury stamps on the chair's entry in the report's ``reviewers`` 

428#: array. The chair is the consensus record, not a panelist ballot, so it renders 

429#: as the jury verdict rather than as one more review verdict. 

430CHAIR_ROLE = "chair" 

431 

432#: ai-jury ballot tokens → keel verdict vocabulary. ai-jury emits one machine 

433#: token per ballot (``REQUEST_CHANGES``, never ``REQUEST CHANGES``) in either the 

434#: code or the ``--issue`` vocabulary; keel's verdicts are ``LGTM`` / 

435#: ``REQUEST_CHANGES`` / ``COMMENT`` / ``ABSTAIN``. An unknown token is carried 

436#: through verbatim rather than folded into ``LGTM``: inventing an approval for a 

437#: stance keel does not recognise is the one mapping error that cannot be undone. 

438_VERDICT = { 

439 "APPROVE": "LGTM", 

440 "READY": "LGTM", 

441 "REQUEST_CHANGES": "REQUEST_CHANGES", 

442 "NEEDS_INFO": "REQUEST_CHANGES", 

443 "COMMENT": "COMMENT", 

444 "UNCLEAR": "COMMENT", 

445 "ABSTAIN": "ABSTAIN", 

446 "NO_QUORUM": "ABSTAIN", 

447} 

448 

449#: Files a ballot's scope line names before it starts counting instead. 

450_SCOPE_FILES = 8 

451 

452#: The verification status ai-jury stamps on a consensus group the verification 

453#: round upheld. Only these findings gate: an unsupported or unverified claim is 

454#: reported, never merged against. 

455VERIFIED_STATUS = "verified" 

456 

457 

458class JuryReportError(ValueError): 

459 """Raised when an ai-jury report cannot be read as a panel of ballots.""" 

460 

461 

462def map_verdict(verdict: str) -> str: 

463 """Map an ai-jury ballot token onto keel's verdict vocabulary.""" 

464 token = (verdict or "").strip().upper().replace(" ", "_").replace("-", "_") 

465 if not token: 

466 return "ABSTAIN" 

467 return _VERDICT.get(token, token) 

468 

469 

470@dataclass(frozen=True) 

471class Ballot: 

472 """One panelist's own stance, with the provenance that makes it evidence.""" 

473 

474 reviewer: str 

475 verdict: str 

476 vendor: str | None = None 

477 model: str | None = None 

478 verified_count: int = 0 

479 round1_ok: bool = True 

480 findings: tuple[dict[str, Any], ...] = () 

481 scope: str | None = None 

482 testing: str | None = None 

483 counts_as_review: bool | None = None 

484 scope_substantive: bool | None = None 

485 abstention_cause: str | None = None 

486 

487 def as_review(self) -> dict[str, Any]: 

488 """This ballot in the ``keel review --reviews`` bundle shape.""" 

489 return { 

490 "reviewer": self.reviewer, 

491 "verdict": self.verdict, 

492 "scope": ballot_scope(self), 

493 "findings": [dict(finding) for finding in self.findings], 

494 "testing": ballot_testing(self), 

495 "vendor": self.vendor, 

496 "model": self.model, 

497 } 

498 

499 

500def _review_ballots(ballots: tuple[Ballot, ...]) -> tuple[Ballot, ...]: 

501 """Ballots that count as reviews — the one definition :class:`Panel` consumes.""" 

502 return tuple(ballot for ballot in ballots if ballot_is_review(ballot)) 

503 

504 

505@dataclass(frozen=True) 

506class Panel: 

507 """A parsed ai-jury panel: the panelist ballots and the chair's consensus.""" 

508 

509 ballots: tuple[Ballot, ...] = () 

510 chair: Ballot | None = None 

511 verified: tuple[dict[str, Any], ...] = () 

512 

513 @property 

514 def size(self) -> int: 

515 """Reviews this panel produced — the reviewer count the evidence gate sizes. 

516 

517 Aligns with ai-jury's ``is_review``: an abstention, an empty ballot, or a 

518 ``counts_as_review: false`` record is not a review and does not inflate 

519 ``panelists`` / ``jury_panel_size``. :func:`parse_panel` already drops 

520 those from :attr:`ballots`; this property re-applies the same predicate 

521 so a hand-built panel cannot disagree with the posting path. 

522 """ 

523 return len(_review_ballots(self.ballots)) 

524 

525 @property 

526 def vendors(self) -> tuple[str, ...]: 

527 """Distinct declared vendors across the reviews, in panel order. 

528 

529 Lower-cased and de-duplicated exactly as :func:`keel.evidence.distinct_vendor_check` 

530 reads the posted ``vendor:`` lines, so the count declared on the jury verdict and 

531 the count the evidence gate recomputes from the verdicts cannot disagree. 

532 Abstaining seats are not reviews and do not contribute a vendor. 

533 """ 

534 seen: list[str] = [] 

535 for ballot in _review_ballots(self.ballots): 

536 vendor = (ballot.vendor or "").strip().lower() 

537 if vendor and vendor not in seen: 

538 seen.append(vendor) 

539 return tuple(seen) 

540 

541 def reviews(self) -> tuple[dict[str, Any], ...]: 

542 """Review ballots in the ``--reviews`` bundle shape. 

543 

544 Non-reviews are omitted: posting them as head-pinned ``review-verdict-*`` 

545 evidence is the defect this mapping exists to close. 

546 """ 

547 return tuple(ballot.as_review() for ballot in _review_ballots(self.ballots)) 

548 

549 

550def _finding_record(raw: Any) -> dict[str, Any] | None: 

551 """One ai-jury finding in keel's finding shape (``file``→``path``, ``claim``→``message``).""" 

552 if not isinstance(raw, dict): 

553 return None 

554 line = raw.get("line") 

555 return { 

556 "severity": map_severity(raw.get("severity", "")), 

557 "path": raw.get("file") or None, 

558 "line": line if isinstance(line, int) else None, 

559 "message": raw.get("claim") or "(jury finding)", 

560 } 

561 

562 

563def _ballot_findings(raw: Any, findings: list[Any]) -> tuple[dict[str, Any], ...]: 

564 """Resolve a ballot's ``findings`` index list against the report's findings array. 

565 

566 Out-of-range and non-integer indexes are dropped rather than raising: the 

567 ballot's stance is the evidence, and a report whose indexes do not line up 

568 must still produce a verdict that says so with the findings it *can* resolve. 

569 """ 

570 if not isinstance(raw, list): 

571 return () 

572 records: list[dict[str, Any]] = [] 

573 for index in raw: 

574 if not isinstance(index, int) or isinstance(index, bool): 

575 continue 

576 if not 0 <= index < len(findings): 

577 continue 

578 record = _finding_record(findings[index]) 

579 if record is not None: 

580 records.append(record) 

581 return tuple(records) 

582 

583 

584def _text_field(raw: dict[str, Any], key: str) -> str | None: 

585 """A non-empty string field, or ``None`` when absent / blank / the wrong type.""" 

586 value = raw.get(key) 

587 if isinstance(value, str) and value.strip(): 

588 return value.strip() 

589 return None 

590 

591 

592def _bool_field(raw: dict[str, Any], key: str) -> bool | None: 

593 """A JSON boolean field, or ``None`` when absent or not a bool. 

594 

595 Integers are refused: ``1``/``0`` are not the schema ≥1.2 flags, and treating 

596 them as booleans would let a malformed report opt a ballot into the review 

597 count. 

598 """ 

599 value = raw.get(key) 

600 if isinstance(value, bool): 

601 return value 

602 return None 

603 

604 

605def _ballot(raw: Any, findings: list[Any], *, position: int) -> Ballot: 

606 if not isinstance(raw, dict): 

607 raise JuryReportError(f"jury report reviewer #{position} must be a JSON object") 

608 name = raw.get("name") 

609 if not isinstance(name, str) or not name.strip(): 

610 raise JuryReportError(f"jury report reviewer #{position} requires a non-empty 'name'") 

611 vendor = raw.get("vendor") 

612 model = raw.get("model") 

613 verified = raw.get("verified_count") 

614 return Ballot( 

615 reviewer=name.strip(), 

616 verdict=map_verdict(raw.get("verdict", "")), 

617 vendor=vendor.strip() if isinstance(vendor, str) and vendor.strip() else None, 

618 model=model.strip() if isinstance(model, str) and model.strip() else None, 

619 verified_count=verified 

620 if isinstance(verified, int) and not isinstance(verified, bool) 

621 else 0, 

622 round1_ok=bool(raw.get("round1_ok", True)), 

623 findings=_ballot_findings(raw.get("findings"), findings), 

624 scope=_text_field(raw, "scope"), 

625 testing=_text_field(raw, "testing"), 

626 counts_as_review=_bool_field(raw, "counts_as_review"), 

627 scope_substantive=_bool_field(raw, "scope_substantive"), 

628 abstention_cause=_text_field(raw, "abstention_cause"), 

629 ) 

630 

631 

632def _verified_records(data: dict) -> tuple[dict[str, Any], ...]: 

633 """Consensus-group representatives the verification round upheld. 

634 

635 These are the findings that gate. ai-jury verifies a consensus group and 

636 stamps ``verification_status``; keel's own rule — critical/major block — 

637 applies to the *upheld* ones only, so a claim the panel could not support 

638 never holds a merge. 

639 """ 

640 records: list[dict[str, Any]] = [] 

641 for group in data.get("consensus") or []: 

642 if not isinstance(group, dict): 

643 continue 

644 if (group.get("verification_status") or "") != VERIFIED_STATUS: 

645 continue 

646 record = _finding_record(group.get("representative")) 

647 if record is not None: 

648 reviewers = group.get("reviewers") 

649 record["reviewers"] = ( 

650 [name for name in reviewers if isinstance(name, str)] 

651 if isinstance(reviewers, list) 

652 else [] 

653 ) 

654 records.append(record) 

655 return tuple(records) 

656 

657 

658def parse_panel(data: dict | str) -> Panel | None: 

659 """Parse an ai-jury JSON report into a :class:`Panel`, or ``None``. 

660 

661 ``None`` means *this is not a report carrying per-reviewer ballots* — an 

662 unparseable document, or a pre-schema-1.1 report with no ``reviewers`` array. 

663 The caller turns that into an actionable error (upgrade ai-jury, or supply a 

664 ``--reviews`` bundle); it is deliberately not an exception, because "not a 

665 ballot report" is the same question :func:`parse_report` answers for findings. 

666 

667 A report that *does* carry ballots but carries them malformed raises 

668 :class:`JuryReportError`: dropping a panelist would silently post fewer 

669 verdicts than the panel produced, which is the one failure this whole path 

670 exists to prevent. 

671 

672 Only ballots that :func:`ballot_is_review` accepts enter :attr:`Panel.ballots`. 

673 An ``ABSTAIN``, a ``counts_as_review: false`` record, or an older empty 

674 ballot is parsed and then dropped, so it cannot inflate ``panel.size`` or 

675 become a posted ``review-verdict-*``. 

676 """ 

677 if isinstance(data, str): 

678 try: 

679 data, _end = json.JSONDecoder().raw_decode(data.lstrip()) 

680 except json.JSONDecodeError: 

681 return None 

682 if not isinstance(data, dict): 

683 return None 

684 raw_reviewers = data.get("reviewers") 

685 if not isinstance(raw_reviewers, list): 

686 return None 

687 findings = list(data.get("findings") or []) 

688 ballots: list[Ballot] = [] 

689 chair: Ballot | None = None 

690 for position, raw in enumerate(raw_reviewers, start=1): 

691 ballot = _ballot(raw, findings, position=position) 

692 if isinstance(raw, dict) and (raw.get("role") or "") == CHAIR_ROLE: 

693 chair = ballot 

694 continue 

695 if ballot_is_review(ballot): 

696 ballots.append(ballot) 

697 return Panel(ballots=tuple(ballots), chair=chair, verified=_verified_records(data)) 

698 

699 

700def _finding_paths(ballot: Ballot) -> list[str]: 

701 """Distinct file paths this ballot's own findings named, in first-seen order.""" 

702 files: list[str] = [] 

703 for finding in ballot.findings: 

704 path = finding.get("path") 

705 if isinstance(path, str) and path.strip() and path not in files: 

706 files.append(path.strip()) 

707 return files 

708 

709 

710def ballot_is_review(ballot: Ballot) -> bool: 

711 """Whether this panelist ballot counts as a review (ai-jury ``is_review``). 

712 

713 A review is a panelist whose scope is substantive and whose verdict is not 

714 ``ABSTAIN``. The chair is split off before this predicate runs. 

715 

716 The flags a schema ≥1.2 report declares — ``counts_as_review`` and 

717 ``scope_substantive`` — can only ever **remove** a ballot here, never admit 

718 one that carries nothing. ai-jury derives ``counts_as_review`` from 

719 ``scope_substantive``, which is itself derived from the ``scope`` prose, so a 

720 record claiming ``counts_as_review: true`` with no ``scope`` and no finding 

721 is not a clean review it produced; it is internally inconsistent, and 

722 admitting it means keel writing the substance the report failed to supply. 

723 That is the escape hatch #1150 is about, so the ambiguous record fails closed 

724 like every other one. 

725 

726 What is left is the fact the scope line reads: a ballot counts when the 

727 report gave prose to post or a path to name. Older reports carry no flags and 

728 are decided by that same fact, so a schema-1.1 empty ``APPROVE`` is dropped 

729 rather than dressed up. 

730 """ 

731 if ballot.verdict == "ABSTAIN": 

732 return False 

733 if ballot.counts_as_review is False or ballot.scope_substantive is False: 

734 return False 

735 return bool(ballot.scope) or bool(_finding_paths(ballot)) 

736 

737 

738def _abstention_scope(ballot: Ballot) -> str: 

739 """An explicitly anchorless scope: no ``Checked …``, no path, no backtick.""" 

740 cause = (ballot.abstention_cause or "").replace("_", " ") 

741 if cause: 

742 return f"ai-jury panelist {ballot.reviewer} did not review ({cause})." 

743 return f"ai-jury panelist {ballot.reviewer} did not review." 

744 

745 

746def ballot_scope(ballot: Ballot) -> str: 

747 """The scope line keel renders for a panelist ballot. 

748 

749 A ballot that is not a review never gets a substance-passing opener: keel 

750 used to always start with ``Checked the changed-file diff…``, which is its 

751 own :func:`keel.evidence.verdict_substance` escape hatch, so an ``ABSTAIN`` 

752 still passed the gate by construction. 

753 

754 Every remaining branch renders something the report actually supplied. A 

755 schema ≥1.2 ballot carries its own ``scope`` and that prose is posted as 

756 written — including the clean review that read the diff and found nothing, 

757 which ai-jury describes itself rather than leaving keel to. Otherwise the 

758 ballot named paths, and the ``checked …`` line is built from them. There is 

759 no third case: :func:`ballot_is_review` admits a ballot only when one of 

760 those two is true, so this function never has to invent a scope for a ballot 

761 it was told to post. 

762 """ 

763 if not ballot_is_review(ballot): 

764 return _abstention_scope(ballot) 

765 if ballot.scope: 

766 return ballot.scope 

767 # Non-empty: `ballot_is_review` accepted this ballot, and with no `scope` 

768 # prose the only way it could have is by naming a path. 

769 files = _finding_paths(ballot) 

770 opening = f"Checked the changed-file diff as ai-jury panelist {ballot.reviewer}" 

771 listed = ", ".join(files[:_SCOPE_FILES]) 

772 more = len(files) - _SCOPE_FILES 

773 suffix = f" (+{more} more)" if more > 0 else "" 

774 return f"{opening}; named {len(files)} file(s): {listed}{suffix}." 

775 

776 

777def ballot_testing(ballot: Ballot) -> str: 

778 """The testing line keel renders for a panelist ballot. 

779 

780 Schema ≥1.2 reports carry their own ``testing`` prose; that is preferred 

781 when present. Otherwise the panel's verification round *is* the ballot's 

782 testing note: it is the only check ai-jury performs on a reviewer's claims, 

783 and a ballot whose claims were never upheld must say so rather than borrow 

784 the PR's own testing section. 

785 """ 

786 if ballot.testing: 

787 return ballot.testing 

788 if ballot.verified_count > 0: 

789 note = ( 

790 f"ai-jury verification upheld {ballot.verified_count} consensus " 

791 "group(s) this panelist joined." 

792 ) 

793 else: 

794 note = "ai-jury verification upheld no consensus group from this panelist." 

795 if not ballot.round1_ok: 

796 return f"The panelist's adapter reported a failed run; its output was still read. {note}" 

797 return note 

798 

799 

800def verified_findings(panel: Panel) -> list[Finding]: 

801 """Verified consensus findings as keel :class:`~keel.findings.Finding`s. 

802 

803 This is the s9 input: ``critical``/``major`` block, ``minor`` is a gated 

804 suggestion, ``nit`` is advisory — the same mapping a host reviewer's findings 

805 get, which is the whole point of the panel being the review rather than a 

806 second opinion beside it. 

807 """ 

808 out: list[Finding] = [] 

809 for record in panel.verified: 

810 reviewers = record.get("reviewers") or [] 

811 source = f"jury:{reviewers[0]}" if reviewers else "jury:consensus" 

812 path = record.get("path") 

813 line = record.get("line") 

814 out.append( 

815 Finding( 

816 severity=record["severity"], 

817 message=record["message"], 

818 source=source, 

819 path=path, 

820 line=line, 

821 anchorable=bool(path) and isinstance(line, int), 

822 ) 

823 ) 

824 return out 

825 

826 

827def jury_verdict(panel: Panel) -> dict[str, Any]: 

828 """The ``render_jury_verdict`` arguments for a parsed panel. 

829 

830 The chair's ballot is the consensus record — that is what the jury verdict 

831 comment has always been — and the panel's own size and vendor count ride 

832 along on it, because the posted verdict is the only channel by which either 

833 reaches a hosted evidence check (see :func:`keel.artifacts.render_jury_verdict`). 

834 """ 

835 chair = panel.chair 

836 summary = [f"{record['severity']}: {record['message']}" for record in panel.verified] 

837 return { 

838 "verdict": chair.verdict if chair is not None else "ABSTAIN", 

839 "participants": [ 

840 f"{ballot.reviewer} ({ballot.vendor})" if ballot.vendor else ballot.reviewer 

841 for ballot in _review_ballots(panel.ballots) 

842 ], 

843 "participating_vendors": len(panel.vendors), 

844 "panelists": panel.size, 

845 "findings_summary": summary, 

846 "remaining_risks": None if summary else "none identified", 

847 }