Coverage for src/keel/ship.py: 100%

178 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""The deterministic ship decisions — keel's value-add as pure functions. 

2 

3The agentic steps and the git/gh plumbing live in the adapter + I/O layer; the 

4*decisions* (how many reviewers, whether to merge / defer / block, whether to keep 

5fixing) are pure and live here, so they are reproducible and fully unit-tested. 

6""" 

7 

8from __future__ import annotations 

9 

10from collections.abc import Mapping, Sequence 

11from dataclasses import dataclass 

12from typing import Any 

13 

14from . import classify 

15from . import team as team_policy 

16from .findings import Verdict, decision_for 

17from .window import is_merge_open 

18 

19#: Hard cap on review→fix rounds (matches ship's budget). 

20MAX_FIX_ROUNDS = 3 

21 

22#: GitHub check-rollup conclusions that count as "not failing". 

23CI_OK_STATES = frozenset({"SUCCESS", "NEUTRAL", "SKIPPED"}) 

24 

25POSTING_MODES = frozenset({"inline", "summary"}) 

26 

27# A cross-vendor jury needs at least this many distinct vendors to gate. Below it 

28# the panel cannot produce cross-vendor consensus, so the verdict is advisory — 

29# and a run where no agent produced output counts as zero, which is how "a jury 

30# that did not complete cleanly never gates" falls out of the same comparison. 

31MINIMUM_JURY_VENDORS = 2 

32 

33#: What ``keel merge --hotfix`` actually skips, named by the keys the ``keel.merge.v1`` 

34#: record marks ``{"bypassed": true, "reason": "hotfix"}``. 

35#: 

36#: The contract used to publish a single ``hotfix_bypasses_window_only: True``, which 

37#: has been false since the gates-SHA bypass landed: a consumer reading it would treat a 

38#: hotfix merge as gates-verified when no gates-pass ledger record was ever matched 

39#: (#1078). Keep this tuple derived from :func:`keel.cli._cmd_merge` — the agreement is 

40#: pinned by ``tests/test_cli.py``, which reads it back off a real ``--hotfix`` run. 

41HOTFIX_BYPASSES = ("gates_sha", "window") 

42 

43#: What ``--hotfix`` never skips. All but ``findings`` are keys of the same merge record, 

44#: present and un-bypassed on a hotfix run. ``findings`` blocks earlier — at the review 

45#: verdict this contract's ``finding_policy`` block governs — so a blocked change never 

46#: reaches the merge command to have a key there at all. 

47HOTFIX_NEVER_BYPASSES = ("checkpoint_gate", "ci", "evidence", "findings", "lock") 

48 

49REVIEW_FOCUS_A = ( 

50 "logic correctness", 

51 "null safety", 

52 "language interop", 

53) 

54REVIEW_FOCUS_B = ( 

55 "platform compatibility", 

56 "lifecycle safety", 

57 "API compatibility", 

58 "threading", 

59) 

60REVIEW_FOCUS_C = ( 

61 "test coverage", 

62 "docs gate", 

63 "scope creep", 

64 "CI prediction", 

65 "security", 

66) 

67 

68 

69def reviewer_count(tier: int) -> int: 

70 """Reviewers for a risk tier: TIER-3→3, TIER-2→2, TIER-1→1 (default 2).""" 

71 return {3: 3, 2: 2, 1: 1}.get(tier, 2) 

72 

73 

74def reviewer_focuses(count: int) -> tuple[dict[str, Any], ...]: 

75 """Focus coverage for each reviewer slot. Lower counts merge focus; none are dropped. 

76 

77 Zero slots is not "one slot with everything merged in": it is a tier whose 

78 ``knobs.team`` policy made the jury the review panel (#1014), so there is no host 

79 reviewer to carry a focus. The panel's own coverage is the jury's business. 

80 """ 

81 if count <= 0: 

82 return () 

83 if count <= 1: 

84 return ( 

85 { 

86 "slot": "A", 

87 "focus": list(REVIEW_FOCUS_A + REVIEW_FOCUS_B + REVIEW_FOCUS_C), 

88 "merged_from": ["A", "B", "C"], 

89 }, 

90 ) 

91 if count == 2: 

92 return ( 

93 { 

94 "slot": "A", 

95 "focus": list(REVIEW_FOCUS_A + REVIEW_FOCUS_B), 

96 "merged_from": ["A", "B"], 

97 }, 

98 { 

99 "slot": "C", 

100 "focus": list(REVIEW_FOCUS_C), 

101 "merged_from": ["C"], 

102 }, 

103 ) 

104 return ( 

105 {"slot": "A", "focus": list(REVIEW_FOCUS_A), "merged_from": ["A"]}, 

106 {"slot": "B", "focus": list(REVIEW_FOCUS_B), "merged_from": ["B"]}, 

107 {"slot": "C", "focus": list(REVIEW_FOCUS_C), "merged_from": ["C"]}, 

108 ) 

109 

110 

111def resolve_jury( 

112 *, 

113 tier: int | None, 

114 gates: tuple[str, ...] = (), 

115 jury: bool = False, 

116 no_jury: bool = False, 

117 jury_advisory: bool = False, 

118 participating_vendors: int | None = None, 

119 panel_is_jury: bool = False, 

120 policy_mode: str | None = None, 

121 minimum_vendors: int = MINIMUM_JURY_VENDORS, 

122 panel_unavailable: bool = False, 

123) -> dict[str, Any]: 

124 """Resolve the cross-vendor jury mode using ship flag precedence. 

125 

126 ``panel_unavailable`` outranks everything, including ``tier == 3``'s auto-on (#1066). 

127 It is set only when a measured probe found the panel unstaffable *and* the project's 

128 ``team.jury.on_unavailable`` allowed a host bench in its place, and it turns the jury 

129 off because there is no panel to produce a verdict: leaving ``tier-3 auto`` standing 

130 would require a ``jury-verdict`` artifact from a panel this machine just established 

131 cannot convene, which is the tier stuck all over again one layer down. It is not the 

132 flag route #1014 closed — a preference still cannot do this, and the reason string 

133 names the fallback so nothing reads it as a plain ``--no-jury``. 

134 

135 ``panel_is_jury`` is a ``knobs.team`` tier whose review policy is ``jury``, and it 

136 **outranks every per-run jury flag**. At such a tier the panel *is* the review: there 

137 are no host reviewer slots, so a flag that turned the jury off or made it advisory 

138 would leave that tier with no required review evidence at all — a stricter policy 

139 producing a weaker gate. It is also the only answer the six commands that resolve this 

140 contract can agree on, because they are not all *given* the flags: every surface 

141 accepts them since #1043, but keel's CI passes ``--no-jury`` to ``evidence-verify`` on 

142 every run and to ``ship``/``plan`` on none. So on a panel tier the verdict stays 

143 required, whatever was typed; the flag is recorded in ``assignment.warnings`` instead 

144 of applied. Below that 

145 tier ``--no-jury`` keeps its pre-existing meaning and still beats the tier-3 auto-jury. 

146 

147 ``policy_mode`` is ``team.jury.mode``, which can make an enabled jury *advisory* — a 

148 project cannot promote a jury that ``--no-jury`` turned off. Pairing it with a jury 

149 *panel* is refused by :func:`keel.team.team_issues`, because "the panel is the review" 

150 and "the panel does not gate" together mean the tier has no enforceable review at all. 

151 ``minimum_vendors`` is ``team.jury.min_vendors``, which may raise 

152 :data:`MINIMUM_JURY_VENDORS` but never lowers it (the schema's floor is 2). 

153 

154 ``participating_vendors`` is the count of distinct vendors that actually took 

155 part in the panel. Below :data:`MINIMUM_JURY_VENDORS` a gating mode is 

156 downgraded to advisory **unless the panel is the tier's review**, because a 

157 panel that small cannot produce cross-vendor consensus — and a run where no 

158 agent returned output is simply zero, so "a jury that did not complete cleanly 

159 never gates" needs no separate branch. ``None`` means the panel is not known 

160 yet (planning, ``keel plan``, any caller resolving the contract before s8 

161 runs), and leaves the mode alone. 

162 

163 The downgrade must live here rather than in adapter prose: the evidence gate 

164 derives its ``jury-verdict`` requirement from this ``mode``, so a mode that 

165 ignores the real panel makes the gate demand a verdict the jury step would 

166 decline to treat as gating. 

167 

168 On a **panel tier** the downgrade is suppressed for the mirror-image reason. 

169 There the verdict is not a second opinion beside a host bench — it is the 

170 tier's own consensus record, and dropping it because the panel came back 

171 short lets a short panel excuse itself from the one artifact that says so. 

172 The short panel is still refused, by 

173 :func:`keel.evidence.panel_vendor_check`; what it may not do is quietly stop 

174 being required. ``downgraded`` reports ``False`` there, and the reason string 

175 is left alone, so nothing downstream reads a relaxation that did not happen. 

176 """ 

177 if panel_unavailable: 

178 enabled = False 

179 reason = "jury panel unavailable; host bench fallback (team.jury.on_unavailable)" 

180 elif panel_is_jury: 

181 enabled = True 

182 reason = "team.review panel" 

183 ignored = [ 

184 flag 

185 for flag, passed in (("--no-jury", no_jury), ("--jury-advisory", jury_advisory)) 

186 if passed 

187 ] 

188 if ignored: 

189 reason = f"{reason} ({' and '.join(ignored)} does not apply: the panel is the review)" 

190 elif no_jury: 

191 enabled = False 

192 reason = "--no-jury" 

193 elif jury: 

194 enabled = True 

195 reason = "--jury" 

196 elif tier == 3: 

197 enabled = True 

198 reason = "tier-3 auto" 

199 else: 

200 enabled = False 

201 reason = "default" 

202 # A panel tier gates, full stop: it has no host reviewers to fall back on, so an 

203 # advisory panel there is a tier with nothing required of it. 

204 advisory = not panel_is_jury and (jury_advisory or policy_mode == "advisory") 

205 mode = "off" if not enabled else ("advisory" if advisory else "gating") 

206 # …and the vendor downgrade is the same door, so it carries the same guard. Round 3 

207 # of #1014 closed the flag route only; a panel tier reaching this with one 

208 # participating vendor still came out advisory, which drops `jury-verdict` from the 

209 # required evidence of a tier whose panel is the *whole* review. That is the short 

210 # panel excusing itself from the verdict it came back short on. Elsewhere the 

211 # downgrade is right and stays: a jury sitting beside a host bench that did not 

212 # convene cross-vendor should not gate, because the bench still reviewed the change. 

213 downgraded = ( 

214 not panel_is_jury 

215 and mode == "gating" 

216 and participating_vendors is not None 

217 and participating_vendors < minimum_vendors 

218 ) 

219 if downgraded: 

220 mode = "advisory" 

221 reason = ( 

222 f"{reason}; downgraded to advisory " 

223 f"({participating_vendors} participating vendor(s), " 

224 f"minimum {minimum_vendors})" 

225 ) 

226 return { 

227 "enabled": enabled, 

228 "mode": mode, 

229 "reason": reason, 

230 "configured_gate": "jury" in gates, 

231 "fail_soft": True, 

232 "minimum_vendors": minimum_vendors, 

233 "participating_vendors": participating_vendors, 

234 "downgraded": downgraded, 

235 "verified_consensus_gates": enabled and mode == "gating", 

236 "severity_policy": { 

237 "critical": "block", 

238 "major": "block", 

239 "minor": "gated-suggestion", 

240 "nit": "advisory", 

241 }, 

242 } 

243 

244 

245def panel_fell_back(assignment: dict[str, Any] | None) -> bool: 

246 """Did a measured probe move this tier's review off the panel and onto a host bench? 

247 

248 Read off the resolved assignment rather than re-derived, so every surface that resolves 

249 the review contract reaches the same answer from the same measurement (#1066). Total: 

250 an assignment from before the block existed, or one with no availability recorded, 

251 reads as "no fallback" — the answer that leaves the contract exactly as it was. 

252 """ 

253 if not isinstance(assignment, dict): 

254 return False 

255 jury = assignment.get("jury") 

256 availability = jury.get("availability") if isinstance(jury, dict) else None 

257 if not isinstance(availability, dict): 

258 return False 

259 return availability.get("decision") == team_policy.JURY_ON_UNAVAILABLE[0] 

260 

261 

262def check_reviewer_override(reviewer_override: int | None) -> None: 

263 """Refuse a reviewer count keel has no reviewer vocabulary for. 

264 

265 Extracted so :func:`assess` can apply it **before** it resolves the team: the 

266 assignment is resolved first, and an out-of-range override reaching that resolver 

267 produced an ``IndexError`` from inside it instead of the documented ``ValueError`` 

268 the caller has always been able to catch. 

269 """ 

270 if reviewer_override is not None and reviewer_override not in {1, 2, 3}: 

271 raise ValueError("reviewer_override must be one of 1, 2, or 3") 

272 

273 

274def _jury_panel_size(jury_record: dict[str, Any], panel_size: int | None) -> int: 

275 """How many verdicts a jury-panel tier requires (#1015). 

276 

277 The panel *is* the review there, so its ballots are the required s7 verdicts — 

278 ``keel review --from-jury`` posts one head-pinned verdict per ballot — and the 

279 required count is the panel's own size, declared as ``panelists: <N>`` on the 

280 posted jury verdict. 

281 

282 ``minimum_vendors`` is a **floor, not a fallback**: the answer is 

283 ``max(declared, minimum_vendors)``, so a declared count can only ever *raise* 

284 the requirement. Taking the declared count verbatim let a verdict lower it — 

285 ``panelists: 1`` against a minimum of 2 asked for one ballot, while the 

286 unmeasured cases (absent, ``0``, negative) still asked for two, so the one 

287 shape that means "the panel came back short" was the one shape that relaxed 

288 the gate. The declared count is attacker-adjacent evidence in exactly the way 

289 the vendor count is: it is read off a comment, and it must not be able to 

290 shrink what the tier owes. 

291 

292 **What this deliberately does not do is move the bench.** Neither a jury flag 

293 nor the measured participating-vendor count may change *who reviews*, only 

294 whether the panel's verdict gates. The bench is a pure function of config + 

295 tier + role + ``--reviewers``/``--review-delegate`` (:func:`keel.team._review_seats`), 

296 and for the same reason: the six commands that resolve this contract are not 

297 *given* the other inputs uniformly. All six accept the jury flags since #1043, 

298 but keel's CI passes ``--no-jury`` to ``evidence-verify`` on every run and to 

299 ``ship``/``plan`` on none, and only the surfaces that can read the PR's posted 

300 jury verdict — ``evidence-verify`` and ``keel merge`` — ever see a vendor 

301 count. A bench that 

302 moved with either input would have ``keel plan`` requiring the panel's ballots 

303 while ``evidence-verify`` demanded a host bench of the same PR, which is the 

304 contract disagreement #1014 exists to prevent, reintroduced along a new axis. 

305 

306 A short panel therefore does not buy fewer eyes: the ballots stay required in 

307 full, the jury verdict stays required (:func:`resolve_jury` suppresses the 

308 downgrade on a panel tier), and a panel that spans too few vendors is refused 

309 by :func:`keel.evidence.panel_vendor_check` rather than quietly swapped for a 

310 bench nobody dispatched. 

311 

312 **Why the planning surfaces publish the floor rather than the real count.** 

313 ``keel plan`` is offline by construction and has no pull request to read a 

314 verdict from. ``keel ship --pr N`` does have one — it already makes three 

315 GitHub reads for CI status, and could make a fourth for the posted 

316 ``panelists`` — and deliberately does not. Two reasons, both about keeping one 

317 answer rather than two: the floor is *provably conservative* (this function 

318 only ever raises, so a planning surface can under-state what will be required 

319 and never over-state it), and ``keel ship`` without ``--pr``, and every dry 

320 run, must resolve the same contract with no verdict in reach — so the floor 

321 has to be right on its own regardless. Reading it only sometimes would buy a 

322 number that is sharper on some runs and identical on the rest, at the cost of 

323 a contract whose value depends on which flags the caller happened to pass. 

324 """ 

325 floor = jury_record["minimum_vendors"] 

326 if isinstance(panel_size, int) and panel_size > floor: 

327 return panel_size 

328 return floor 

329 

330 

331def resolve_review_contract( 

332 *, 

333 tier: int | None, 

334 reviewer_override: int | None = None, 

335 review_comments: str = "inline", 

336 gates: tuple[str, ...] = (), 

337 policy_pack: dict[str, Any] | None = None, 

338 jury: bool = False, 

339 no_jury: bool = False, 

340 jury_advisory: bool = False, 

341 require_distinct_vendors: bool | None = None, 

342 jury_participating_vendors: int | None = None, 

343 jury_panel_size: int | None = None, 

344 assignment: dict[str, Any] | None = None, 

345 learnings: Mapping[str, Any] | None = None, 

346) -> dict[str, Any]: 

347 """Machine-readable review, jury, test, and merge-gate plan for ship-like flows. 

348 

349 ``assignment`` is the resolved ``knobs.team`` team (:func:`keel.team.resolve_assignment`). 

350 When one is supplied it owns the reviewer bench — how many slots there are, who sits in 

351 each, and whether the jury is the panel instead — so the contract a host executes and 

352 the assignment it renders cannot disagree. Without one the tier-derived counts stand, 

353 which is every pre-#1014 caller. 

354 

355 ``require_distinct_vendors`` is tri-state at the config boundary, but ``None`` — unset 

356 — resolves to ``False`` on every tier (#1065): the independence claim is opt-in, and a 

357 bool is the project's explicit answer. 

358 

359 ``jury_panel_size`` is the number of ballots a jury panel actually returned, which 

360 only a run that has seen the panel can know (a posted jury verdict declares it; see 

361 :func:`keel.evidence.jury_panel_size`). On a tier whose panel *is* the review it 

362 becomes the required reviewer count, so the panel sizes its own bench (#1015). 

363 """ 

364 check_reviewer_override(reviewer_override) 

365 if review_comments not in POSTING_MODES: 

366 raise ValueError("review_comments must be 'inline' or 'summary'") 

367 if assignment is None: 

368 count = reviewer_override if reviewer_override is not None else reviewer_count(tier or 2) 

369 source = ( 

370 "override" 

371 if reviewer_override is not None 

372 else ("risk-tier" if tier is not None else "unresolved") 

373 ) 

374 panel, slots = "reviewers", [] 

375 panel_is_jury = False 

376 panel_unavailable = False 

377 else: 

378 count = assignment["reviewer_count"] 

379 source = assignment["reviewer_source"] 

380 panel = assignment["review_panel"] 

381 slots = list(assignment["reviewers"]) 

382 panel_is_jury = bool(assignment["jury"]["panel_is_review"]) 

383 panel_unavailable = panel_fell_back(assignment) 

384 jury_record = resolve_jury( 

385 tier=tier, 

386 gates=gates, 

387 jury=jury, 

388 no_jury=no_jury, 

389 jury_advisory=jury_advisory, 

390 participating_vendors=jury_participating_vendors, 

391 panel_is_jury=panel_is_jury, 

392 policy_mode=None if assignment is None else assignment["jury"]["mode"], 

393 minimum_vendors=( 

394 MINIMUM_JURY_VENDORS if assignment is None else assignment["jury"]["min_vendors"] 

395 ), 

396 panel_unavailable=panel_unavailable, 

397 ) 

398 # The probe's verdict travels *on the contract*, not only in the assignment (#1066). 

399 # `evidence-verify`, the ledger and the closure comment all read the contract, and a 

400 # fallback that only the assignment recorded would be exactly the silent downgrade 

401 # ai-jury #682 was opened for: a review that says nothing about the panel it replaced. 

402 jury_record["panel_unavailable"] = panel_unavailable 

403 jury_record["availability"] = None if assignment is None else assignment["jury"]["availability"] 

404 if panel_is_jury: 

405 count, source = _jury_panel_size(jury_record, jury_panel_size), "jury" 

406 pack = policy_pack or {} 

407 review_policy = pack.get("review", {}) if isinstance(pack.get("review", {}), dict) else {} 

408 return { 

409 "reviewers": { 

410 "count": count, 

411 "source": source, 

412 "tier": tier, 

413 "independent": True, 

414 "self_review_counts_toward_lgtm": False, 

415 "minimum_lgtm": count, 

416 "require_distinct_vendors": team_policy.require_distinct_vendors( 

417 require_distinct_vendors 

418 ), 

419 "orchestrator_owns_writes": True, 

420 "panel": panel, 

421 # Per-slot provider/model/effort, so a host dispatches the configured vendor 

422 # for slot B instead of running one vendor N times (#1014). Empty for every 

423 # caller that resolves no team, which keeps the pre-#1014 contract intact. 

424 "slots": slots, 

425 # A panel picks its own coverage; keel's A/B/C focus slices describe a bench 

426 # keel staffs, and handing them to ai-jury would be keel briefing reviewers it 

427 # never dispatched. 

428 "focuses": ([] if panel == team_policy.JURY_PANEL else list(reviewer_focuses(count))), 

429 "project_additions": list(review_policy.get("additions", [])), 

430 "required_sections": list(review_policy.get("required_sections", [])), 

431 # The lessons this project already recorded about work of this shape 

432 # (#1155), same shape as `project_additions` and for the same reason: 

433 # a reviewer who is told what went wrong last time can check the 

434 # implementation against it. Empty for every project with no 

435 # learnings on disk, which is every project until it has some. 

436 "past_learnings": list((learnings or {}).get("hits", [])), 

437 }, 

438 "posting": { 

439 "mode": review_comments, 

440 "inline_default": True, 

441 "per_reviewer_inline_fallback": "summary", 

442 "summary_mode": review_comments == "summary", 

443 }, 

444 "jury": jury_record, 

445 "finding_policy": { 

446 "critical": "block", 

447 "major": "block", 

448 "minor": "gated-suggestion", 

449 "nit": "advisory", 

450 "suggestions_require_fix_or_explicit_deferral": True, 

451 "parser_source": "reviewer-returned-findings", 

452 }, 

453 "fixloop": { 

454 "max_rounds": MAX_FIX_ROUNDS, 

455 "blocker_rerun": "full-review", 

456 "suggestion_only_rerun": "narrowed-originating-focus", 

457 }, 

458 "ci": { 

459 "failure_before_pending": True, 

460 "empty_check_set_allowed_for_docs_only": True, 

461 "retry_budget": 3, 

462 }, 

463 "test_gates": { 

464 "configured_gates": list(gates), 

465 "no_jury_preserves_review_and_test_gates": True, 

466 }, 

467 "merge_gate": { 

468 "merge_window_applies_to": "literal-merge-only", 

469 "merge_lock_scope": "literal-merge-only", 

470 "final_mergeability_recheck_inside_lock": True, 

471 "hotfix_bypasses": list(HOTFIX_BYPASSES), 

472 "hotfix_never_bypasses": list(HOTFIX_NEVER_BYPASSES), 

473 "pr_merged_state_authoritative": True, 

474 }, 

475 "closeout": { 

476 "comment_targets": ["issue", "pull_request"], 

477 "capture_marker_required": True, 

478 "status_done_after_merge_only": True, 

479 }, 

480 } 

481 

482 

483@dataclass(frozen=True) 

484class MergeDecision: 

485 action: str # "merge" | "defer" | "block" 

486 reason: str 

487 

488 

489#: The built-in jury gate writes ``jury:<reviewer>`` (and ``jury:consensus`` / 

490#: ``jury:incomplete-run`` / …) as a finding's source: one gate with several voices. 

491_JURY_SOURCE_PREFIX = "jury:" 

492 

493 

494def _gate_id(source: str) -> str: 

495 """The gate a finding's ``source`` belongs to. 

496 

497 Command gates write their ``spec.id`` verbatim, and nothing forbids a colon in an 

498 extension's id, so only the jury's own ``jury:`` prefix is collapsed — splitting 

499 every source on ``:`` would turn an operator's ``sec:scan`` gate into ``sec``. 

500 """ 

501 return "jury" if source.startswith(_JURY_SOURCE_PREFIX) else source 

502 

503 

504def blocking_sources(verdict: Verdict) -> tuple[str, ...]: 

505 """The distinct gates whose findings block this verdict, sorted. 

506 

507 Ship's verdict is built from gate outcomes, so a finding's ``source`` is the gate's 

508 id, except for the jury's ``jury:<voice>`` sources, which :func:`_gate_id` folds 

509 back to the one gate they belong to. 

510 """ 

511 return tuple( 

512 sorted( 

513 { 

514 _gate_id(finding.source) 

515 for finding in verdict.findings 

516 if finding.source and decision_for(finding.severity) == "block" 

517 } 

518 ) 

519 ) 

520 

521 

522def block_reason(verdict: Verdict) -> str: 

523 """The reason a blocked verdict gives, naming what blocked it when it can. 

524 

525 "blocking findings present" is true and useless next to a reviewer verdict that says 

526 "none blocking": the findings it means are the ones a failed ``on_fail: block`` gate 

527 produced, and the line was the one place the operator looked that did not say which 

528 gate (#1007). A verdict blocked with no attributable source keeps the old wording. 

529 """ 

530 sources = blocking_sources(verdict) 

531 if not sources: 

532 return "blocking findings present" 

533 return f"blocking findings from gate(s): {', '.join(sources)}" 

534 

535 

536def decide_merge( 

537 verdict: Verdict, 

538 *, 

539 window_open: bool, 

540 is_blocker: bool = False, 

541 unrun_blocking_gates: tuple[str, ...] = (), 

542) -> MergeDecision: 

543 """Decide what to do with a green-or-not PR given the window. 

544 

545 * blocking findings ⇒ **block** (never merges); 

546 * a required gate that nobody ran ⇒ **block** (no verdict exists to clear it); 

547 * outside the merge window and not a blocker ⇒ **defer** to the morning queue; 

548 * otherwise ⇒ **merge**. A blocker bypasses the window (but never the findings). 

549 

550 ``unrun_blocking_gates`` names ``on_fail: block`` gates this run did not execute — 

551 agentic gates reach the command-only runner, which does not dispatch them. The 

552 assessment must say so: :func:`keel.ledger.record_gates_passed` refuses to certify 

553 such a record, so reporting "clear to merge" would promise a merge that 

554 ``keel merge`` will then refuse, with the operator given no reason why. 

555 """ 

556 if verdict.blocked: 

557 return MergeDecision("block", block_reason(verdict)) 

558 if unrun_blocking_gates: 

559 listed = ", ".join(unrun_blocking_gates) 

560 return MergeDecision( 

561 "block", 

562 f"required gate(s) not run: {listed} — record a result with " 

563 "--gate-result <id>=pass|fail once the gate has been dispatched", 

564 ) 

565 if not window_open and not is_blocker: 

566 return MergeDecision("defer", "outside merge window (night no-merge)") 

567 reason = "blocker bypass" if (is_blocker and not window_open) else "clear to merge" 

568 return MergeDecision("merge", reason) 

569 

570 

571def should_run_fixloop(verdict: Verdict, *, current_round: int, cap: int = MAX_FIX_ROUNDS) -> bool: 

572 """True if there are blocking findings and the fix budget is not exhausted.""" 

573 return verdict.blocked and current_round < cap 

574 

575 

576def ci_passing(ci_conclusion: str | None) -> bool | None: 

577 """Interpret a check-rollup string (e.g. ``"SUCCESS,FAILURE"``). ``None`` == unknown.""" 

578 if ci_conclusion is None: 

579 return None 

580 parts = [p.strip().upper() for p in ci_conclusion.split(",") if p.strip()] 

581 if not parts: 

582 return None 

583 # ⚡ Bolt: ~3.4x faster validation using C-level frozenset.issuperset 

584 # instead of generator expression 

585 return CI_OK_STATES.issuperset(parts) 

586 

587 

588def ci_ran(ci_conclusion: str | None) -> bool | None: 

589 """Did any check report for this head? ``None`` == we could not find out. 

590 

591 Separate from :func:`ci_passing` on purpose. "Every check passed" and "no check 

592 ran" are not the same fact, and folding them together is the defect in #675: an 

593 empty rollup used to reach the merge decision as *unknown*, and unknown did not 

594 block, so a PR nothing had verified assessed identically to a green one — then 

595 that assessment was written into the run ledger as evidence. 

596 

597 ``""`` is ``gh`` reporting an empty rollup (a fact about the PR) and returns 

598 **False**. ``None`` is ``gh`` never answered, or no PR was supplied at all (a 

599 fact about the runner) and stays **None** — keel does not block on what it 

600 could not observe, it blocks on having observed nothing. 

601 """ 

602 if ci_conclusion is None: 

603 return None 

604 return bool(ci_conclusion.strip()) 

605 

606 

607def missing_ci_workflows( 

608 workflow_names: Sequence[str] | None, 

609 ci_workflows: dict[str, str] | None, 

610) -> tuple[str, ...]: 

611 """Declared workflows in ``ci_workflows`` that reported nothing for this head. 

612 

613 ``knobs.ci_workflows`` is the project stating which workflows gate a merge, so 

614 presence can be checked against a **declaration** instead of inferred from an 

615 empty set — the difference between "I saw no failures" and "I saw the things 

616 that were supposed to run". 

617 

618 ``workflow_names`` must be *workflow* names (:func:`keel.github.ci_workflow_names`), 

619 not job names. The distinction is not cosmetic: `ci_workflows` is keyed ``CI``, 

620 while the rollup reports ``test (py3.13 / ubuntu-latest)``, so comparing against 

621 job names would report every declared workflow missing on any repo using a matrix. 

622 Matching is exact and case-insensitive — a prefix rule would let an unrelated 

623 ``testing-utils`` satisfy a declared ``test``. 

624 

625 ``()`` when nothing is declared or the names could not be read — absence of a 

626 declaration is not evidence of a missing run. 

627 """ 

628 if not ci_workflows or workflow_names is None: 

629 return () 

630 reported = {name.strip().lower() for name in workflow_names if name.strip()} 

631 return tuple( 

632 sorted(declared for declared in ci_workflows if declared.strip().lower() not in reported) 

633 ) 

634 

635 

636def is_hotfix(labels: list[str] | tuple[str, ...], *, hotfix_label: str = "hotfix") -> bool: 

637 """True if the issue/PR carries the hotfix label (case-insensitive).""" 

638 # ⚡ Bolt Optimization: Unroll any() generator and pre-compute lower() target 

639 target = hotfix_label.lower() 

640 for label in labels: 

641 if label.strip().lower() == target: 

642 return True 

643 return False 

644 

645 

646@dataclass(frozen=True) 

647class ShipAssessment: 

648 tier: int 

649 reviewers: int 

650 window_open: bool 

651 ci_ok: bool | None 

652 merge: MergeDecision 

653 halted: bool = False # pause mode + outside window ⇒ pipeline halted 

654 bypassed_window: bool = False # hotfix merged outside the window (audited) 

655 review_contract: dict[str, Any] | None = None 

656 #: Did any check report? False == the rollup was empty (nothing verified this 

657 #: head); None == keel could not find out. Distinct from ``ci_ok`` (#675). 

658 ci_ran: bool | None = None 

659 #: Declared ``knobs.ci_workflows`` that produced no check for this head. 

660 missing_workflows: tuple[str, ...] = () 

661 #: The resolved ``knobs.team`` assignment: who implements, gates, reviews, juries. 

662 assignment: dict[str, Any] | None = None 

663 #: What ``policy_pack.capture.learning.source`` retrieved for this task (#1155), 

664 #: as :func:`keel.capture.learning_retrieval_as_dict` builds it. ``None`` for 

665 #: every caller that measured nothing, which is not the same as a project whose 

666 #: directory is empty — that one retrieves and finds no hits. 

667 learnings: dict[str, Any] | None = None 

668 

669 

670def assess( 

671 *, 

672 changed_files: list[str] | None, 

673 gate_verdict: Verdict, 

674 tier3_globs: tuple[str, ...] = (), 

675 docs_globs: tuple[str, ...] = (), 

676 allowlist_globs: tuple[str, ...] = (), 

677 patches: dict[str, str] | None = None, 

678 timezone: str | None = None, 

679 merge_window: str | None = None, 

680 merge_window_mode: str = "freeze", 

681 ci_conclusion: str | None = None, 

682 ci_check_names: Sequence[str] | None = None, 

683 ci_workflow_names: Sequence[str] | None = None, 

684 ci_workflows: dict[str, str] | None = None, 

685 now=None, 

686 is_blocker: bool = False, 

687 unrun_blocking_gates: tuple[str, ...] = (), 

688 reviewer_override: int | None = None, 

689 review_comments: str = "inline", 

690 gates: tuple[str, ...] = (), 

691 policy_pack: dict[str, Any] | None = None, 

692 jury: bool = False, 

693 no_jury: bool = False, 

694 jury_advisory: bool = False, 

695 team: team_policy.TeamPolicy | None = None, 

696 legacy_agents: dict[str, team_policy.Seat] | None = None, 

697 role: str | None = None, 

698 delegate: str | None = None, 

699 review_delegates: Sequence[str] = (), 

700 #: ``--team`` / ``--effort``: the bench this run is staffed from (#1017). Resolved 

701 #: here as well as in ``cli._review_assignment`` because ``keel ship`` *replaces* the 

702 #: planned assignment with this one once the real tier is known — dropping them here 

703 #: published a ship contract whose team disagreed with the plan the operator read. 

704 team_profile: str | None = None, 

705 effort: str | None = None, 

706 host_agent: str = team_policy.HOST_DEFAULT, 

707 require_distinct_vendors: bool | None = None, 

708 #: The s7 panel-availability probe (#1066), measured by 

709 #: :func:`keel.providerprobe.jury_availability` and passed in for the same reason 

710 #: ``team_profile``/``effort`` are: ``keel ship`` **replaces** the planned assignment 

711 #: with this one once the real tier is known, so a measurement that stopped here would 

712 #: publish a ship contract naming a panel the plan had already found unstaffable. 

713 jury_availability: Mapping[str, Any] | None = None, 

714 #: The retrieved past learnings for this task (#1155), measured in `cli` for the 

715 #: same reason ``jury_availability`` is: reading a directory is I/O, and this 

716 #: function is pure. ``None`` means nothing was retrieved *or attempted*. 

717 learnings: Mapping[str, Any] | None = None, 

718) -> ShipAssessment: 

719 """The whole deterministic ship decision in one place: tier → reviewers, window, 

720 CI, and the final merge action. Pure — identical inputs give identical output. 

721 

722 ``merge_window_mode`` 'pause' halts the pipeline outside the window; 'freeze' 

723 (default) only blocks the merge. ``is_blocker`` (a hotfix) bypasses the window — 

724 but never the findings or a failing CI. 

725 

726 ``patches`` is the per-file diff, keyed by path. Without it this classified 

727 from filenames alone and so could not apply the diff-based TIER-3 downgrade, 

728 which made the assessment a human reads disagree with the evidence gate that 

729 enforces it (#845). ``None`` keeps the old behaviour — no diff is no evidence, 

730 and the path decides. 

731 

732 ``changed_files`` is ``None`` when git could not be read (as 

733 :func:`keel.git.changed_files` reports it), which is deliberately *not* the same 

734 as ``[]``. An empty list classifies as the default tier; an unreadable one 

735 classifies fail-closed at :data:`keel.classify.UNKNOWN_TIER`, so a change nobody 

736 could see never buys itself a lighter review contract.""" 

737 tier = ( 

738 classify.UNKNOWN_TIER 

739 if changed_files is None 

740 else classify.tier_for_files( 

741 changed_files, 

742 tier3_globs=tier3_globs, 

743 docs_globs=docs_globs, 

744 allowlist_globs=allowlist_globs, 

745 patches=patches, 

746 ) 

747 ) 

748 check_reviewer_override(reviewer_override) 

749 assignment = team_policy.resolve_assignment( 

750 team if team is not None else team_policy.TeamPolicy(), 

751 tier=tier, 

752 role=role, 

753 default_count=reviewer_count(tier), 

754 reviewer_override=reviewer_override, 

755 delegate=delegate, 

756 review_delegates=review_delegates, 

757 host_agent=host_agent, 

758 legacy=legacy_agents, 

759 jury_disabled=no_jury, 

760 jury_advisory=jury_advisory, 

761 team_profile=team_profile, 

762 effort=effort, 

763 jury_availability=jury_availability, 

764 ) 

765 reviewers = assignment["reviewer_count"] 

766 window_open = ( 

767 is_merge_open(timezone, merge_window, now=now) if (timezone and merge_window) else True 

768 ) 

769 halted = (merge_window_mode == "pause") and not window_open and not is_blocker 

770 ci_ok = ci_passing(ci_conclusion) 

771 ran = ci_ran(ci_conclusion) 

772 docs_only = changed_files is not None and classify.is_docs_only(list(changed_files), docs_globs) 

773 missing = () if docs_only else missing_ci_workflows(ci_workflow_names, ci_workflows) 

774 if ci_ok is False: 

775 merge = MergeDecision("block", "CI failing") 

776 elif ran is False and not docs_only: 

777 # Fail closed, and say which it was: an operator needs "nothing verified 

778 # this commit" to read differently from "a check went red". 

779 # 

780 # The docs-only carve-out is not a softening — it is what `keel merge` 

781 # already applies to its own `no-checks` state (cli._ci_state), and this 

782 # assessment must not contradict the gate it is predicting. A docs-only 

783 # change legitimately matches no workflow's path filter; anything else 

784 # with an empty rollup was simply never verified. 

785 merge = MergeDecision("block", "no CI ran — nothing verified this commit") 

786 elif missing: 

787 merge = MergeDecision("block", f"declared CI workflow(s) never ran: {', '.join(missing)}") 

788 else: 

789 merge = decide_merge( 

790 gate_verdict, 

791 window_open=window_open, 

792 is_blocker=is_blocker, 

793 unrun_blocking_gates=unrun_blocking_gates, 

794 ) 

795 bypassed = is_blocker and not window_open and merge.action == "merge" 

796 review_contract = resolve_review_contract( 

797 tier=tier, 

798 reviewer_override=reviewer_override, 

799 review_comments=review_comments, 

800 gates=gates, 

801 policy_pack=policy_pack, 

802 jury=jury, 

803 no_jury=no_jury, 

804 jury_advisory=jury_advisory, 

805 require_distinct_vendors=require_distinct_vendors, 

806 assignment=assignment, 

807 learnings=learnings, 

808 ) 

809 return ShipAssessment( 

810 tier, 

811 reviewers, 

812 window_open, 

813 ci_ok, 

814 merge, 

815 halted, 

816 bypassed, 

817 review_contract, 

818 ran, 

819 missing, 

820 assignment, 

821 None if learnings is None else dict(learnings), 

822 )