Coverage for src/keel/gates.py: 100%

176 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""Plan + run quality gates — built-in gates and project Lego gates, uniformly. 

2 

3A *gate* is anything that can pass/fail and produce findings: the built-in 

4``build`` / ``lint`` / ``jury`` gates (from ``project.yaml``'s ``gates:`` list), 

5plus the project's blocking-capable extension hooks. :func:`plan_gates` 

6turns a config + loaded extensions into an ordered list of :class:`GateSpec`; 

7:func:`run_gates` executes them through an injected ``runner`` with fail-soft 

8semantics, normalising everything into :class:`keel.findings.Finding`. 

9""" 

10 

11from __future__ import annotations 

12 

13from collections.abc import Callable, Sequence 

14from dataclasses import dataclass, replace 

15from typing import TYPE_CHECKING 

16 

17from . import revertcheck, tdd 

18from .findings import Finding 

19 

20if TYPE_CHECKING: # pragma: no cover 

21 from .config import ProjectConfig 

22 from .extensions import Extension 

23 

24#: Built-in gate names accepted in ``project.yaml``'s ``gates:`` list. 

25#: 

26#: ``tdd-order`` is deliberately not among them: it is not a gate a project *lists*, it 

27#: is the gate ``implement_mode: tdd`` brings with it. Naming it here would let a project 

28#: ask for the verification of a commit order it never asked its implementer to produce. 

29#: 

30#: ``revert-check`` (#1289) is listed, and **opt-in**: it re-runs the test command once per 

31#: production change, so a project turns it on knowing the cost (:mod:`keel.revertcheck`). 

32BUILTIN_GATES: tuple[str, ...] = ("build", "lint", "jury", revertcheck.GATE_ID) 

33 

34#: Directories a source scanner must not walk, as **prefix-independent globs**. 

35#: 

36#: ``bandit -r .`` walks everything under the working directory, including trees 

37#: version control is told to ignore: an installed ``.venv`` (its dependencies 

38#: alone produce hundreds of findings) and nested checkouts under a harness's 

39#: worktree directory. Test code is excluded on purpose — bandit's heuristics 

40#: target application code, and tests legitimately use temp paths, ``urlopen``, 

41#: and subprocesses, so their findings are false by construction and crowd out 

42#: real ones. 

43#: 

44#: The patterns are globs rather than fixed paths because a fixed ``./tests`` 

45#: does not match ``./.claude/worktrees/<name>/tests`` — the same prefix-anchoring 

46#: mistake as the coverage ``omit`` pattern in #820. 

47_SCAN_EXCLUDE_GLOBS = "*/tests/*,*/.venv/*,*/venv/*,*/node_modules/*,*/site-packages/*" 

48 

49#: Declarative security & SAST presets supported in ``policy_pack.presets``. 

50POLICY_PACK_PRESETS: dict[str, tuple[str, str, str, str]] = { 

51 # preset: (gate_id, phase, on_fail, run_cmd) 

52 "gitleaks": ("gitleaks", "guard", "block", "gitleaks detect --no-git -v"), 

53 "semgrep": ("semgrep", "test", "suggest", "semgrep scan"), 

54 "bandit": ("bandit", "test", "suggest", f"bandit -r . -ll -x '{_SCAN_EXCLUDE_GLOBS}'"), 

55 "trivy": ("trivy", "test", "warn", "trivy fs ."), 

56} 

57 

58#: The finding an unconfigured ``build`` gate blocks with (#1328). One of three texts 

59#: about the same state, each worded for where it is read: this is the gate's finding; 

60#: ``keel init`` / ``keel setup`` print ``cli._UNSET_BUILD_NOTE`` on the terminal; the 

61#: scaffolded file carries ``scaffold._UNSET_BUILD_COMMENT`` above ``knobs:``. All three 

62#: name ``knobs.build_gate_cmd``. 

63UNCONFIGURED_BUILD_GATE = ( 

64 "no build gate configured: set knobs.build_gate_cmd in .keel/project.yaml " 

65 "to the command that runs your tests" 

66) 

67 

68#: The id and finding of the outcome a run with **nothing to judge** blocks with (#1364). 

69#: ``gates: []`` (or no ``gates:`` key, which loads the same), or a list whose only entry 

70#: plans nothing — ``lint`` with no ``knobs.lint_cmd`` — with no extension or preset 

71#: adding a gate, used to run zero gates, print nothing, exit 0, and let a dry ``keel 

72#: ship`` say MERGE, while ``keel merge`` refused the empty record. The id is the config 

73#: key, so the finding reads ``gates: …`` wherever findings are printed. 

74NO_GATES_ID = "gates" 

75#: 

76#: The remedy names only gates that judge wherever they run. ``jury`` is not one of them: 

77#: with no ``jury`` binary on the host, or an empty diff, the jury gate judges nothing 

78#: (#1368 review) — it reports ``SKIPPED``, and a plan with no other gate blocks on it 

79#: (:func:`lone_jury_cannot_judge`, #1369). 

80NO_GATES_PLANNED = ( 

81 "no gate configured: gates: in .keel/project.yaml plans nothing to run and no " 

82 "extension adds a gate — list build (with knobs.build_gate_cmd) or lint (with " 

83 "knobs.lint_cmd), or add a gate extension" 

84) 

85 

86#: The finding an unconfigured built-in ``lint`` gate fails with. :func:`plan_gates` does 

87#: not plan ``lint`` without a command, so only a spec built elsewhere reaches it; it names 

88#: the knob all the same, as every knob-backed command gate's finding does. 

89UNCONFIGURED_LINT_GATE = ( 

90 "no lint command configured: set knobs.lint_cmd in .keel/project.yaml " 

91 "to the command that lints your code, or remove lint from gates:" 

92) 

93 

94#: Built-in command gates whose command is a knob -> the finding naming that knob. 

95_UNCONFIGURED_BUILTIN: dict[str, str] = { 

96 "build": UNCONFIGURED_BUILD_GATE, 

97 "lint": UNCONFIGURED_LINT_GATE, 

98} 

99 

100#: ``GateSpec.source`` prefixes keel itself writes; any other source is an extension file. 

101_KEEL_SOURCES: tuple[str, ...] = ("builtin", "policy_pack:", "implement_mode:") 

102 

103# A failed gate with no explicit findings is reported at this severity. 

104_ON_FAIL_SEVERITY: dict[str, str] = {"block": "major", "suggest": "minor", "warn": "nit"} 

105 

106 

107class GateError(ValueError): 

108 """Raised when a config references an unknown built-in gate.""" 

109 

110 

111#: The backbone steps a gate can run at. `keel run-gates --phases` validates against 

112#: this, so a typo is refused rather than scoping the run to nothing (#1172). 

113BACKBONE_PHASES: tuple[str, ...] = ("guard", "test", "pre-merge") 

114 

115 

116@dataclass(frozen=True) 

117class GateSpec: 

118 """A planned gate. ``phase`` is the backbone step it runs at.""" 

119 

120 id: str 

121 kind: str # command | agentic | builtin 

122 phase: str # backbone step name, e.g. "guard", "test", or "pre-merge" 

123 on_fail: str # block | suggest | warn 

124 run: str | None = None 

125 prompt: str | None = None 

126 agent: str = "inherit" 

127 source: str = "builtin" 

128 required_capabilities: tuple[str, ...] = () 

129 optional_capabilities: tuple[str, ...] = () 

130 #: Resolved wall-clock limit for a ``command`` gate, in seconds. ``None`` means 

131 #: the runner's own fallback applies (a spec built outside :func:`plan_gates`). 

132 timeout: int | None = None 

133 

134 

135@dataclass(frozen=True) 

136class PanelReuse: 

137 """Where a jury gate's verdict came from when keel did not convene a panel (#1437). 

138 

139 The ``keel.jury-verdict.v1`` comment the head's panel already posted: its GitHub id 

140 and URL when the payload carried them, and the head it is pinned to. 

141 """ 

142 

143 head_sha: str 

144 comment_id: int | None = None 

145 url: str | None = None 

146 

147 def as_dict(self) -> dict[str, object]: 

148 return {"head_sha": self.head_sha, "comment_id": self.comment_id, "url": self.url} 

149 

150 

151@dataclass(frozen=True) 

152class GateOutcome: 

153 """Result of running one gate.""" 

154 

155 gate: str 

156 ok: bool 

157 findings: tuple[Finding, ...] = () 

158 error: str | None = None 

159 skipped: bool = False 

160 #: True when this gate was killed by its wall-clock limit rather than returning a 

161 #: verdict. Purely descriptive: a timed-out gate is still ``ok=False`` with an 

162 #: unchanged severity, so it blocks the merge exactly as a failure does. Only the 

163 #: label and the operator-facing explanation differ — a hanging command is a real 

164 #: defect and must stay red. 

165 timed_out: bool = False 

166 #: True when *this runner did not execute the gate at all* — an ``agentic`` gate 

167 #: reached the command-only runner, which the agent-dispatch layer runs instead. 

168 #: Distinct from ``ok`` on purpose: "not my job" must never be recorded as "ran and 

169 #: passed", or a blocking review gate nobody executed would authorize the merge. 

170 #: ``ok`` stays True so a soft gate does not spuriously fail the run; consumers that 

171 #: certify (see :func:`keel.ledger.record_gates_passed`) must refuse a *blocking* 

172 #: gate that was never run. 

173 not_run: bool = False 

174 #: The gate's declared severity (``block`` / ``suggest`` / ``warn``), carried from 

175 #: its :class:`GateSpec` so a consumer reading only outcomes can tell whether a 

176 #: ``not_run`` gate was one the project required. 

177 on_fail: str = "block" 

178 #: True when the gate **cannot judge**: a ``command`` gate with no command (an unset 

179 #: or blank ``knobs.build_gate_cmd``), or the :data:`NO_GATES_ID` outcome of a run 

180 #: that planned nothing. Always ``ok=False``. Distinct from an ordinary failure 

181 #: because no implementer can turn it green — only the project's config can — so the 

182 #: s4 loop stops on it at once instead of spending its budget (#1364). 

183 unconfigured: bool = False 

184 #: Set on the jury gate when its verdict was **reused** from the panel already posted 

185 #: for this head rather than from a panel keel convened (#1437). ``None`` means the 

186 #: runner produced the outcome itself. Descriptive only: a reused outcome is judged by 

187 #: ``ok`` and its findings exactly like one keel ran. 

188 reused_from: PanelReuse | None = None 

189 

190 

191# runner(spec) -> (ok, findings[, timed_out[, not_run[, skipped]]]). May raise; run_gates 

192# handles it fail-soft. The shorter forms stay supported for runners that cannot time out, 

193# that execute every gate they are given, or whose gates always judge. ``skipped`` is a 

194# gate that reached its runner and judged nothing (the jury with no CLI, #1369): reported 

195# ``SKIPPED``, never ``ok``, and honoured only on a passing result. 

196GateRunner = Callable[ 

197 [GateSpec], 

198 "tuple[bool, list[Finding]] | tuple[bool, list[Finding], bool] " 

199 "| tuple[bool, list[Finding], bool, bool] " 

200 "| tuple[bool, list[Finding], bool, bool, bool]", 

201] 

202 

203 

204def plan_gates( 

205 config: ProjectConfig, 

206 loaded: dict[str, list[Extension]], 

207 *, 

208 implement_mode: str | None = None, 

209) -> tuple[GateSpec, ...]: 

210 """Order gates by backbone phase: guard, built-in test gates, test hooks, pre-merge. 

211 

212 ``implement_mode`` is the resolved s4 profile for *this run* (:func:`keel.tdd.resolve_mode`); 

213 ``None`` reads the project's ``knobs.implement_mode``, which is what every caller that 

214 has no per-run ``--tdd`` flag wants. In ``tdd`` mode the pure :data:`keel.tdd.GATE_ID` 

215 gate is appended last at the ``test`` phase: it is the only gate whose verdict depends 

216 on the other gates' (the branch must be test-first **and** green), so it is planned to 

217 run after them. 

218 

219 Every gate that shells out gets its wall-clock ``timeout`` resolved here, so the 

220 planner is the single place budgets are decided: 

221 

222 * ``command`` gates, most specific first — the extension's own ``timeout:`` 

223 frontmatter → ``knobs.gate_timeout_s`` → :data:`keel.model.DEFAULT_GATE_TIMEOUT_S`; 

224 * the ``jury`` builtin, which also shells out (via ``run_argv``) — 

225 ``knobs.jury_timeout_s``, kept separate because a cross-vendor panel and a test 

226 suite have unrelated runtimes. 

227 

228 ``agentic`` gates carry ``None``: the agent-dispatch layer runs those, nothing 

229 shells out for them, and a number there would advertise a limit never applied. 

230 """ 

231 project_timeout = config.knobs.gate_timeout_s 

232 

233 def _timeout_for(e: Extension) -> int | None: 

234 if e.kind != "command": 

235 return None 

236 return e.timeout if e.timeout is not None else project_timeout 

237 

238 specs: list[GateSpec] = [] 

239 presets = ( 

240 tuple(config.policy_pack.get("presets", ())) if isinstance(config.policy_pack, dict) else () 

241 ) 

242 

243 for e in loaded.get("guard", []): 

244 specs.append( 

245 GateSpec( 

246 e.id, 

247 e.kind, 

248 "guard", 

249 e.on_fail, 

250 run=e.run, 

251 prompt=e.prompt, 

252 agent=e.agent, 

253 source=e.source, 

254 required_capabilities=e.required_capabilities, 

255 optional_capabilities=e.optional_capabilities, 

256 timeout=_timeout_for(e), 

257 ) 

258 ) 

259 

260 if "gitleaks" in presets: 

261 gid, phase, on_fail, run_cmd = POLICY_PACK_PRESETS["gitleaks"] 

262 specs.append( 

263 GateSpec( 

264 gid, 

265 "command", 

266 phase, 

267 on_fail, 

268 run=run_cmd, 

269 source="policy_pack:preset:gitleaks", 

270 timeout=project_timeout, 

271 ) 

272 ) 

273 

274 for name in config.gates: 

275 if name == "build": 

276 specs.append( 

277 GateSpec( 

278 "build", 

279 "command", 

280 "test", 

281 "block", 

282 run=config.knobs.build_gate_cmd, 

283 timeout=project_timeout, 

284 ) 

285 ) 

286 elif name == "lint": 

287 # lint is optional: absent, empty or blank means off. A blank command is not a 

288 # command, and planning one only to fail it would block every run for a key 

289 # that says "no lint" (#1368 review). 

290 if (config.knobs.lint_cmd or "").strip(): 

291 specs.append( 

292 GateSpec( 

293 "lint", 

294 "command", 

295 "test", 

296 "block", 

297 run=config.knobs.lint_cmd, 

298 timeout=project_timeout, 

299 ) 

300 ) 

301 elif name == "jury": 

302 specs.append( 

303 GateSpec("jury", "builtin", "test", "block", timeout=config.knobs.jury_timeout_s) 

304 ) 

305 elif name == revertcheck.GATE_ID: 

306 # `pre-merge`, so the s4 loop (`--phases guard,test`) defers it rather than 

307 # paying for a revert per change on every iteration; s8 runs every phase. 

308 # No `timeout`: its bounds are `knobs.revert_check`'s, applied per run. 

309 specs.append(GateSpec(revertcheck.GATE_ID, "builtin", "pre-merge", "block")) 

310 else: 

311 raise GateError( 

312 f"unknown built-in gate {name!r}; valid: {', '.join(BUILTIN_GATES)} " 

313 "(project gates belong in extension slots, not in gates:)" 

314 ) 

315 

316 for preset_name in ("semgrep", "bandit", "trivy"): 

317 if preset_name in presets: 

318 gid, phase, on_fail, run_cmd = POLICY_PACK_PRESETS[preset_name] 

319 specs.append( 

320 GateSpec( 

321 gid, 

322 "command", 

323 phase, 

324 on_fail, 

325 run=run_cmd, 

326 source=f"policy_pack:preset:{preset_name}", 

327 timeout=project_timeout, 

328 ) 

329 ) 

330 

331 for slot, phase in (("tester", "test"), ("test", "test"), ("pre-merge", "pre-merge")): 

332 for e in loaded.get(slot, []): 

333 specs.append( 

334 GateSpec( 

335 e.id, 

336 e.kind, 

337 phase, 

338 e.on_fail, 

339 run=e.run, 

340 prompt=e.prompt, 

341 agent=e.agent, 

342 source=e.source, 

343 required_capabilities=e.required_capabilities, 

344 optional_capabilities=e.optional_capabilities, 

345 timeout=_timeout_for(e), 

346 ) 

347 ) 

348 mode = implement_mode or config.knobs.implement_mode 

349 if mode == tdd.TDD_MODE: 

350 specs.append( 

351 GateSpec( 

352 tdd.GATE_ID, 

353 "builtin", 

354 "test", 

355 "block", 

356 source=f"implement_mode:{tdd.TDD_MODE}", 

357 ) 

358 ) 

359 return tuple(specs) 

360 

361 

362def command_unset(spec: GateSpec) -> bool: 

363 """Is ``spec`` a ``command`` gate with nothing to run — no command, or only whitespace? 

364 

365 A blank command is not a command: ``sh -c ' '`` exits 0, so a ``" "`` build command 

366 used to report ``ok build`` for a gate that ran nothing (#1364). The schema refuses a 

367 blank ``knobs.build_gate_cmd`` too; this is the same test for any spec, whoever built it. 

368 """ 

369 return spec.kind == "command" and not (spec.run or "").strip() 

370 

371 

372def unconfigured_finding(spec: GateSpec) -> Finding: 

373 """The finding a ``command`` gate with no command fails with (#1328). 

374 

375 Returned by the command runner, and re-applied by :func:`run_gates` after any runner 

376 returns (#1364): a gate outside the run's ``--phases`` scope has to stay ``not_run`` 

377 like any other, so the verdict belongs after the scope test, which only the runner 

378 sees — but a runner that answers "ok, ran" for such a gate must not make it a pass. 

379 

380 It names what to set: the knob for a knob-backed built-in (``build`` -> 

381 ``knobs.build_gate_cmd``, ``lint`` -> ``knobs.lint_cmd``), and ``run:`` in the file for 

382 an extension gate. A spec keel did not plan from either has nothing to name. 

383 """ 

384 if spec.source == "builtin" and spec.id in _UNCONFIGURED_BUILTIN: 

385 message = _UNCONFIGURED_BUILTIN[spec.id] 

386 elif not spec.source.startswith(_KEEL_SOURCES): 

387 message = f"gate {spec.id!r} has no command configured: set run: in {spec.source}" 

388 else: 

389 message = f"gate {spec.id!r} has no command configured" 

390 return Finding(_ON_FAIL_SEVERITY[spec.on_fail], message, spec.id) 

391 

392 

393#: Gates evaluated after the others, because each reads their verdict: ``tdd-order`` 

394#: (the branch must be test-first *and* green) and ``revert-check`` (a revert against a red 

395#: suite proves nothing, so it does not spend a run when the others are red). 

396DEFERRED_GATES: tuple[str, ...] = (revertcheck.GATE_ID, tdd.GATE_ID) 

397 

398 

399def split_deferred( 

400 specs: Sequence[GateSpec], 

401) -> tuple[tuple[GateSpec, ...], tuple[GateSpec, ...]]: 

402 """Split planned gates into "run now" and "run after the rest" (:data:`DEFERRED_GATES`). 

403 

404 These gates read the others' verdict, and a runner cannot: :func:`run_gates` hands each 

405 spec to the runner independently, and may run them concurrently. So the caller runs 

406 the first group, summarises it, and only then evaluates the deferred ones — rather than 

407 depending on a list order that a future ``concurrency > 1`` would quietly invalidate. 

408 """ 

409 now = tuple(spec for spec in specs if spec.id not in DEFERRED_GATES) 

410 later = tuple(spec for spec in specs if spec.id in DEFERRED_GATES) 

411 return now, later 

412 

413 

414def nothing_to_judge(specs: Sequence[GateSpec]) -> GateOutcome | None: 

415 """The blocking outcome for a plan with no gate to judge the change, else ``None``. 

416 

417 Keyed on the **plan**, not on the ``gates:`` key alone: ``gates: []`` beside a 

418 ``tester`` extension is a documented way to run project gates, and that run judges. 

419 Neither deferred gate counts (:data:`DEFERRED_GATES`): ``tdd-order`` reads the *other* 

420 gates' verdict, and "green" over no gates is vacuous; ``revert-check`` runs at the 

421 ``pre-merge`` phase, so an s4 loop over it alone would judge nothing, and it needs a 

422 suite the other gates proved green. Independent of ``--phases``: this is not a gate 

423 outside the run's scope but the absence of any gate in every scope, which is why it 

424 is reported apart from the planned gates rather than as a ``not_run`` one (#1364). 

425 """ 

426 now, _later = split_deferred(specs) 

427 if now: 

428 return None 

429 finding = Finding(_ON_FAIL_SEVERITY["block"], NO_GATES_PLANNED, NO_GATES_ID) 

430 return GateOutcome(NO_GATES_ID, False, (finding,), unconfigured=True) 

431 

432 

433#: The id of the built-in jury gate, as ``gates:`` lists it. 

434JURY_ID = "jury" 

435 

436#: The finding a jury that could not run blocks with when no other gate is planned (#1369). 

437#: Its remedy names the same gates :data:`NO_GATES_PLANNED` does, for the same reason. 

438LONE_JURY_JUDGED_NOTHING = ( 

439 "no gate judged this change: jury is the only gate planned and it did not run — list " 

440 "build (with knobs.build_gate_cmd) or lint (with knobs.lint_cmd) beside it in gates:" 

441) 

442 

443 

444def lone_jury_cannot_judge( 

445 specs: Sequence[GateSpec], outcomes: Sequence[GateOutcome] 

446) -> list[GateOutcome]: 

447 """Fail a jury that could not run when it is the only gate planned (#1369). 

448 

449 ``specs`` are the gates this run executed (the ``now`` half of :func:`split_deferred`) 

450 and ``outcomes`` their results, in the same order. The jury builtin with no ``jury`` 

451 CLI on the host, or on an empty diff, judges nothing and comes back ``skipped``. 

452 Beside another gate that stays the documented s8 no-op: the other gate judged, the 

453 jury's ``nit`` says it did not, and the run reads ``SKIPPED jury``. With **no** other 

454 gate, nothing judged the change — the #1364 case of a plan with nothing to judge, 

455 reached through a gate that is planned but cannot run — so its outcome becomes a 

456 ``FAIL`` that cannot judge (``unconfigured``): a ``major`` naming what to list beside 

457 it, ahead of the runner's ``nit`` saying why the jury did not run. 

458 

459 Keyed on the plan, like :func:`nothing_to_judge`: any other gate counts, a soft or a 

460 ``not_run`` one included (a plan of soft gates alone is a separate question), and 

461 ``tdd-order`` is not in ``specs`` — it reads the other gates' verdict. 

462 """ 

463 result = list(outcomes) 

464 if len(specs) != 1: 

465 return result 

466 spec, outcome = specs[0], result[0] 

467 if spec.kind != "builtin" or spec.id != JURY_ID or not outcome.skipped or outcome.error: 

468 return result 

469 # The runner's own finding stays beside the new one: it says *why* the jury did not run. 

470 found = (Finding(_ON_FAIL_SEVERITY["block"], LONE_JURY_JUDGED_NOTHING, JURY_ID),) 

471 return [ 

472 GateOutcome( 

473 JURY_ID, False, found + outcome.findings, on_fail=outcome.on_fail, unconfigured=True 

474 ) 

475 ] 

476 

477 

478def mark_jury_reused( 

479 outcomes: Sequence[GateOutcome], reuse: PanelReuse | None 

480) -> list[GateOutcome]: 

481 """Stamp the jury outcome with the posted panel it was read from (#1437). 

482 

483 ``None`` leaves every outcome as it is. A jury outcome the runner did not produce from 

484 ``reuse`` — not run, or rewritten by :func:`lone_jury_cannot_judge` — is never one the 

485 caller handed a reused verdict for, so only an executed jury outcome is stamped. 

486 """ 

487 if reuse is None: 

488 return list(outcomes) 

489 return [ 

490 replace(outcome, reused_from=reuse) 

491 if outcome.gate == JURY_ID and not outcome.not_run and not outcome.unconfigured 

492 else outcome 

493 for outcome in outcomes 

494 ] 

495 

496 

497def run_gates( 

498 specs, 

499 runner: GateRunner, 

500 *, 

501 fail_soft: bool = True, 

502 concurrency: int = 1, 

503) -> list[GateOutcome]: 

504 """Run each gate via ``runner``; normalise to outcomes (fail-soft by default). 

505 

506 When ``concurrency > 1``, independent gates are executed concurrently using 

507 standard library ``concurrent.futures.ThreadPoolExecutor``, while preserving 

508 exact deterministic outcome ordering. 

509 """ 

510 

511 def _run_single(spec: GateSpec) -> GateOutcome: 

512 try: 

513 # tuple() first: the runner contract has always been "any 2-iterable", 

514 # so indexing the raw return would reject a generator that used to work. 

515 result = tuple(runner(spec)) 

516 # Runners that cannot time out may return the 2-tuple form; runners that 

517 # execute every gate they are given may omit the not-run flag. 

518 ok, found = result[0], result[1] 

519 timed_out = result[2] is True if len(result) > 2 else False 

520 not_run = result[3] is True if len(result) > 3 else False 

521 skipped = result[4] is True if len(result) > 4 else False 

522 except Exception as exc: # noqa: BLE001 - fail-soft is the contract 

523 if not fail_soft: 

524 raise 

525 if spec.on_fail == "block": 

526 # A hard gate that errors must still block (can't silently pass). 

527 finding = Finding("major", f"gate {spec.id!r} errored: {exc}", spec.id) 

528 return GateOutcome(spec.id, False, (finding,), error=str(exc), on_fail=spec.on_fail) 

529 # Soft gate broke -> degrade to a no-op (logged), never abort. 

530 return GateOutcome( 

531 spec.id, True, (), error=str(exc), skipped=True, on_fail=spec.on_fail 

532 ) 

533 

534 if not not_run and command_unset(spec): 

535 # Belt and braces (#1364): the unset-command verdict used to live only in 

536 # `command_gate_runner`, so any other runner answering `(True, [])` for every 

537 # spec would pass a gate that ran nothing. Re-checked here, after the runner, 

538 # so a gate the runner scoped out (`--phases`) is still NOT-RUN. 

539 return GateOutcome( 

540 spec.id, 

541 False, 

542 (unconfigured_finding(spec),), 

543 on_fail=spec.on_fail, 

544 unconfigured=True, 

545 ) 

546 found = tuple(found) 

547 if ok: 

548 return GateOutcome( 

549 spec.id, True, found, skipped=skipped, not_run=not_run, on_fail=spec.on_fail 

550 ) 

551 if not found: 

552 sev = _ON_FAIL_SEVERITY[spec.on_fail] 

553 found = (Finding(sev, f"gate {spec.id!r} failed", spec.id),) 

554 # ok stays False for a timeout: the merge gate is unchanged, only the label. 

555 # not_run rides along on this branch too: dropping it would let a future 

556 # runner that reports a not-run gate as *failing* certify the merge anyway. 

557 return GateOutcome( 

558 spec.id, False, found, timed_out=timed_out, not_run=not_run, on_fail=spec.on_fail 

559 ) 

560 

561 spec_list = list(specs) 

562 if concurrency <= 1 or len(spec_list) <= 1: 

563 return [_run_single(s) for s in spec_list] 

564 

565 from concurrent.futures import ThreadPoolExecutor 

566 

567 with ThreadPoolExecutor(max_workers=concurrency) as executor: 

568 return list(executor.map(_run_single, spec_list)) 

569 

570 

571def unrun_blocking(outcomes: list[GateOutcome]) -> tuple[str, ...]: 

572 """Names of ``on_fail: block`` gates this run did not execute, in outcome order.""" 

573 return tuple(o.gate for o in outcomes if o.not_run and o.on_fail == "block") 

574 

575 

576def apply_recorded_results( 

577 outcomes: list[GateOutcome], results: dict[str, str] 

578) -> tuple[list[GateOutcome], list[str]]: 

579 """Fold externally-executed gate verdicts into ``outcomes``. 

580 

581 ``results`` maps a gate id to ``"pass"`` or ``"fail"``. It exists because the 

582 command-only runner cannot execute ``agentic`` gates — the agent-dispatch layer 

583 does — and without a way to report back, such a gate stays ``not_run`` forever and 

584 :func:`keel.ledger.record_gates_passed` can never certify the run. That would make 

585 a blocking agentic gate a permanent merge block rather than a gate. 

586 

587 **Only a ``not_run`` outcome is replaced.** A gate keel executed has a measured 

588 verdict, and letting a recorded one override it would turn this channel into a way 

589 to certify a run whose gates were observed failing — the same fail-open this whole 

590 series exists to close, arriving from the other direction. Results naming an 

591 executed gate are returned in ``rejected`` so the caller can refuse loudly rather 

592 than silently discard them. 

593 

594 A recorded result clears ``not_run``, because the gate *was* run; a ``fail`` 

595 additionally produces a finding at the gate's declared severity, exactly as an 

596 in-process failure would. A not-run gate can be neither timed out nor skipped, so 

597 the rebuilt outcome carries neither. 

598 

599 Returns ``(outcomes, rejected)``. Ids matching no outcome at all are left to the 

600 CLI, which validates them against the plan. 

601 """ 

602 applied: list[GateOutcome] = [] 

603 rejected: list[str] = [] 

604 for outcome in outcomes: 

605 verdict = results.get(outcome.gate) 

606 if verdict is None: 

607 applied.append(outcome) 

608 continue 

609 if not outcome.not_run: 

610 rejected.append(outcome.gate) 

611 applied.append(outcome) 

612 continue 

613 if verdict == "pass": 

614 applied.append( 

615 GateOutcome( 

616 outcome.gate, 

617 True, 

618 outcome.findings, 

619 error=outcome.error, 

620 on_fail=outcome.on_fail, 

621 ) 

622 ) 

623 continue 

624 found = outcome.findings or ( 

625 Finding( 

626 _ON_FAIL_SEVERITY[outcome.on_fail], 

627 f"gate {outcome.gate!r} failed (reported by the dispatching agent)", 

628 outcome.gate, 

629 ), 

630 ) 

631 applied.append( 

632 GateOutcome(outcome.gate, False, found, error=outcome.error, on_fail=outcome.on_fail) 

633 ) 

634 return applied, rejected 

635 

636 

637def collect_findings(outcomes: list[GateOutcome]) -> list[Finding]: 

638 """Flatten all findings across gate outcomes (in outcome order).""" 

639 out: list[Finding] = [] 

640 for o in outcomes: 

641 out.extend(o.findings) 

642 return out