Coverage for src/keel/gates.py: 100%
176 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""Plan + run quality gates — built-in gates and project Lego gates, uniformly.
3A *gate* is anything that can pass/fail and produce findings: the built-in
4``build`` / ``lint`` / ``jury`` gates (from ``project.yaml``'s ``gates:`` list),
5plus the project's blocking-capable extension hooks. :func:`plan_gates`
6turns a config + loaded extensions into an ordered list of :class:`GateSpec`;
7:func:`run_gates` executes them through an injected ``runner`` with fail-soft
8semantics, normalising everything into :class:`keel.findings.Finding`.
9"""
11from __future__ import annotations
13from collections.abc import Callable, Sequence
14from dataclasses import dataclass, replace
15from typing import TYPE_CHECKING
17from . import revertcheck, tdd
18from .findings import Finding
20if TYPE_CHECKING: # pragma: no cover
21 from .config import ProjectConfig
22 from .extensions import Extension
24#: Built-in gate names accepted in ``project.yaml``'s ``gates:`` list.
25#:
26#: ``tdd-order`` is deliberately not among them: it is not a gate a project *lists*, it
27#: is the gate ``implement_mode: tdd`` brings with it. Naming it here would let a project
28#: ask for the verification of a commit order it never asked its implementer to produce.
29#:
30#: ``revert-check`` (#1289) is listed, and **opt-in**: it re-runs the test command once per
31#: production change, so a project turns it on knowing the cost (:mod:`keel.revertcheck`).
32BUILTIN_GATES: tuple[str, ...] = ("build", "lint", "jury", revertcheck.GATE_ID)
34#: Directories a source scanner must not walk, as **prefix-independent globs**.
35#:
36#: ``bandit -r .`` walks everything under the working directory, including trees
37#: version control is told to ignore: an installed ``.venv`` (its dependencies
38#: alone produce hundreds of findings) and nested checkouts under a harness's
39#: worktree directory. Test code is excluded on purpose — bandit's heuristics
40#: target application code, and tests legitimately use temp paths, ``urlopen``,
41#: and subprocesses, so their findings are false by construction and crowd out
42#: real ones.
43#:
44#: The patterns are globs rather than fixed paths because a fixed ``./tests``
45#: does not match ``./.claude/worktrees/<name>/tests`` — the same prefix-anchoring
46#: mistake as the coverage ``omit`` pattern in #820.
47_SCAN_EXCLUDE_GLOBS = "*/tests/*,*/.venv/*,*/venv/*,*/node_modules/*,*/site-packages/*"
49#: Declarative security & SAST presets supported in ``policy_pack.presets``.
50POLICY_PACK_PRESETS: dict[str, tuple[str, str, str, str]] = {
51 # preset: (gate_id, phase, on_fail, run_cmd)
52 "gitleaks": ("gitleaks", "guard", "block", "gitleaks detect --no-git -v"),
53 "semgrep": ("semgrep", "test", "suggest", "semgrep scan"),
54 "bandit": ("bandit", "test", "suggest", f"bandit -r . -ll -x '{_SCAN_EXCLUDE_GLOBS}'"),
55 "trivy": ("trivy", "test", "warn", "trivy fs ."),
56}
58#: The finding an unconfigured ``build`` gate blocks with (#1328). One of three texts
59#: about the same state, each worded for where it is read: this is the gate's finding;
60#: ``keel init`` / ``keel setup`` print ``cli._UNSET_BUILD_NOTE`` on the terminal; the
61#: scaffolded file carries ``scaffold._UNSET_BUILD_COMMENT`` above ``knobs:``. All three
62#: name ``knobs.build_gate_cmd``.
63UNCONFIGURED_BUILD_GATE = (
64 "no build gate configured: set knobs.build_gate_cmd in .keel/project.yaml "
65 "to the command that runs your tests"
66)
68#: The id and finding of the outcome a run with **nothing to judge** blocks with (#1364).
69#: ``gates: []`` (or no ``gates:`` key, which loads the same), or a list whose only entry
70#: plans nothing — ``lint`` with no ``knobs.lint_cmd`` — with no extension or preset
71#: adding a gate, used to run zero gates, print nothing, exit 0, and let a dry ``keel
72#: ship`` say MERGE, while ``keel merge`` refused the empty record. The id is the config
73#: key, so the finding reads ``gates: …`` wherever findings are printed.
74NO_GATES_ID = "gates"
75#:
76#: The remedy names only gates that judge wherever they run. ``jury`` is not one of them:
77#: with no ``jury`` binary on the host, or an empty diff, the jury gate judges nothing
78#: (#1368 review) — it reports ``SKIPPED``, and a plan with no other gate blocks on it
79#: (:func:`lone_jury_cannot_judge`, #1369).
80NO_GATES_PLANNED = (
81 "no gate configured: gates: in .keel/project.yaml plans nothing to run and no "
82 "extension adds a gate — list build (with knobs.build_gate_cmd) or lint (with "
83 "knobs.lint_cmd), or add a gate extension"
84)
86#: The finding an unconfigured built-in ``lint`` gate fails with. :func:`plan_gates` does
87#: not plan ``lint`` without a command, so only a spec built elsewhere reaches it; it names
88#: the knob all the same, as every knob-backed command gate's finding does.
89UNCONFIGURED_LINT_GATE = (
90 "no lint command configured: set knobs.lint_cmd in .keel/project.yaml "
91 "to the command that lints your code, or remove lint from gates:"
92)
94#: Built-in command gates whose command is a knob -> the finding naming that knob.
95_UNCONFIGURED_BUILTIN: dict[str, str] = {
96 "build": UNCONFIGURED_BUILD_GATE,
97 "lint": UNCONFIGURED_LINT_GATE,
98}
100#: ``GateSpec.source`` prefixes keel itself writes; any other source is an extension file.
101_KEEL_SOURCES: tuple[str, ...] = ("builtin", "policy_pack:", "implement_mode:")
103# A failed gate with no explicit findings is reported at this severity.
104_ON_FAIL_SEVERITY: dict[str, str] = {"block": "major", "suggest": "minor", "warn": "nit"}
107class GateError(ValueError):
108 """Raised when a config references an unknown built-in gate."""
111#: The backbone steps a gate can run at. `keel run-gates --phases` validates against
112#: this, so a typo is refused rather than scoping the run to nothing (#1172).
113BACKBONE_PHASES: tuple[str, ...] = ("guard", "test", "pre-merge")
116@dataclass(frozen=True)
117class GateSpec:
118 """A planned gate. ``phase`` is the backbone step it runs at."""
120 id: str
121 kind: str # command | agentic | builtin
122 phase: str # backbone step name, e.g. "guard", "test", or "pre-merge"
123 on_fail: str # block | suggest | warn
124 run: str | None = None
125 prompt: str | None = None
126 agent: str = "inherit"
127 source: str = "builtin"
128 required_capabilities: tuple[str, ...] = ()
129 optional_capabilities: tuple[str, ...] = ()
130 #: Resolved wall-clock limit for a ``command`` gate, in seconds. ``None`` means
131 #: the runner's own fallback applies (a spec built outside :func:`plan_gates`).
132 timeout: int | None = None
135@dataclass(frozen=True)
136class PanelReuse:
137 """Where a jury gate's verdict came from when keel did not convene a panel (#1437).
139 The ``keel.jury-verdict.v1`` comment the head's panel already posted: its GitHub id
140 and URL when the payload carried them, and the head it is pinned to.
141 """
143 head_sha: str
144 comment_id: int | None = None
145 url: str | None = None
147 def as_dict(self) -> dict[str, object]:
148 return {"head_sha": self.head_sha, "comment_id": self.comment_id, "url": self.url}
151@dataclass(frozen=True)
152class GateOutcome:
153 """Result of running one gate."""
155 gate: str
156 ok: bool
157 findings: tuple[Finding, ...] = ()
158 error: str | None = None
159 skipped: bool = False
160 #: True when this gate was killed by its wall-clock limit rather than returning a
161 #: verdict. Purely descriptive: a timed-out gate is still ``ok=False`` with an
162 #: unchanged severity, so it blocks the merge exactly as a failure does. Only the
163 #: label and the operator-facing explanation differ — a hanging command is a real
164 #: defect and must stay red.
165 timed_out: bool = False
166 #: True when *this runner did not execute the gate at all* — an ``agentic`` gate
167 #: reached the command-only runner, which the agent-dispatch layer runs instead.
168 #: Distinct from ``ok`` on purpose: "not my job" must never be recorded as "ran and
169 #: passed", or a blocking review gate nobody executed would authorize the merge.
170 #: ``ok`` stays True so a soft gate does not spuriously fail the run; consumers that
171 #: certify (see :func:`keel.ledger.record_gates_passed`) must refuse a *blocking*
172 #: gate that was never run.
173 not_run: bool = False
174 #: The gate's declared severity (``block`` / ``suggest`` / ``warn``), carried from
175 #: its :class:`GateSpec` so a consumer reading only outcomes can tell whether a
176 #: ``not_run`` gate was one the project required.
177 on_fail: str = "block"
178 #: True when the gate **cannot judge**: a ``command`` gate with no command (an unset
179 #: or blank ``knobs.build_gate_cmd``), or the :data:`NO_GATES_ID` outcome of a run
180 #: that planned nothing. Always ``ok=False``. Distinct from an ordinary failure
181 #: because no implementer can turn it green — only the project's config can — so the
182 #: s4 loop stops on it at once instead of spending its budget (#1364).
183 unconfigured: bool = False
184 #: Set on the jury gate when its verdict was **reused** from the panel already posted
185 #: for this head rather than from a panel keel convened (#1437). ``None`` means the
186 #: runner produced the outcome itself. Descriptive only: a reused outcome is judged by
187 #: ``ok`` and its findings exactly like one keel ran.
188 reused_from: PanelReuse | None = None
191# runner(spec) -> (ok, findings[, timed_out[, not_run[, skipped]]]). May raise; run_gates
192# handles it fail-soft. The shorter forms stay supported for runners that cannot time out,
193# that execute every gate they are given, or whose gates always judge. ``skipped`` is a
194# gate that reached its runner and judged nothing (the jury with no CLI, #1369): reported
195# ``SKIPPED``, never ``ok``, and honoured only on a passing result.
196GateRunner = Callable[
197 [GateSpec],
198 "tuple[bool, list[Finding]] | tuple[bool, list[Finding], bool] "
199 "| tuple[bool, list[Finding], bool, bool] "
200 "| tuple[bool, list[Finding], bool, bool, bool]",
201]
204def plan_gates(
205 config: ProjectConfig,
206 loaded: dict[str, list[Extension]],
207 *,
208 implement_mode: str | None = None,
209) -> tuple[GateSpec, ...]:
210 """Order gates by backbone phase: guard, built-in test gates, test hooks, pre-merge.
212 ``implement_mode`` is the resolved s4 profile for *this run* (:func:`keel.tdd.resolve_mode`);
213 ``None`` reads the project's ``knobs.implement_mode``, which is what every caller that
214 has no per-run ``--tdd`` flag wants. In ``tdd`` mode the pure :data:`keel.tdd.GATE_ID`
215 gate is appended last at the ``test`` phase: it is the only gate whose verdict depends
216 on the other gates' (the branch must be test-first **and** green), so it is planned to
217 run after them.
219 Every gate that shells out gets its wall-clock ``timeout`` resolved here, so the
220 planner is the single place budgets are decided:
222 * ``command`` gates, most specific first — the extension's own ``timeout:``
223 frontmatter → ``knobs.gate_timeout_s`` → :data:`keel.model.DEFAULT_GATE_TIMEOUT_S`;
224 * the ``jury`` builtin, which also shells out (via ``run_argv``) —
225 ``knobs.jury_timeout_s``, kept separate because a cross-vendor panel and a test
226 suite have unrelated runtimes.
228 ``agentic`` gates carry ``None``: the agent-dispatch layer runs those, nothing
229 shells out for them, and a number there would advertise a limit never applied.
230 """
231 project_timeout = config.knobs.gate_timeout_s
233 def _timeout_for(e: Extension) -> int | None:
234 if e.kind != "command":
235 return None
236 return e.timeout if e.timeout is not None else project_timeout
238 specs: list[GateSpec] = []
239 presets = (
240 tuple(config.policy_pack.get("presets", ())) if isinstance(config.policy_pack, dict) else ()
241 )
243 for e in loaded.get("guard", []):
244 specs.append(
245 GateSpec(
246 e.id,
247 e.kind,
248 "guard",
249 e.on_fail,
250 run=e.run,
251 prompt=e.prompt,
252 agent=e.agent,
253 source=e.source,
254 required_capabilities=e.required_capabilities,
255 optional_capabilities=e.optional_capabilities,
256 timeout=_timeout_for(e),
257 )
258 )
260 if "gitleaks" in presets:
261 gid, phase, on_fail, run_cmd = POLICY_PACK_PRESETS["gitleaks"]
262 specs.append(
263 GateSpec(
264 gid,
265 "command",
266 phase,
267 on_fail,
268 run=run_cmd,
269 source="policy_pack:preset:gitleaks",
270 timeout=project_timeout,
271 )
272 )
274 for name in config.gates:
275 if name == "build":
276 specs.append(
277 GateSpec(
278 "build",
279 "command",
280 "test",
281 "block",
282 run=config.knobs.build_gate_cmd,
283 timeout=project_timeout,
284 )
285 )
286 elif name == "lint":
287 # lint is optional: absent, empty or blank means off. A blank command is not a
288 # command, and planning one only to fail it would block every run for a key
289 # that says "no lint" (#1368 review).
290 if (config.knobs.lint_cmd or "").strip():
291 specs.append(
292 GateSpec(
293 "lint",
294 "command",
295 "test",
296 "block",
297 run=config.knobs.lint_cmd,
298 timeout=project_timeout,
299 )
300 )
301 elif name == "jury":
302 specs.append(
303 GateSpec("jury", "builtin", "test", "block", timeout=config.knobs.jury_timeout_s)
304 )
305 elif name == revertcheck.GATE_ID:
306 # `pre-merge`, so the s4 loop (`--phases guard,test`) defers it rather than
307 # paying for a revert per change on every iteration; s8 runs every phase.
308 # No `timeout`: its bounds are `knobs.revert_check`'s, applied per run.
309 specs.append(GateSpec(revertcheck.GATE_ID, "builtin", "pre-merge", "block"))
310 else:
311 raise GateError(
312 f"unknown built-in gate {name!r}; valid: {', '.join(BUILTIN_GATES)} "
313 "(project gates belong in extension slots, not in gates:)"
314 )
316 for preset_name in ("semgrep", "bandit", "trivy"):
317 if preset_name in presets:
318 gid, phase, on_fail, run_cmd = POLICY_PACK_PRESETS[preset_name]
319 specs.append(
320 GateSpec(
321 gid,
322 "command",
323 phase,
324 on_fail,
325 run=run_cmd,
326 source=f"policy_pack:preset:{preset_name}",
327 timeout=project_timeout,
328 )
329 )
331 for slot, phase in (("tester", "test"), ("test", "test"), ("pre-merge", "pre-merge")):
332 for e in loaded.get(slot, []):
333 specs.append(
334 GateSpec(
335 e.id,
336 e.kind,
337 phase,
338 e.on_fail,
339 run=e.run,
340 prompt=e.prompt,
341 agent=e.agent,
342 source=e.source,
343 required_capabilities=e.required_capabilities,
344 optional_capabilities=e.optional_capabilities,
345 timeout=_timeout_for(e),
346 )
347 )
348 mode = implement_mode or config.knobs.implement_mode
349 if mode == tdd.TDD_MODE:
350 specs.append(
351 GateSpec(
352 tdd.GATE_ID,
353 "builtin",
354 "test",
355 "block",
356 source=f"implement_mode:{tdd.TDD_MODE}",
357 )
358 )
359 return tuple(specs)
362def command_unset(spec: GateSpec) -> bool:
363 """Is ``spec`` a ``command`` gate with nothing to run — no command, or only whitespace?
365 A blank command is not a command: ``sh -c ' '`` exits 0, so a ``" "`` build command
366 used to report ``ok build`` for a gate that ran nothing (#1364). The schema refuses a
367 blank ``knobs.build_gate_cmd`` too; this is the same test for any spec, whoever built it.
368 """
369 return spec.kind == "command" and not (spec.run or "").strip()
372def unconfigured_finding(spec: GateSpec) -> Finding:
373 """The finding a ``command`` gate with no command fails with (#1328).
375 Returned by the command runner, and re-applied by :func:`run_gates` after any runner
376 returns (#1364): a gate outside the run's ``--phases`` scope has to stay ``not_run``
377 like any other, so the verdict belongs after the scope test, which only the runner
378 sees — but a runner that answers "ok, ran" for such a gate must not make it a pass.
380 It names what to set: the knob for a knob-backed built-in (``build`` ->
381 ``knobs.build_gate_cmd``, ``lint`` -> ``knobs.lint_cmd``), and ``run:`` in the file for
382 an extension gate. A spec keel did not plan from either has nothing to name.
383 """
384 if spec.source == "builtin" and spec.id in _UNCONFIGURED_BUILTIN:
385 message = _UNCONFIGURED_BUILTIN[spec.id]
386 elif not spec.source.startswith(_KEEL_SOURCES):
387 message = f"gate {spec.id!r} has no command configured: set run: in {spec.source}"
388 else:
389 message = f"gate {spec.id!r} has no command configured"
390 return Finding(_ON_FAIL_SEVERITY[spec.on_fail], message, spec.id)
393#: Gates evaluated after the others, because each reads their verdict: ``tdd-order``
394#: (the branch must be test-first *and* green) and ``revert-check`` (a revert against a red
395#: suite proves nothing, so it does not spend a run when the others are red).
396DEFERRED_GATES: tuple[str, ...] = (revertcheck.GATE_ID, tdd.GATE_ID)
399def split_deferred(
400 specs: Sequence[GateSpec],
401) -> tuple[tuple[GateSpec, ...], tuple[GateSpec, ...]]:
402 """Split planned gates into "run now" and "run after the rest" (:data:`DEFERRED_GATES`).
404 These gates read the others' verdict, and a runner cannot: :func:`run_gates` hands each
405 spec to the runner independently, and may run them concurrently. So the caller runs
406 the first group, summarises it, and only then evaluates the deferred ones — rather than
407 depending on a list order that a future ``concurrency > 1`` would quietly invalidate.
408 """
409 now = tuple(spec for spec in specs if spec.id not in DEFERRED_GATES)
410 later = tuple(spec for spec in specs if spec.id in DEFERRED_GATES)
411 return now, later
414def nothing_to_judge(specs: Sequence[GateSpec]) -> GateOutcome | None:
415 """The blocking outcome for a plan with no gate to judge the change, else ``None``.
417 Keyed on the **plan**, not on the ``gates:`` key alone: ``gates: []`` beside a
418 ``tester`` extension is a documented way to run project gates, and that run judges.
419 Neither deferred gate counts (:data:`DEFERRED_GATES`): ``tdd-order`` reads the *other*
420 gates' verdict, and "green" over no gates is vacuous; ``revert-check`` runs at the
421 ``pre-merge`` phase, so an s4 loop over it alone would judge nothing, and it needs a
422 suite the other gates proved green. Independent of ``--phases``: this is not a gate
423 outside the run's scope but the absence of any gate in every scope, which is why it
424 is reported apart from the planned gates rather than as a ``not_run`` one (#1364).
425 """
426 now, _later = split_deferred(specs)
427 if now:
428 return None
429 finding = Finding(_ON_FAIL_SEVERITY["block"], NO_GATES_PLANNED, NO_GATES_ID)
430 return GateOutcome(NO_GATES_ID, False, (finding,), unconfigured=True)
433#: The id of the built-in jury gate, as ``gates:`` lists it.
434JURY_ID = "jury"
436#: The finding a jury that could not run blocks with when no other gate is planned (#1369).
437#: Its remedy names the same gates :data:`NO_GATES_PLANNED` does, for the same reason.
438LONE_JURY_JUDGED_NOTHING = (
439 "no gate judged this change: jury is the only gate planned and it did not run — list "
440 "build (with knobs.build_gate_cmd) or lint (with knobs.lint_cmd) beside it in gates:"
441)
444def lone_jury_cannot_judge(
445 specs: Sequence[GateSpec], outcomes: Sequence[GateOutcome]
446) -> list[GateOutcome]:
447 """Fail a jury that could not run when it is the only gate planned (#1369).
449 ``specs`` are the gates this run executed (the ``now`` half of :func:`split_deferred`)
450 and ``outcomes`` their results, in the same order. The jury builtin with no ``jury``
451 CLI on the host, or on an empty diff, judges nothing and comes back ``skipped``.
452 Beside another gate that stays the documented s8 no-op: the other gate judged, the
453 jury's ``nit`` says it did not, and the run reads ``SKIPPED jury``. With **no** other
454 gate, nothing judged the change — the #1364 case of a plan with nothing to judge,
455 reached through a gate that is planned but cannot run — so its outcome becomes a
456 ``FAIL`` that cannot judge (``unconfigured``): a ``major`` naming what to list beside
457 it, ahead of the runner's ``nit`` saying why the jury did not run.
459 Keyed on the plan, like :func:`nothing_to_judge`: any other gate counts, a soft or a
460 ``not_run`` one included (a plan of soft gates alone is a separate question), and
461 ``tdd-order`` is not in ``specs`` — it reads the other gates' verdict.
462 """
463 result = list(outcomes)
464 if len(specs) != 1:
465 return result
466 spec, outcome = specs[0], result[0]
467 if spec.kind != "builtin" or spec.id != JURY_ID or not outcome.skipped or outcome.error:
468 return result
469 # The runner's own finding stays beside the new one: it says *why* the jury did not run.
470 found = (Finding(_ON_FAIL_SEVERITY["block"], LONE_JURY_JUDGED_NOTHING, JURY_ID),)
471 return [
472 GateOutcome(
473 JURY_ID, False, found + outcome.findings, on_fail=outcome.on_fail, unconfigured=True
474 )
475 ]
478def mark_jury_reused(
479 outcomes: Sequence[GateOutcome], reuse: PanelReuse | None
480) -> list[GateOutcome]:
481 """Stamp the jury outcome with the posted panel it was read from (#1437).
483 ``None`` leaves every outcome as it is. A jury outcome the runner did not produce from
484 ``reuse`` — not run, or rewritten by :func:`lone_jury_cannot_judge` — is never one the
485 caller handed a reused verdict for, so only an executed jury outcome is stamped.
486 """
487 if reuse is None:
488 return list(outcomes)
489 return [
490 replace(outcome, reused_from=reuse)
491 if outcome.gate == JURY_ID and not outcome.not_run and not outcome.unconfigured
492 else outcome
493 for outcome in outcomes
494 ]
497def run_gates(
498 specs,
499 runner: GateRunner,
500 *,
501 fail_soft: bool = True,
502 concurrency: int = 1,
503) -> list[GateOutcome]:
504 """Run each gate via ``runner``; normalise to outcomes (fail-soft by default).
506 When ``concurrency > 1``, independent gates are executed concurrently using
507 standard library ``concurrent.futures.ThreadPoolExecutor``, while preserving
508 exact deterministic outcome ordering.
509 """
511 def _run_single(spec: GateSpec) -> GateOutcome:
512 try:
513 # tuple() first: the runner contract has always been "any 2-iterable",
514 # so indexing the raw return would reject a generator that used to work.
515 result = tuple(runner(spec))
516 # Runners that cannot time out may return the 2-tuple form; runners that
517 # execute every gate they are given may omit the not-run flag.
518 ok, found = result[0], result[1]
519 timed_out = result[2] is True if len(result) > 2 else False
520 not_run = result[3] is True if len(result) > 3 else False
521 skipped = result[4] is True if len(result) > 4 else False
522 except Exception as exc: # noqa: BLE001 - fail-soft is the contract
523 if not fail_soft:
524 raise
525 if spec.on_fail == "block":
526 # A hard gate that errors must still block (can't silently pass).
527 finding = Finding("major", f"gate {spec.id!r} errored: {exc}", spec.id)
528 return GateOutcome(spec.id, False, (finding,), error=str(exc), on_fail=spec.on_fail)
529 # Soft gate broke -> degrade to a no-op (logged), never abort.
530 return GateOutcome(
531 spec.id, True, (), error=str(exc), skipped=True, on_fail=spec.on_fail
532 )
534 if not not_run and command_unset(spec):
535 # Belt and braces (#1364): the unset-command verdict used to live only in
536 # `command_gate_runner`, so any other runner answering `(True, [])` for every
537 # spec would pass a gate that ran nothing. Re-checked here, after the runner,
538 # so a gate the runner scoped out (`--phases`) is still NOT-RUN.
539 return GateOutcome(
540 spec.id,
541 False,
542 (unconfigured_finding(spec),),
543 on_fail=spec.on_fail,
544 unconfigured=True,
545 )
546 found = tuple(found)
547 if ok:
548 return GateOutcome(
549 spec.id, True, found, skipped=skipped, not_run=not_run, on_fail=spec.on_fail
550 )
551 if not found:
552 sev = _ON_FAIL_SEVERITY[spec.on_fail]
553 found = (Finding(sev, f"gate {spec.id!r} failed", spec.id),)
554 # ok stays False for a timeout: the merge gate is unchanged, only the label.
555 # not_run rides along on this branch too: dropping it would let a future
556 # runner that reports a not-run gate as *failing* certify the merge anyway.
557 return GateOutcome(
558 spec.id, False, found, timed_out=timed_out, not_run=not_run, on_fail=spec.on_fail
559 )
561 spec_list = list(specs)
562 if concurrency <= 1 or len(spec_list) <= 1:
563 return [_run_single(s) for s in spec_list]
565 from concurrent.futures import ThreadPoolExecutor
567 with ThreadPoolExecutor(max_workers=concurrency) as executor:
568 return list(executor.map(_run_single, spec_list))
571def unrun_blocking(outcomes: list[GateOutcome]) -> tuple[str, ...]:
572 """Names of ``on_fail: block`` gates this run did not execute, in outcome order."""
573 return tuple(o.gate for o in outcomes if o.not_run and o.on_fail == "block")
576def apply_recorded_results(
577 outcomes: list[GateOutcome], results: dict[str, str]
578) -> tuple[list[GateOutcome], list[str]]:
579 """Fold externally-executed gate verdicts into ``outcomes``.
581 ``results`` maps a gate id to ``"pass"`` or ``"fail"``. It exists because the
582 command-only runner cannot execute ``agentic`` gates — the agent-dispatch layer
583 does — and without a way to report back, such a gate stays ``not_run`` forever and
584 :func:`keel.ledger.record_gates_passed` can never certify the run. That would make
585 a blocking agentic gate a permanent merge block rather than a gate.
587 **Only a ``not_run`` outcome is replaced.** A gate keel executed has a measured
588 verdict, and letting a recorded one override it would turn this channel into a way
589 to certify a run whose gates were observed failing — the same fail-open this whole
590 series exists to close, arriving from the other direction. Results naming an
591 executed gate are returned in ``rejected`` so the caller can refuse loudly rather
592 than silently discard them.
594 A recorded result clears ``not_run``, because the gate *was* run; a ``fail``
595 additionally produces a finding at the gate's declared severity, exactly as an
596 in-process failure would. A not-run gate can be neither timed out nor skipped, so
597 the rebuilt outcome carries neither.
599 Returns ``(outcomes, rejected)``. Ids matching no outcome at all are left to the
600 CLI, which validates them against the plan.
601 """
602 applied: list[GateOutcome] = []
603 rejected: list[str] = []
604 for outcome in outcomes:
605 verdict = results.get(outcome.gate)
606 if verdict is None:
607 applied.append(outcome)
608 continue
609 if not outcome.not_run:
610 rejected.append(outcome.gate)
611 applied.append(outcome)
612 continue
613 if verdict == "pass":
614 applied.append(
615 GateOutcome(
616 outcome.gate,
617 True,
618 outcome.findings,
619 error=outcome.error,
620 on_fail=outcome.on_fail,
621 )
622 )
623 continue
624 found = outcome.findings or (
625 Finding(
626 _ON_FAIL_SEVERITY[outcome.on_fail],
627 f"gate {outcome.gate!r} failed (reported by the dispatching agent)",
628 outcome.gate,
629 ),
630 )
631 applied.append(
632 GateOutcome(outcome.gate, False, found, error=outcome.error, on_fail=outcome.on_fail)
633 )
634 return applied, rejected
637def collect_findings(outcomes: list[GateOutcome]) -> list[Finding]:
638 """Flatten all findings across gate outcomes (in outcome order)."""
639 out: list[Finding] = []
640 for o in outcomes:
641 out.extend(o.findings)
642 return out