Coverage for src/keel/wizard.py: 100%
428 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""The pure question/answer planner behind every keel `--wizard` (#1018).
3`ship.md` has promised an interactive picker "built from a best-effort tool/model
4probe" since the adapter was written, but no core code existed for it, so every host
5improvised the choices — different questions, different defaults, and options naming
6providers that are not installed on the operator's machine.
8This module is that picker's pure half. It takes two inputs and never performs I/O:
10* the **provider probe report** — the exact document ``keel doctor --providers --json``
11 prints (:func:`keel.providerprobe.build_report`), which is the single source of truth
12 for what is usable here;
13* the current **team policy** (:class:`keel.team.TeamPolicy`, ``knobs.team``), which
14 supplies every default so a wizard run with no answers reproduces today's behaviour.
16Out of it comes a :class:`Resolution`: the implementer/gate/reviewer seats, the jury
17mode and the review-comments mode, rendered either as the literal ``keel ship`` flag
18set (so the adapter passes flags on, exactly as ``ship.md``'s worked example does) or
19as a ``knobs.team`` block ``keel init --wizard`` writes.
21Two properties are load-bearing:
23* **A provider the probe did not mark available is never offered, and can never be
24 selected.** Every question is closed over :class:`Catalog`, and
25 :meth:`Question.normalize` refuses a value outside its own choices — so an answer
26 file, an injected seam and a typed answer are all held to the same list.
27* **Deterministic.** No wall-clock, no randomness, no I/O, no default that depends on
28 the order a dict happened to iterate in. Identical inputs give identical questions,
29 identical defaults and an identical resolution, which is what lets a host replay a
30 wizard run from its recorded answers.
32The interactive half is :func:`run`, which is pure given its ``ask``/``notify`` seams;
33the CLI supplies the real ``input``-based ones and the ``isatty`` guard.
34"""
36from __future__ import annotations
38from collections.abc import Callable, Iterable, Mapping, Sequence
39from dataclasses import dataclass, field
40from typing import Any
42from . import team
43from .team import Seat
44from .vocab import EFFORTS, supports_effort
46#: JSON-stable schema id of :meth:`Resolution.as_dict`.
47SCHEMA_VERSION = "keel.wizard.v1"
49#: Which wizard is being run. ``run`` resolves per-run flags for ``keel ship`` /
50#: ``keel work-block``; ``config`` resolves the ``knobs.team`` block ``keel init``
51#: writes. The questions differ because the *artefacts* differ: a run has one reviewer
52#: bench (``--reviewers`` / ``--review-delegate`` are per-slot flags and the risk tier
53#: is not known until s5 classifies the diff), while a config names one bench **per
54#: tier**. Same planner, same catalogue, same defaults.
55SCOPE_RUN = "run"
56SCOPE_CONFIG = "config"
57SCOPES = (SCOPE_RUN, SCOPE_CONFIG)
59#: First question: take every default, or answer the rest. Mirrors the Quick-start vs
60#: Customize fast path ``ship.md`` documents.
61QUICK_START = "quick-start"
62CUSTOMIZE = "customize"
64#: :attr:`Candidate.source` of a machine-level ``~/.keel/providers.yaml`` entry. Reachable
65#: per run, never nameable in a committed policy — see :func:`committable`.
66REGISTRY_SOURCE = "registry"
68#: Answer meaning "leave this unset" — no model override, no reasoning effort, no gate
69#: review. Spelled once so a caller cannot confuse it with a provider named ``none``:
70#: a provider by that name would still be offered under its own name, and the sentinel
71#: is only ever compared against, never dispatched.
72NONE = "none"
74#: The jury is off. ``gating``/``advisory`` are :data:`keel.team.JURY_MODES`.
75JURY_OFF = "off"
76JURY_ANSWERS = (*team.JURY_MODES, JURY_OFF)
78#: How review findings are posted, matching ``keel ship --review-comments``.
79REVIEW_COMMENT_MODES = ("inline", "summary")
81#: The tier a **run** wizard derives its offered bench at. `keel ship` classifies the
82#: real tier at s1, after the wizard has run, so the offer has to name some tier; this
83#: is the one an empty changeset already classifies as. It is only ever a *default* —
84#: an unanswered bench question emits no `--reviewers`, so the run still gets the bench
85#: its real tier earns (see :meth:`Resolution.flags`).
86RUN_BENCH_TIER = "2"
88#: Reviewer seats a tier gets when nothing else says otherwise — one at tier 1, two at
89#: tier 2, three at tier 3, clamped to the providers that are actually available.
90DEFAULT_BENCH = {"1": 1, "2": 2, "3": 3}
92#: Questions only :data:`SCOPE_CONFIG` asks, because they land in ``knobs.team`` and
93#: ``keel ship`` has **no flag that carries them**. The gate seat has no ``--gate``, and
94#: reasoning effort has no ``--effort`` (``--delegate`` splits ``provider:model`` and
95#: stops there). Asking them in a run produced an answer that changed nothing: the echo
96#: named a gate while the published ``assignment.gate`` still held the policy's — two
97#: documents disagreeing about the same seat, which is the defect #1014 round 2 fixed
98#: for reviewers. A question the run cannot honour is not asked.
99#: **#1049 re-opens ``implement.effort`` for runs**: it adds ``--effort`` (and ``--team``)
100#: to ``keel ship``, at which point a run *can* carry an effort and the key moves back to
101#: the run column — here, and in the table at ``docs/keel/cli.md#ship-wizard-questions``
102#: that ``tests/test_wizard.py`` holds against this tuple.
103CONFIG_ONLY_KEYS = ("implement.effort", "gate.provider")
105#: Every key the planner can ask under any scope or branch. Used to tell a misspelled
106#: ``--wizard-answer`` from a correctly spelled one this run never reaches.
107#: ``tests/test_wizard.py`` walks the planner and fails if a question escapes this list.
108QUESTION_KEYS = (
109 "mode",
110 "implement.provider",
111 "implement.model",
112 "implement.effort",
113 "gate.provider",
114 "jury",
115 "review",
116 "review.1",
117 "review.2",
118 "review.3",
119 "review_comments",
120)
122#: Re-asks a single question gets before the planner stops arguing and takes the
123#: default. A wizard that can loop forever on a stubborn ``ask`` seam is a hang, and a
124#: hang is the one thing ``--wizard`` promises never to be.
125MAX_ATTEMPTS = 3
128def _effort_capable(vendor: str) -> bool:
129 """Can ``vendor`` express a reasoning-effort request in its own spelling?
131 Read from the leaf :mod:`keel.vocab` rather than from :mod:`keel.delegate`: the
132 answer is vocabulary, not dispatch, and importing the executor would drag the whole
133 config graph into :mod:`keel.scaffold` — a module whose entire job is to run before
134 a config exists. Until #1050 that cost a function-local import; now it does not.
135 """
136 return supports_effort(vendor)
139def _efforts() -> tuple[str, ...]:
140 """keel's vendor-neutral effort vocabulary (see :func:`_effort_capable`)."""
141 return tuple(EFFORTS)
144@dataclass(frozen=True)
145class Candidate:
146 """One provider the probe reported as **available**, ready to be offered."""
148 name: str
149 vendor: str
150 transport: str
151 source: str
152 #: Models the provider listed for itself (``agy models``, Ollama ``/api/tags``).
153 models: tuple[str, ...] = ()
154 #: Can it drive git/PR steps itself? Only a ``cli`` transport can.
155 tools: bool = False
157 @property
158 def effort(self) -> bool:
159 """True when a reasoning effort can be asked for on this seat."""
160 return _effort_capable(self.vendor)
162 def detail(self) -> str:
163 """The one-line description shown beside this provider in a question."""
164 parts = [self.transport, self.source]
165 parts.append("tools" if self.tools else "no tools")
166 if self.models:
167 parts.append(f"{len(self.models)} model(s)")
168 return " · ".join(parts)
171@dataclass(frozen=True)
172class Catalog:
173 """Every available provider, in the probe's order. Empty means "nothing usable"."""
175 candidates: tuple[Candidate, ...] = ()
177 @classmethod
178 def from_report(cls, report: Any) -> Catalog:
179 """Build a catalogue from a ``keel doctor --providers`` document.
181 Fail-soft on purpose: the report can arrive from a file an operator wrote or
182 from another keel's ``--json`` output, so a malformed row is skipped rather
183 than raising. A row that is not marked ``available`` is never a candidate —
184 that single rule is what makes an unavailable provider unofferable.
185 """
186 rows = report.get("providers") if isinstance(report, Mapping) else None
187 candidates = []
188 for row in rows if isinstance(rows, list) else ():
189 candidate = _candidate(row)
190 if candidate is not None:
191 candidates.append(candidate)
192 return cls(tuple(candidates))
194 def names(self) -> tuple[str, ...]:
195 return tuple(candidate.name for candidate in self.candidates)
197 def get(self, name: str | None) -> Candidate | None:
198 for candidate in self.candidates:
199 if candidate.name == name:
200 return candidate
201 return None
203 def has(self, name: str | None) -> bool:
204 return self.get(name) is not None
206 def spread(self) -> tuple[Candidate, ...]:
207 """Candidates re-ordered so distinct vendors come first.
209 A default bench of two seats should be two *vendors* where the machine has
210 two — one vendor reviewing twice is one opinion twice, which is the same rule
211 :data:`keel.providers.REVIEW_VENDOR_MINIMUM` states for the jury.
212 """
213 seen: dict[str, Candidate] = {}
214 rest: list[Candidate] = []
215 for candidate in self.candidates:
216 if candidate.vendor in seen:
217 rest.append(candidate)
218 else:
219 seen[candidate.vendor] = candidate
220 return (*seen.values(), *rest)
223def _candidate(row: Any) -> Candidate | None:
224 """One probe row -> a :class:`Candidate`, or ``None`` when it is not offerable."""
225 if not isinstance(row, Mapping) or not row.get("available"):
226 return None
227 name = row.get("name")
228 vendor = row.get("vendor")
229 if not isinstance(name, str) or not name.strip():
230 return None
231 capabilities = row.get("capabilities")
232 capabilities = capabilities if isinstance(capabilities, Mapping) else {}
233 models = row.get("models")
234 return Candidate(
235 name=name.strip(),
236 vendor=vendor.strip() if isinstance(vendor, str) and vendor.strip() else name.strip(),
237 transport=_word(row.get("transport"), "cli"),
238 source=_word(row.get("source"), "builtin"),
239 models=tuple(m.strip() for m in models if isinstance(m, str) and m.strip())
240 if isinstance(models, list)
241 else (),
242 tools=bool(capabilities.get("tools")),
243 )
246def _word(value: Any, fallback: str) -> str:
247 return value.strip() if isinstance(value, str) and value.strip() else fallback
250@dataclass(frozen=True)
251class Choice:
252 """One offered option: its literal answer value and a one-line description."""
254 value: str
255 detail: str = ""
258@dataclass(frozen=True)
259class Question:
260 """One question, closed over the values it will accept."""
262 key: str
263 prompt: str
264 choices: tuple[Choice, ...]
265 default: str
266 help: str = ""
267 #: Comma-separated answer (a reviewer bench); otherwise exactly one value.
268 multi: bool = False
269 max_values: int = 1
271 def values(self) -> tuple[str, ...]:
272 return tuple(choice.value for choice in self.choices)
274 def ordered(self) -> tuple[Choice, ...]:
275 """Choices with the default first — the shape ``ship.md`` asks every question for."""
276 default = [choice for choice in self.choices if choice.value == self.default]
277 return (*default, *(choice for choice in self.choices if choice.value != self.default))
279 def text(self) -> str:
280 """The full prompt block: the question, its help, then the options."""
281 lines = [self.prompt if not self.help else f"{self.prompt} — {self.help}"]
282 for choice in self.ordered():
283 marker = " (default)" if choice.value == self.default else ""
284 detail = f" — {choice.detail}" if choice.detail else ""
285 lines.append(f" {choice.value}{marker}{detail}")
286 if self.multi:
287 lines.append(f" (comma-separate up to {self.max_values})")
288 return "\n".join(lines)
290 def normalize(self, raw: str | None) -> tuple[str | None, str | None]:
291 """``(value, error)`` — exactly one is ``None``. A blank answer is the default."""
292 text = (raw or "").strip()
293 if not text:
294 return self.default, None
295 tokens = [t.strip() for t in text.split(",")] if self.multi else [text]
296 tokens = [token for token in tokens if token]
297 if not tokens:
298 return self.default, None
299 allowed = self.values()
300 unknown = [token for token in tokens if token not in allowed]
301 if unknown:
302 return None, (
303 f"{self.key}: {unknown[0]!r} is not on offer here; choose from {', '.join(allowed)}"
304 )
305 if len(tokens) > self.max_values:
306 return None, f"{self.key}: at most {self.max_values} value(s), got {len(tokens)}"
307 if len(tokens) > 1 and team.JURY_PANEL in tokens:
308 return None, (
309 f"{self.key}: {team.JURY_PANEL!r} is the whole panel and cannot be "
310 "combined with named seats"
311 )
312 return ",".join(tokens), None
315@dataclass(frozen=True)
316class State:
317 """A wizard mid-flight: what is on offer, what the defaults are, what was answered."""
319 catalog: Catalog
320 policy: team.TeamPolicy = field(default_factory=team.TeamPolicy)
321 scope: str = SCOPE_RUN
322 #: Defaults carried in from the parsed flags, so the wizard starts where the
323 #: command line already is.
324 review_comments: str = "inline"
325 jury: str = JURY_OFF
326 delegate: str | None = None
327 #: Questions the operator answered with a value of their own.
328 answers: Mapping[str, str] = field(default_factory=dict)
329 #: Questions the operator was asked and accepted the default for. Deliberately
330 #: **not** the same thing as an answer: an accepted default means "do what this
331 #: command would have done anyway", and :meth:`Resolution.flags` must not then
332 #: materialise that default as an explicit flag. Writing back a default the
333 #: operator never chose is how a quick-start run on a tier-3 change ended up
334 #: passing `--reviewers 2 --no-jury` and quietly dropping a reviewer and the
335 #: gating jury.
336 defaulted: frozenset[str] = frozenset()
338 def _replace(self, **changes: Any) -> State:
339 base = {
340 "catalog": self.catalog,
341 "policy": self.policy,
342 "scope": self.scope,
343 "review_comments": self.review_comments,
344 "jury": self.jury,
345 "delegate": self.delegate,
346 "answers": self.answers,
347 "defaulted": self.defaulted,
348 }
349 return State(**{**base, **changes})
351 def with_answer(self, key: str, value: str) -> State:
352 """Record a value the operator chose. This one *does* become a flag."""
353 return self._replace(answers={**self.answers, key: value})
355 def with_default(self, key: str) -> State:
356 """Record that the operator accepted this question's default: no flag."""
357 return self._replace(defaulted=self.defaulted | {key})
359 def settled(self, key: str) -> bool:
360 """Has this question been put to the operator and disposed of?"""
361 return key in self.answers or key in self.defaulted
363 def questions(self) -> tuple[Question, ...]:
364 """Every question this scope asks, given the answers so far."""
365 return _walk(self)[0]
367 def next_question(self) -> Question | None:
368 """The first question still unanswered, or ``None`` when the wizard is done."""
369 for question in self.questions():
370 if not self.settled(question.key):
371 return question
372 return None
374 def resolve(self) -> Resolution:
375 """The resolved seats/flags for the answers so far (unanswered = default)."""
376 return _walk(self)[1]
379def committable(catalog: Catalog) -> Catalog:
380 """Only the providers a **committed** ``knobs.team`` may name.
382 ``keel validate`` resolves a policy against the built-in vendors and the project's own
383 ``knobs.delegate_profiles`` — never the machine-level ``~/.keel/providers.yaml``,
384 because a policy that validates only on its author's laptop is not a policy
385 (``docs/keel/configuration.md#team``). A registry provider stays reachable *per run*
386 through ``--delegate``, so the run wizard still offers it; the config wizard must not,
387 or the file it writes would fail the very next ``keel validate``.
388 """
389 return Catalog(tuple(c for c in catalog.candidates if c.source != REGISTRY_SOURCE))
392def start(
393 catalog: Catalog,
394 *,
395 policy: team.TeamPolicy | None = None,
396 scope: str = SCOPE_RUN,
397 review_comments: str = "inline",
398 jury: str = JURY_OFF,
399 delegate: str | None = None,
400) -> State:
401 """A fresh :class:`State`. ``jury``/``review_comments``/``delegate`` are the parsed flags."""
402 if scope not in SCOPES:
403 raise ValueError(f"unknown wizard scope {scope!r}; valid: {', '.join(SCOPES)}")
404 return State(
405 catalog=committable(catalog) if scope == SCOPE_CONFIG else catalog,
406 policy=policy if policy is not None else team.TeamPolicy(),
407 scope=scope,
408 review_comments=review_comments if review_comments in REVIEW_COMMENT_MODES else "inline",
409 jury=jury if jury in JURY_ANSWERS else JURY_OFF,
410 delegate=delegate,
411 )
414@dataclass(frozen=True)
415class Resolution:
416 """What the wizard decided: seats, panel, and the flags that express them."""
418 scope: str
419 implement: Seat
420 gate: Seat | None = None
421 #: ``run`` scope: the bench for this run. Always seats, never the jury sentinel —
422 #: `keel ship` spells a bench with ``--reviewers <1|2|3>`` and ``--review-delegate``,
423 #: and neither can say "the panel *is* the review", so a run cannot express one
424 #: (#1015 / #1046 may change that). A tier whose *policy* is the panel keeps it in
425 #: :attr:`review_by_tier`, which only the config scope writes.
426 review: tuple[Seat, ...] = ()
427 #: ``config`` scope: tier -> bench (or the jury panel).
428 review_by_tier: Mapping[str, tuple[Seat, ...] | str] = field(default_factory=dict)
429 jury: str = JURY_OFF
430 review_comments: str = "inline"
431 quick_start: bool = True
432 #: Question keys the operator answered with a value of their own. Everything else
433 #: resolved to a default, and a default is *not* a decision: see :meth:`flags`.
434 answered: frozenset[str] = frozenset()
436 def flags(self) -> tuple[str, ...]:
437 """The literal ``keel ship`` / ``keel work-block`` flag set, in a stable order.
439 **Only an answered question produces a flag.** Every value below also has a
440 resolved default, and materialising those defaults as flags is not neutral —
441 it overrides the very policy they were read from. The reviewer bench a run
442 wizard shows is derived at a nominal tier because the real one is not
443 classified until s5, and the jury default is "whatever the flags and
444 `knobs.team` already say"; writing either back turned a quick-start run on a
445 tier-3 change into `--reviewers 2 --no-jury`, dropping a reviewer and the
446 gating jury. An unanswered question therefore emits nothing at all and the
447 command resolves it exactly as it would have without `--wizard`.
448 """
449 flags: list[str] = []
450 if {"implement.provider", "implement.model"} & self.answered:
451 flags += ["--delegate", seat_token(self.implement)]
452 if "review" in self.answered and self.review:
453 flags += ["--reviewers", str(len(self.review))]
454 for seat in self.review:
455 flags += ["--review-delegate", seat_token(seat)]
456 if "review_comments" in self.answered:
457 flags += ["--review-comments", self.review_comments]
458 if "jury" in self.answered:
459 flags += {
460 "gating": ["--jury"],
461 "advisory": ["--jury-advisory"],
462 JURY_OFF: ["--no-jury"],
463 }[self.jury]
464 return tuple(flags)
466 def team_block(self) -> dict[str, Any]:
467 """The ``knobs.team`` block, in the shape :mod:`keel.team` parses (#1014)."""
468 block: dict[str, Any] = {"implement": {"default": _seat_block(self.implement)}}
469 if self.gate is not None:
470 block["gate"] = {**_seat_block(self.gate), "distinct_from": team.IMPLEMENTER}
471 by_tier = {
472 tier: value if isinstance(value, str) else [_seat_block(s) for s in value]
473 for tier, value in sorted(self.review_by_tier.items())
474 }
475 if by_tier:
476 block["review"] = {"by_tier": by_tier}
477 if self.jury != JURY_OFF:
478 block["jury"] = {"mode": self.jury, "min_vendors": team.DEFAULT_MIN_VENDORS}
479 block["fix"] = {"provider": team.IMPLEMENTER}
480 return block
482 def as_dict(self) -> dict[str, Any]:
483 """JSON-stable echo of everything the wizard resolved."""
484 return {
485 "schema_version": SCHEMA_VERSION,
486 "scope": self.scope,
487 "quick_start": self.quick_start,
488 "flags": list(self.flags()),
489 "implement": self.implement.as_dict(),
490 "gate": None if self.gate is None else self.gate.as_dict(),
491 "review": [seat.as_dict() for seat in self.review],
492 "review_by_tier": {
493 tier: value if isinstance(value, str) else [s.as_dict() for s in value]
494 for tier, value in sorted(self.review_by_tier.items())
495 },
496 "jury": self.jury,
497 "review_comments": self.review_comments,
498 "team": self.team_block(),
499 }
502def seat_token(seat: Seat) -> str:
503 """``provider`` or ``provider:model`` — the spelling ``--delegate`` takes."""
504 return f"{seat.provider}:{seat.model}" if seat.model else seat.provider
507def _seat_block(seat: Seat) -> dict[str, Any]:
508 block: dict[str, Any] = {"provider": seat.provider}
509 if seat.model:
510 block["model"] = seat.model
511 if seat.effort:
512 block["effort"] = seat.effort
513 return block
516def _default_implement(state: State) -> Seat:
517 """The implementer to start from: the flag, then the policy, then what is available.
519 A default that names a provider this machine cannot reach is *not* a default — it
520 would put an unusable option in front of the operator as the recommended one. It
521 degrades to the first available candidate instead.
522 """
523 if state.delegate:
524 seat = team.seat_from_token(state.delegate)
525 if state.catalog.has(seat.provider):
526 return seat
527 if state.policy.implement is not None and state.catalog.has(state.policy.implement.provider):
528 return state.policy.implement
529 return Seat(provider=state.catalog.names()[0])
532def _default_gate(state: State, implementer: str) -> Seat | None:
533 """The gate seat, dropped when it is unavailable or would be the implementer."""
534 gate = state.policy.gate
535 if gate is None or not state.catalog.has(gate.provider) or gate.provider == implementer:
536 return None
537 return gate
540def _default_bench(
541 state: State,
542 tier: str,
543 implementer: str,
544 *,
545 panel_allowed: bool,
546) -> tuple[Seat, ...] | str:
547 """The reviewer bench for ``tier``: the policy's, filtered to what is available.
549 A policy that makes the panel the review keeps it — but only while the jury can
550 gate. Offering `jury` as the *default* beside an advisory jury would hand the
551 operator a pre-filled answer `keel validate` refuses.
552 """
553 if _configured_bench(state, tier) == team.JURY_PANEL and panel_allowed:
554 return team.JURY_PANEL
555 return _seat_bench(state, tier, implementer)
558def _configured_bench(state: State, tier: str) -> tuple[Seat, ...] | str | None:
559 """What the policy says about ``tier``: its own seats, ``review.default``'s, or nothing."""
560 configured = state.policy.review_by_tier.get(tier)
561 return state.policy.review if configured is None else configured
564def _seat_bench(state: State, tier: str, implementer: str) -> tuple[Seat, ...]:
565 """The policy's seats for ``tier``, filtered to this machine — **never** the panel.
567 The run scope's only bench source, and the config scope's whenever the jury cannot
568 gate: both need seats, and neither may fall back to a sentinel that one of them has
569 no spelling for and the other would fail validation on.
570 """
571 configured = _configured_bench(state, tier)
572 if isinstance(configured, tuple):
573 seats = tuple(seat for seat in configured if state.catalog.has(seat.provider))
574 if seats:
575 return seats
576 return _spread_bench(state, tier, implementer)
579def _spread_bench(state: State, tier: str, implementer: str) -> tuple[Seat, ...]:
580 """The tier-sized bench this machine can staff, preferring seats off the implementer."""
581 count = min(DEFAULT_BENCH[tier], len(state.catalog.candidates))
582 spread = state.catalog.spread()
583 preferred = [c for c in spread if c.name != implementer] + [
584 c for c in spread if c.name == implementer
585 ]
586 return tuple(Seat(provider=candidate.name) for candidate in preferred[:count])
589def _bench_answer(bench: tuple[Seat, ...] | str) -> str:
590 if isinstance(bench, str):
591 return bench
592 return ",".join(seat.provider for seat in bench)
595def _offerable_gate(state: State, name: str, implementer: str) -> bool:
596 """Is ``name`` a gate this catalogue offers, and not the implementer's own seat?"""
597 return state.catalog.has(name) and name != implementer
600def _seats_from(
601 value: str,
602 models: Mapping[str, str | None],
603 *,
604 catalog: Catalog,
605 fallback: tuple[Seat, ...],
606) -> tuple[Seat, ...]:
607 """Reviewer seats from a bench answer, dropping anything this machine cannot run.
609 The third guard of the same shape as the implementer's and the gate's: `normalize`
610 already refuses an off-offer token, but an answer seated straight onto `State`
611 reaches here unfiltered, and a bench naming a provider the probe never found is a
612 reviewer slot nobody can staff. An answer that filters down to nothing falls back
613 to the bench this question offered as its default.
615 It never returns the jury sentinel. Whether a panel is allowed at all is the
616 caller's decision, taken structurally rather than only through the offered choices:
617 a run has no flag that spells one, and a config may name one only beside a *gating*
618 jury — an advisory panel leaves that tier with no host reviewers and no required
619 evidence, which `team._review_issues` refuses. An answer of ``jury`` reaching here
620 is therefore just a name no provider has, and falls back like any other.
621 """
622 seats = tuple(
623 Seat(provider=name, model=models.get(name))
624 for name in value.split(",")
625 if catalog.has(name)
626 )
627 return seats or fallback
630def _policy_models(policy: team.TeamPolicy) -> dict[str, str | None]:
631 """Provider -> the model the policy seats it on, across every reviewer bench.
633 A reviewer question answers with provider *names* only, so a bench the operator
634 keeps unchanged would otherwise lose the model its policy pinned. First seat wins;
635 the walk order is sorted, so the answer does not depend on dict iteration order.
636 """
637 models: dict[str, str | None] = {}
638 for _, seat in _policy_seats(policy):
639 if seat.model and seat.provider not in models:
640 models[seat.provider] = seat.model
641 return models
644def _provider_choices(catalog: Catalog, *, skip: str | None = None) -> tuple[Choice, ...]:
645 return tuple(
646 Choice(candidate.name, candidate.detail())
647 for candidate in catalog.candidates
648 if candidate.name != skip
649 )
652class WizardError(Exception):
653 """The wizard cannot run at all: the probe offered nothing to choose between."""
656def _walk(state: State) -> tuple[tuple[Question, ...], Resolution]:
657 """The one traversal: it emits the questions *and* the resolution they resolve to.
659 Every value is read through the same ``ask`` closure, so a question's default and
660 the value used when it is unanswered are the same expression — the two can never
661 drift, which is what makes a quick-start run and an all-defaults customized run
662 identical by construction.
663 """
664 if not state.catalog.candidates:
665 raise WizardError(
666 "no provider is available on this machine — run `keel doctor --providers` "
667 "to see why; the wizard has nothing it could offer"
668 )
669 questions: list[Question] = []
670 answered: set[str] = set()
671 asking = True
672 config_scope = state.scope == SCOPE_CONFIG
674 def ask(
675 key: str,
676 prompt: str,
677 choices: tuple[Choice, ...],
678 default: str,
679 *,
680 help: str = "",
681 multi: bool = False,
682 max_values: int = 1,
683 ) -> str:
684 question = Question(
685 key=key,
686 prompt=prompt,
687 choices=choices,
688 default=default,
689 help=help,
690 multi=multi,
691 max_values=max_values,
692 )
693 if asking:
694 questions.append(question)
695 if key in state.answers:
696 answered.add(key)
697 return state.answers[key]
698 return default
700 mode = ask(
701 "mode",
702 "Start style",
703 (
704 Choice(QUICK_START, "take every default below and ask nothing else"),
705 Choice(CUSTOMIZE, "answer each question"),
706 ),
707 QUICK_START,
708 help="quick-start resolves every option to its default",
709 )
710 # Quick-start still *computes* every value below — it just stops asking. The
711 # resolution is therefore the same object either way, which is the property that
712 # lets `--wizard` promise it "cannot produce a config the grammar could not".
713 asking = mode == CUSTOMIZE
715 implement_default = _default_implement(state)
716 provider = ask(
717 "implement.provider",
718 "Implementer provider",
719 _provider_choices(state.catalog),
720 implement_default.provider,
721 help="who writes the change at s4",
722 )
723 candidate = state.catalog.get(provider)
724 if candidate is None:
725 # Only reachable when a caller seats an answer directly on `State` instead of
726 # going through `normalize`/`apply_answers`. The offer stands: an answer that
727 # is not on it resolves to the default rather than to an unusable provider.
728 provider = implement_default.provider
729 candidate = state.catalog.get(provider)
730 model = implement_default.model if provider == implement_default.provider else None
731 if candidate.models:
732 model_answer = ask(
733 "implement.model",
734 "Implementer model",
735 (
736 Choice(NONE, f"the provider's own default for {provider}"),
737 *(Choice(name) for name in candidate.models),
738 ),
739 model if model in candidate.models else NONE,
740 help=f"models {provider} lists for itself",
741 )
742 model = None if model_answer == NONE else model_answer
743 # A run carries no effort: `--delegate` splits `provider:model` and there is no
744 # `--effort` on `keel ship`, so an answer here would change nothing this command
745 # publishes. It stays a `knobs.team` question (see :data:`CONFIG_ONLY_KEYS`).
746 effort = (
747 implement_default.effort
748 if config_scope and provider == implement_default.provider
749 else None
750 )
751 # A vendor that spells effort as a model suffix has nowhere to put one without a
752 # model, and `keel validate` says so (`team.effort_needs_model`). Asking anyway
753 # would let the wizard write a config keel then refuses to load — the one thing a
754 # scaffolder must never do.
755 if (
756 config_scope
757 and candidate.effort
758 and not (team.effort_needs_model(candidate.vendor) and model is None)
759 ):
760 effort_answer = ask(
761 "implement.effort",
762 "Implementer reasoning effort",
763 (Choice(NONE, "no effort request"), *(Choice(name) for name in _efforts())),
764 effort if effort in _efforts() else NONE,
765 help=f"{provider} can express this in its own spelling",
766 )
767 effort = None if effort_answer == NONE else effort_answer
768 elif config_scope and team.effort_needs_model(candidate.vendor):
769 # A policy default that carried an effort loses it along with the model it
770 # was a suffix of; keeping it would write exactly the pair keel rejects.
771 effort = None
772 implement = Seat(provider=provider, model=model, effort=effort)
774 gate = None
775 if config_scope:
776 gate_default = _default_gate(state, provider)
777 gate_answer = ask(
778 "gate.provider",
779 "Gate reviewer",
780 (
781 Choice(NONE, "no mandatory second opinion"),
782 *_provider_choices(state.catalog, skip=provider),
783 ),
784 gate_default.provider if gate_default is not None else NONE,
785 help="one mandatory second opinion, never the seat that wrote the change",
786 )
787 if gate_answer != NONE and not _offerable_gate(state, gate_answer, provider):
788 # Same guard as the implementer's, for the same reason: an answer seated
789 # directly on `State` bypasses `normalize`, and a gate naming an unusable
790 # provider — or the implementer itself — is one `keel validate` refuses.
791 gate_answer = gate_default.provider if gate_default is not None else NONE
792 gate = (
793 None
794 if gate_answer == NONE
795 else Seat(provider=gate_answer, distinct_from=team.IMPLEMENTER)
796 )
798 jury = ask(
799 "jury",
800 "Cross-vendor jury",
801 (
802 Choice("gating", "a blocking verdict blocks the merge"),
803 Choice("advisory", "the panel reports and never gates"),
804 Choice(JURY_OFF, "no jury on this run"),
805 ),
806 state.jury if state.jury in JURY_ANSWERS else JURY_OFF,
807 help="the cross-vendor panel",
808 )
810 # `keel ship` spells the bench with `--reviewers <1|2|3>` and `--review-delegate`;
811 # neither can say "the panel *is* the review", so a run-scope `jury` answer emitted
812 # no flags at all and silently did nothing. It stays a `knobs.team` answer until a
813 # run flag can express it (#1015 / #1046).
814 panel_allowed = jury == "gating" and config_scope
815 bench_choices = _provider_choices(state.catalog)
816 if panel_allowed:
817 # Only a *gating* jury may be the review. "The panel is the review" plus "the
818 # panel does not gate" leaves that tier with nothing enforceable — no host
819 # reviewer slots to fall back on, and an advisory verdict is not required
820 # evidence — which `team._review_issues` refuses outright. Offering it for
821 # `advisory` let the wizard write the one combination `keel validate` rejects.
822 bench_choices = (
823 Choice(team.JURY_PANEL, "the cross-vendor panel *is* the review"),
824 *bench_choices,
825 )
826 max_seats = min(team.MAX_REVIEW_SEATS, len(state.catalog.candidates))
827 seat_models = _policy_models(state.policy)
828 review: tuple[Seat, ...] = ()
829 review_by_tier: dict[str, tuple[Seat, ...] | str] = {}
830 if not config_scope:
831 # Seats, never the sentinel: `_seat_bench` and `_seats_from` cannot produce one.
832 run_bench = _seat_bench(state, RUN_BENCH_TIER, provider)
833 review = _seats_from(
834 ask(
835 "review",
836 "Reviewers for this run",
837 bench_choices,
838 _bench_answer(run_bench),
839 help="one seat per slot (A, B, C); leave it to keep the tier's own bench",
840 multi=True,
841 max_values=max_seats,
842 ),
843 seat_models,
844 catalog=state.catalog,
845 fallback=run_bench,
846 )
847 else:
848 for tier in team.TIERS:
849 tier_bench = _default_bench(state, tier, provider, panel_allowed=panel_allowed)
850 answer = ask(
851 f"review.{tier}",
852 f"Reviewers for tier {tier}",
853 bench_choices,
854 _bench_answer(tier_bench),
855 help=f"knobs.team.review.by_tier.{quoted(tier)}",
856 multi=True,
857 max_values=max_seats,
858 )
859 review_by_tier[tier] = (
860 team.JURY_PANEL
861 if answer == team.JURY_PANEL and panel_allowed
862 else _seats_from(
863 answer,
864 seat_models,
865 catalog=state.catalog,
866 fallback=_seat_bench(state, tier, provider),
867 )
868 )
870 review_comments = ask(
871 "review_comments",
872 "Review comment posting",
873 (
874 Choice("inline", "one comment per finding, on the diff"),
875 Choice("summary", "one rolled-up review comment"),
876 ),
877 state.review_comments,
878 help="how findings reach the pull request",
879 )
880 resolution = Resolution(
881 scope=state.scope,
882 implement=implement,
883 gate=gate,
884 review=review,
885 review_by_tier=review_by_tier,
886 jury=jury,
887 review_comments=review_comments,
888 quick_start=mode == QUICK_START,
889 answered=frozenset(answered),
890 )
891 return tuple(questions), resolution
894def quoted(tier: str) -> str:
895 """A tier key as ``knobs.team`` spells it — quoted, because YAML reads bare ``1:`` as int."""
896 return f'"{tier}"'
899def apply_answers(state: State, answers: Mapping[str, str]) -> tuple[State, tuple[str, ...]]:
900 """Feed recorded answers in, without prompting. Returns the state and any errors.
902 This is the non-interactive path: ``--wizard-answer key=value`` on the command
903 line, or a replayed run. An answer naming a provider the probe did not offer is
904 reported and *not* applied — the same wall the interactive path puts up.
906 Supplying any answer other than ``mode`` implies ``mode=customize``. Without that
907 the first question's own default (quick-start) ends the walk before the second
908 question exists, so **every** other answer was rejected as "not a question this
909 wizard asks" — a flag that could only ever set `mode`. An explicit ``mode`` in the
910 answers still wins, including an explicit ``mode=quick-start``, which then really
911 does mean "ignore the rest".
912 """
913 errors: list[str] = []
914 remaining = dict(answers)
915 if remaining and "mode" not in remaining:
916 remaining["mode"] = CUSTOMIZE
917 while remaining:
918 question = state.next_question()
919 if question is None:
920 break
921 if question.key not in remaining:
922 # Asked, not answered: it keeps its default and produces no flag.
923 state = state.with_default(question.key)
924 continue
925 value, error = question.normalize(remaining.pop(question.key))
926 if error is not None:
927 errors.append(error)
928 state = state.with_default(question.key)
929 continue
930 state = state.with_answer(question.key, value)
931 errors.extend(_unused_answer_issues(state, remaining))
932 return state, tuple(errors)
935def _unused_answer_issues(state: State, remaining: Mapping[str, str]) -> list[str]:
936 """Why each leftover answer was never consumed — a typo, or an unreachable branch.
938 The two are different problems with different fixes, and one message for both sent
939 an operator hunting for a misspelling in a key that was spelled perfectly and simply
940 never asked (``review.3`` in a run, ``implement.model`` for a provider that lists
941 none). Say which.
942 """
943 issues = []
944 for key in sorted(remaining):
945 if key not in QUESTION_KEYS:
946 issues.append(
947 f"{key}: not a question this wizard asks; valid keys are {', '.join(QUESTION_KEYS)}"
948 )
949 else:
950 issues.append(
951 f"{key}: a real wizard question, but this run never reaches it — "
952 f"{_unreachable_reason(state, key)}"
953 )
954 return issues
957def _unreachable_reason(state: State, key: str) -> str:
958 if state.answers.get("mode") == QUICK_START:
959 return "you passed mode=quick-start, which answers nothing else"
960 if state.scope == SCOPE_RUN and key in CONFIG_ONLY_KEYS:
961 return (
962 f"{key} is a `keel init --wizard` question: it lands in knobs.team, and "
963 "`keel ship` has no flag that carries it, so a run could not honour an answer"
964 )
965 if state.scope == SCOPE_RUN and key.startswith("review."):
966 return (
967 f"{key} is a `keel init --wizard` question (one bench per risk tier); a run "
968 "asks `review` once, because its tier is not classified until s5"
969 )
970 if state.scope == SCOPE_CONFIG and key == "review":
971 return "a config names one bench per tier, so answer review.1 / review.2 / review.3"
972 if key == "implement.model":
973 return "the chosen implementer lists no models for keel to offer"
974 return (
975 "the chosen implementer has no spelling for reasoning effort, or needs a model chosen first"
976 )
979def run(
980 state: State,
981 ask: Callable[[str, str], str],
982 notify: Callable[[str], None],
983) -> State:
984 """Ask every remaining question through ``ask``. Pure given its seams.
986 ``ask(prompt, default)`` returns the operator's answer. **A blank return means "I
987 accept the default"** — recorded as a default, not as an answer, so it produces no
988 flag and the command resolves that option exactly as it would have without
989 ``--wizard``. An answer outside the question's choices is refused through
990 ``notify`` and asked again, at most :data:`MAX_ATTEMPTS` times — after that the
991 default stands, because a wizard that argues forever is the hang ``--wizard``
992 promises never to be.
993 """
994 question = state.next_question()
995 while question is not None:
996 chosen: str | None = None
997 for _ in range(MAX_ATTEMPTS):
998 raw = ask(question.text(), question.default)
999 if not (raw or "").strip():
1000 break
1001 candidate, error = question.normalize(raw)
1002 if error is None:
1003 chosen = candidate
1004 break
1005 notify(error)
1006 else:
1007 notify(f"{question.key}: keeping the default {question.default!r}")
1008 state = (
1009 state.with_default(question.key)
1010 if chosen is None
1011 else state.with_answer(question.key, chosen)
1012 )
1013 question = state.next_question()
1014 return state
1017def render(resolution: Resolution) -> str:
1018 """The operator-facing echo: the resolved flag set, then the seats behind it."""
1019 flags = resolution.flags()
1020 lines = [
1021 f" flags : {' '.join(flags)}"
1022 if flags
1023 else " flags : (none — every option kept its default, so nothing is overridden)"
1024 ]
1025 seats = [f"implement={seat_token(resolution.implement)}"]
1026 if resolution.implement.effort:
1027 seats.append(f"effort={resolution.implement.effort}")
1028 if resolution.gate is not None:
1029 seats.append(f"gate={seat_token(resolution.gate)} (distinct from the implementer)")
1030 if resolution.review:
1031 seats.append("review=" + ",".join(seat_token(s) for s in resolution.review))
1032 for tier, bench in sorted(resolution.review_by_tier.items()):
1033 rendered = bench if isinstance(bench, str) else ",".join(seat_token(s) for s in bench)
1034 seats.append(f"review[{tier}]={rendered}")
1035 lines.append(f" seats : {' · '.join(seats)}")
1036 return "\n".join(lines)
1039def parse_answer_args(values: Iterable[str]) -> tuple[dict[str, str], tuple[str, ...]]:
1040 """``KEY=VALUE`` strings -> an answer mapping plus the ones that were not pairs."""
1041 answers: dict[str, str] = {}
1042 errors: list[str] = []
1043 for raw in values:
1044 for item in str(raw).split(";"):
1045 text = item.strip()
1046 if not text:
1047 continue
1048 key, sep, value = text.partition("=")
1049 if not sep or not key.strip():
1050 errors.append(f"--wizard-answer {text!r} is not KEY=VALUE")
1051 continue
1052 answers[key.strip()] = value.strip()
1053 return answers, tuple(errors)
1056def unavailable(policy: team.TeamPolicy, catalog: Catalog) -> tuple[str, ...]:
1057 """Providers the policy names that this machine cannot reach, in a stable order.
1059 Advisory, never fatal: a shared ``knobs.team`` legitimately names seats other
1060 machines fill. Saying so once at the top of a wizard run is how an operator learns
1061 why a configured default is not the offered one.
1062 """
1063 names: list[str] = []
1064 for _, seat in _policy_seats(policy):
1065 if seat.kind != "provider" or catalog.has(seat.provider) or seat.provider in names:
1066 continue
1067 names.append(seat.provider)
1068 return tuple(names)
1071def _policy_seats(policy: team.TeamPolicy) -> Sequence[tuple[str, Seat]]:
1072 seats: list[tuple[str, Seat]] = []
1073 if policy.implement is not None:
1074 seats.append(("implement.default", policy.implement))
1075 seats.extend(sorted(policy.implement_by_role.items()))
1076 if policy.gate is not None:
1077 seats.append(("gate", policy.gate))
1078 benches: list[tuple[str, tuple[Seat, ...] | str]] = sorted(policy.review_by_tier.items())
1079 if policy.review is not None:
1080 benches.append(("review.default", policy.review))
1081 for path, bench in benches:
1082 if isinstance(bench, tuple):
1083 seats.extend((path, seat) for seat in bench)
1084 return seats