Coverage for src/keel/wizard.py: 100%

428 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-02 20:26 +0000

1"""The pure question/answer planner behind every keel `--wizard` (#1018). 

2 

3`ship.md` has promised an interactive picker "built from a best-effort tool/model 

4probe" since the adapter was written, but no core code existed for it, so every host 

5improvised the choices — different questions, different defaults, and options naming 

6providers that are not installed on the operator's machine. 

7 

8This module is that picker's pure half. It takes two inputs and never performs I/O: 

9 

10* the **provider probe report** — the exact document ``keel doctor --providers --json`` 

11 prints (:func:`keel.providerprobe.build_report`), which is the single source of truth 

12 for what is usable here; 

13* the current **team policy** (:class:`keel.team.TeamPolicy`, ``knobs.team``), which 

14 supplies every default so a wizard run with no answers reproduces today's behaviour. 

15 

16Out of it comes a :class:`Resolution`: the implementer/gate/reviewer seats, the jury 

17mode and the review-comments mode, rendered either as the literal ``keel ship`` flag 

18set (so the adapter passes flags on, exactly as ``ship.md``'s worked example does) or 

19as a ``knobs.team`` block ``keel init --wizard`` writes. 

20 

21Two properties are load-bearing: 

22 

23* **A provider the probe did not mark available is never offered, and can never be 

24 selected.** Every question is closed over :class:`Catalog`, and 

25 :meth:`Question.normalize` refuses a value outside its own choices — so an answer 

26 file, an injected seam and a typed answer are all held to the same list. 

27* **Deterministic.** No wall-clock, no randomness, no I/O, no default that depends on 

28 the order a dict happened to iterate in. Identical inputs give identical questions, 

29 identical defaults and an identical resolution, which is what lets a host replay a 

30 wizard run from its recorded answers. 

31 

32The interactive half is :func:`run`, which is pure given its ``ask``/``notify`` seams; 

33the CLI supplies the real ``input``-based ones and the ``isatty`` guard. 

34""" 

35 

36from __future__ import annotations 

37 

38from collections.abc import Callable, Iterable, Mapping, Sequence 

39from dataclasses import dataclass, field 

40from typing import Any 

41 

42from . import team 

43from .team import Seat 

44from .vocab import EFFORTS, supports_effort 

45 

46#: JSON-stable schema id of :meth:`Resolution.as_dict`. 

47SCHEMA_VERSION = "keel.wizard.v1" 

48 

49#: Which wizard is being run. ``run`` resolves per-run flags for ``keel ship`` / 

50#: ``keel work-block``; ``config`` resolves the ``knobs.team`` block ``keel init`` 

51#: writes. The questions differ because the *artefacts* differ: a run has one reviewer 

52#: bench (``--reviewers`` / ``--review-delegate`` are per-slot flags and the risk tier 

53#: is not known until s5 classifies the diff), while a config names one bench **per 

54#: tier**. Same planner, same catalogue, same defaults. 

55SCOPE_RUN = "run" 

56SCOPE_CONFIG = "config" 

57SCOPES = (SCOPE_RUN, SCOPE_CONFIG) 

58 

59#: First question: take every default, or answer the rest. Mirrors the Quick-start vs 

60#: Customize fast path ``ship.md`` documents. 

61QUICK_START = "quick-start" 

62CUSTOMIZE = "customize" 

63 

64#: :attr:`Candidate.source` of a machine-level ``~/.keel/providers.yaml`` entry. Reachable 

65#: per run, never nameable in a committed policy — see :func:`committable`. 

66REGISTRY_SOURCE = "registry" 

67 

68#: Answer meaning "leave this unset" — no model override, no reasoning effort, no gate 

69#: review. Spelled once so a caller cannot confuse it with a provider named ``none``: 

70#: a provider by that name would still be offered under its own name, and the sentinel 

71#: is only ever compared against, never dispatched. 

72NONE = "none" 

73 

74#: The jury is off. ``gating``/``advisory`` are :data:`keel.team.JURY_MODES`. 

75JURY_OFF = "off" 

76JURY_ANSWERS = (*team.JURY_MODES, JURY_OFF) 

77 

78#: How review findings are posted, matching ``keel ship --review-comments``. 

79REVIEW_COMMENT_MODES = ("inline", "summary") 

80 

81#: The tier a **run** wizard derives its offered bench at. `keel ship` classifies the 

82#: real tier at s1, after the wizard has run, so the offer has to name some tier; this 

83#: is the one an empty changeset already classifies as. It is only ever a *default* — 

84#: an unanswered bench question emits no `--reviewers`, so the run still gets the bench 

85#: its real tier earns (see :meth:`Resolution.flags`). 

86RUN_BENCH_TIER = "2" 

87 

88#: Reviewer seats a tier gets when nothing else says otherwise — one at tier 1, two at 

89#: tier 2, three at tier 3, clamped to the providers that are actually available. 

90DEFAULT_BENCH = {"1": 1, "2": 2, "3": 3} 

91 

92#: Questions only :data:`SCOPE_CONFIG` asks, because they land in ``knobs.team`` and 

93#: ``keel ship`` has **no flag that carries them**. The gate seat has no ``--gate``, and 

94#: reasoning effort has no ``--effort`` (``--delegate`` splits ``provider:model`` and 

95#: stops there). Asking them in a run produced an answer that changed nothing: the echo 

96#: named a gate while the published ``assignment.gate`` still held the policy's — two 

97#: documents disagreeing about the same seat, which is the defect #1014 round 2 fixed 

98#: for reviewers. A question the run cannot honour is not asked. 

99#: **#1049 re-opens ``implement.effort`` for runs**: it adds ``--effort`` (and ``--team``) 

100#: to ``keel ship``, at which point a run *can* carry an effort and the key moves back to 

101#: the run column — here, and in the table at ``docs/keel/cli.md#ship-wizard-questions`` 

102#: that ``tests/test_wizard.py`` holds against this tuple. 

103CONFIG_ONLY_KEYS = ("implement.effort", "gate.provider") 

104 

105#: Every key the planner can ask under any scope or branch. Used to tell a misspelled 

106#: ``--wizard-answer`` from a correctly spelled one this run never reaches. 

107#: ``tests/test_wizard.py`` walks the planner and fails if a question escapes this list. 

108QUESTION_KEYS = ( 

109 "mode", 

110 "implement.provider", 

111 "implement.model", 

112 "implement.effort", 

113 "gate.provider", 

114 "jury", 

115 "review", 

116 "review.1", 

117 "review.2", 

118 "review.3", 

119 "review_comments", 

120) 

121 

122#: Re-asks a single question gets before the planner stops arguing and takes the 

123#: default. A wizard that can loop forever on a stubborn ``ask`` seam is a hang, and a 

124#: hang is the one thing ``--wizard`` promises never to be. 

125MAX_ATTEMPTS = 3 

126 

127 

128def _effort_capable(vendor: str) -> bool: 

129 """Can ``vendor`` express a reasoning-effort request in its own spelling? 

130 

131 Read from the leaf :mod:`keel.vocab` rather than from :mod:`keel.delegate`: the 

132 answer is vocabulary, not dispatch, and importing the executor would drag the whole 

133 config graph into :mod:`keel.scaffold` — a module whose entire job is to run before 

134 a config exists. Until #1050 that cost a function-local import; now it does not. 

135 """ 

136 return supports_effort(vendor) 

137 

138 

139def _efforts() -> tuple[str, ...]: 

140 """keel's vendor-neutral effort vocabulary (see :func:`_effort_capable`).""" 

141 return tuple(EFFORTS) 

142 

143 

144@dataclass(frozen=True) 

145class Candidate: 

146 """One provider the probe reported as **available**, ready to be offered.""" 

147 

148 name: str 

149 vendor: str 

150 transport: str 

151 source: str 

152 #: Models the provider listed for itself (``agy models``, Ollama ``/api/tags``). 

153 models: tuple[str, ...] = () 

154 #: Can it drive git/PR steps itself? Only a ``cli`` transport can. 

155 tools: bool = False 

156 

157 @property 

158 def effort(self) -> bool: 

159 """True when a reasoning effort can be asked for on this seat.""" 

160 return _effort_capable(self.vendor) 

161 

162 def detail(self) -> str: 

163 """The one-line description shown beside this provider in a question.""" 

164 parts = [self.transport, self.source] 

165 parts.append("tools" if self.tools else "no tools") 

166 if self.models: 

167 parts.append(f"{len(self.models)} model(s)") 

168 return " · ".join(parts) 

169 

170 

171@dataclass(frozen=True) 

172class Catalog: 

173 """Every available provider, in the probe's order. Empty means "nothing usable".""" 

174 

175 candidates: tuple[Candidate, ...] = () 

176 

177 @classmethod 

178 def from_report(cls, report: Any) -> Catalog: 

179 """Build a catalogue from a ``keel doctor --providers`` document. 

180 

181 Fail-soft on purpose: the report can arrive from a file an operator wrote or 

182 from another keel's ``--json`` output, so a malformed row is skipped rather 

183 than raising. A row that is not marked ``available`` is never a candidate — 

184 that single rule is what makes an unavailable provider unofferable. 

185 """ 

186 rows = report.get("providers") if isinstance(report, Mapping) else None 

187 candidates = [] 

188 for row in rows if isinstance(rows, list) else (): 

189 candidate = _candidate(row) 

190 if candidate is not None: 

191 candidates.append(candidate) 

192 return cls(tuple(candidates)) 

193 

194 def names(self) -> tuple[str, ...]: 

195 return tuple(candidate.name for candidate in self.candidates) 

196 

197 def get(self, name: str | None) -> Candidate | None: 

198 for candidate in self.candidates: 

199 if candidate.name == name: 

200 return candidate 

201 return None 

202 

203 def has(self, name: str | None) -> bool: 

204 return self.get(name) is not None 

205 

206 def spread(self) -> tuple[Candidate, ...]: 

207 """Candidates re-ordered so distinct vendors come first. 

208 

209 A default bench of two seats should be two *vendors* where the machine has 

210 two — one vendor reviewing twice is one opinion twice, which is the same rule 

211 :data:`keel.providers.REVIEW_VENDOR_MINIMUM` states for the jury. 

212 """ 

213 seen: dict[str, Candidate] = {} 

214 rest: list[Candidate] = [] 

215 for candidate in self.candidates: 

216 if candidate.vendor in seen: 

217 rest.append(candidate) 

218 else: 

219 seen[candidate.vendor] = candidate 

220 return (*seen.values(), *rest) 

221 

222 

223def _candidate(row: Any) -> Candidate | None: 

224 """One probe row -> a :class:`Candidate`, or ``None`` when it is not offerable.""" 

225 if not isinstance(row, Mapping) or not row.get("available"): 

226 return None 

227 name = row.get("name") 

228 vendor = row.get("vendor") 

229 if not isinstance(name, str) or not name.strip(): 

230 return None 

231 capabilities = row.get("capabilities") 

232 capabilities = capabilities if isinstance(capabilities, Mapping) else {} 

233 models = row.get("models") 

234 return Candidate( 

235 name=name.strip(), 

236 vendor=vendor.strip() if isinstance(vendor, str) and vendor.strip() else name.strip(), 

237 transport=_word(row.get("transport"), "cli"), 

238 source=_word(row.get("source"), "builtin"), 

239 models=tuple(m.strip() for m in models if isinstance(m, str) and m.strip()) 

240 if isinstance(models, list) 

241 else (), 

242 tools=bool(capabilities.get("tools")), 

243 ) 

244 

245 

246def _word(value: Any, fallback: str) -> str: 

247 return value.strip() if isinstance(value, str) and value.strip() else fallback 

248 

249 

250@dataclass(frozen=True) 

251class Choice: 

252 """One offered option: its literal answer value and a one-line description.""" 

253 

254 value: str 

255 detail: str = "" 

256 

257 

258@dataclass(frozen=True) 

259class Question: 

260 """One question, closed over the values it will accept.""" 

261 

262 key: str 

263 prompt: str 

264 choices: tuple[Choice, ...] 

265 default: str 

266 help: str = "" 

267 #: Comma-separated answer (a reviewer bench); otherwise exactly one value. 

268 multi: bool = False 

269 max_values: int = 1 

270 

271 def values(self) -> tuple[str, ...]: 

272 return tuple(choice.value for choice in self.choices) 

273 

274 def ordered(self) -> tuple[Choice, ...]: 

275 """Choices with the default first — the shape ``ship.md`` asks every question for.""" 

276 default = [choice for choice in self.choices if choice.value == self.default] 

277 return (*default, *(choice for choice in self.choices if choice.value != self.default)) 

278 

279 def text(self) -> str: 

280 """The full prompt block: the question, its help, then the options.""" 

281 lines = [self.prompt if not self.help else f"{self.prompt} — {self.help}"] 

282 for choice in self.ordered(): 

283 marker = " (default)" if choice.value == self.default else "" 

284 detail = f" — {choice.detail}" if choice.detail else "" 

285 lines.append(f" {choice.value}{marker}{detail}") 

286 if self.multi: 

287 lines.append(f" (comma-separate up to {self.max_values})") 

288 return "\n".join(lines) 

289 

290 def normalize(self, raw: str | None) -> tuple[str | None, str | None]: 

291 """``(value, error)`` — exactly one is ``None``. A blank answer is the default.""" 

292 text = (raw or "").strip() 

293 if not text: 

294 return self.default, None 

295 tokens = [t.strip() for t in text.split(",")] if self.multi else [text] 

296 tokens = [token for token in tokens if token] 

297 if not tokens: 

298 return self.default, None 

299 allowed = self.values() 

300 unknown = [token for token in tokens if token not in allowed] 

301 if unknown: 

302 return None, ( 

303 f"{self.key}: {unknown[0]!r} is not on offer here; choose from {', '.join(allowed)}" 

304 ) 

305 if len(tokens) > self.max_values: 

306 return None, f"{self.key}: at most {self.max_values} value(s), got {len(tokens)}" 

307 if len(tokens) > 1 and team.JURY_PANEL in tokens: 

308 return None, ( 

309 f"{self.key}: {team.JURY_PANEL!r} is the whole panel and cannot be " 

310 "combined with named seats" 

311 ) 

312 return ",".join(tokens), None 

313 

314 

315@dataclass(frozen=True) 

316class State: 

317 """A wizard mid-flight: what is on offer, what the defaults are, what was answered.""" 

318 

319 catalog: Catalog 

320 policy: team.TeamPolicy = field(default_factory=team.TeamPolicy) 

321 scope: str = SCOPE_RUN 

322 #: Defaults carried in from the parsed flags, so the wizard starts where the 

323 #: command line already is. 

324 review_comments: str = "inline" 

325 jury: str = JURY_OFF 

326 delegate: str | None = None 

327 #: Questions the operator answered with a value of their own. 

328 answers: Mapping[str, str] = field(default_factory=dict) 

329 #: Questions the operator was asked and accepted the default for. Deliberately 

330 #: **not** the same thing as an answer: an accepted default means "do what this 

331 #: command would have done anyway", and :meth:`Resolution.flags` must not then 

332 #: materialise that default as an explicit flag. Writing back a default the 

333 #: operator never chose is how a quick-start run on a tier-3 change ended up 

334 #: passing `--reviewers 2 --no-jury` and quietly dropping a reviewer and the 

335 #: gating jury. 

336 defaulted: frozenset[str] = frozenset() 

337 

338 def _replace(self, **changes: Any) -> State: 

339 base = { 

340 "catalog": self.catalog, 

341 "policy": self.policy, 

342 "scope": self.scope, 

343 "review_comments": self.review_comments, 

344 "jury": self.jury, 

345 "delegate": self.delegate, 

346 "answers": self.answers, 

347 "defaulted": self.defaulted, 

348 } 

349 return State(**{**base, **changes}) 

350 

351 def with_answer(self, key: str, value: str) -> State: 

352 """Record a value the operator chose. This one *does* become a flag.""" 

353 return self._replace(answers={**self.answers, key: value}) 

354 

355 def with_default(self, key: str) -> State: 

356 """Record that the operator accepted this question's default: no flag.""" 

357 return self._replace(defaulted=self.defaulted | {key}) 

358 

359 def settled(self, key: str) -> bool: 

360 """Has this question been put to the operator and disposed of?""" 

361 return key in self.answers or key in self.defaulted 

362 

363 def questions(self) -> tuple[Question, ...]: 

364 """Every question this scope asks, given the answers so far.""" 

365 return _walk(self)[0] 

366 

367 def next_question(self) -> Question | None: 

368 """The first question still unanswered, or ``None`` when the wizard is done.""" 

369 for question in self.questions(): 

370 if not self.settled(question.key): 

371 return question 

372 return None 

373 

374 def resolve(self) -> Resolution: 

375 """The resolved seats/flags for the answers so far (unanswered = default).""" 

376 return _walk(self)[1] 

377 

378 

379def committable(catalog: Catalog) -> Catalog: 

380 """Only the providers a **committed** ``knobs.team`` may name. 

381 

382 ``keel validate`` resolves a policy against the built-in vendors and the project's own 

383 ``knobs.delegate_profiles`` — never the machine-level ``~/.keel/providers.yaml``, 

384 because a policy that validates only on its author's laptop is not a policy 

385 (``docs/keel/configuration.md#team``). A registry provider stays reachable *per run* 

386 through ``--delegate``, so the run wizard still offers it; the config wizard must not, 

387 or the file it writes would fail the very next ``keel validate``. 

388 """ 

389 return Catalog(tuple(c for c in catalog.candidates if c.source != REGISTRY_SOURCE)) 

390 

391 

392def start( 

393 catalog: Catalog, 

394 *, 

395 policy: team.TeamPolicy | None = None, 

396 scope: str = SCOPE_RUN, 

397 review_comments: str = "inline", 

398 jury: str = JURY_OFF, 

399 delegate: str | None = None, 

400) -> State: 

401 """A fresh :class:`State`. ``jury``/``review_comments``/``delegate`` are the parsed flags.""" 

402 if scope not in SCOPES: 

403 raise ValueError(f"unknown wizard scope {scope!r}; valid: {', '.join(SCOPES)}") 

404 return State( 

405 catalog=committable(catalog) if scope == SCOPE_CONFIG else catalog, 

406 policy=policy if policy is not None else team.TeamPolicy(), 

407 scope=scope, 

408 review_comments=review_comments if review_comments in REVIEW_COMMENT_MODES else "inline", 

409 jury=jury if jury in JURY_ANSWERS else JURY_OFF, 

410 delegate=delegate, 

411 ) 

412 

413 

414@dataclass(frozen=True) 

415class Resolution: 

416 """What the wizard decided: seats, panel, and the flags that express them.""" 

417 

418 scope: str 

419 implement: Seat 

420 gate: Seat | None = None 

421 #: ``run`` scope: the bench for this run. Always seats, never the jury sentinel — 

422 #: `keel ship` spells a bench with ``--reviewers <1|2|3>`` and ``--review-delegate``, 

423 #: and neither can say "the panel *is* the review", so a run cannot express one 

424 #: (#1015 / #1046 may change that). A tier whose *policy* is the panel keeps it in 

425 #: :attr:`review_by_tier`, which only the config scope writes. 

426 review: tuple[Seat, ...] = () 

427 #: ``config`` scope: tier -> bench (or the jury panel). 

428 review_by_tier: Mapping[str, tuple[Seat, ...] | str] = field(default_factory=dict) 

429 jury: str = JURY_OFF 

430 review_comments: str = "inline" 

431 quick_start: bool = True 

432 #: Question keys the operator answered with a value of their own. Everything else 

433 #: resolved to a default, and a default is *not* a decision: see :meth:`flags`. 

434 answered: frozenset[str] = frozenset() 

435 

436 def flags(self) -> tuple[str, ...]: 

437 """The literal ``keel ship`` / ``keel work-block`` flag set, in a stable order. 

438 

439 **Only an answered question produces a flag.** Every value below also has a 

440 resolved default, and materialising those defaults as flags is not neutral — 

441 it overrides the very policy they were read from. The reviewer bench a run 

442 wizard shows is derived at a nominal tier because the real one is not 

443 classified until s5, and the jury default is "whatever the flags and 

444 `knobs.team` already say"; writing either back turned a quick-start run on a 

445 tier-3 change into `--reviewers 2 --no-jury`, dropping a reviewer and the 

446 gating jury. An unanswered question therefore emits nothing at all and the 

447 command resolves it exactly as it would have without `--wizard`. 

448 """ 

449 flags: list[str] = [] 

450 if {"implement.provider", "implement.model"} & self.answered: 

451 flags += ["--delegate", seat_token(self.implement)] 

452 if "review" in self.answered and self.review: 

453 flags += ["--reviewers", str(len(self.review))] 

454 for seat in self.review: 

455 flags += ["--review-delegate", seat_token(seat)] 

456 if "review_comments" in self.answered: 

457 flags += ["--review-comments", self.review_comments] 

458 if "jury" in self.answered: 

459 flags += { 

460 "gating": ["--jury"], 

461 "advisory": ["--jury-advisory"], 

462 JURY_OFF: ["--no-jury"], 

463 }[self.jury] 

464 return tuple(flags) 

465 

466 def team_block(self) -> dict[str, Any]: 

467 """The ``knobs.team`` block, in the shape :mod:`keel.team` parses (#1014).""" 

468 block: dict[str, Any] = {"implement": {"default": _seat_block(self.implement)}} 

469 if self.gate is not None: 

470 block["gate"] = {**_seat_block(self.gate), "distinct_from": team.IMPLEMENTER} 

471 by_tier = { 

472 tier: value if isinstance(value, str) else [_seat_block(s) for s in value] 

473 for tier, value in sorted(self.review_by_tier.items()) 

474 } 

475 if by_tier: 

476 block["review"] = {"by_tier": by_tier} 

477 if self.jury != JURY_OFF: 

478 block["jury"] = {"mode": self.jury, "min_vendors": team.DEFAULT_MIN_VENDORS} 

479 block["fix"] = {"provider": team.IMPLEMENTER} 

480 return block 

481 

482 def as_dict(self) -> dict[str, Any]: 

483 """JSON-stable echo of everything the wizard resolved.""" 

484 return { 

485 "schema_version": SCHEMA_VERSION, 

486 "scope": self.scope, 

487 "quick_start": self.quick_start, 

488 "flags": list(self.flags()), 

489 "implement": self.implement.as_dict(), 

490 "gate": None if self.gate is None else self.gate.as_dict(), 

491 "review": [seat.as_dict() for seat in self.review], 

492 "review_by_tier": { 

493 tier: value if isinstance(value, str) else [s.as_dict() for s in value] 

494 for tier, value in sorted(self.review_by_tier.items()) 

495 }, 

496 "jury": self.jury, 

497 "review_comments": self.review_comments, 

498 "team": self.team_block(), 

499 } 

500 

501 

502def seat_token(seat: Seat) -> str: 

503 """``provider`` or ``provider:model`` — the spelling ``--delegate`` takes.""" 

504 return f"{seat.provider}:{seat.model}" if seat.model else seat.provider 

505 

506 

507def _seat_block(seat: Seat) -> dict[str, Any]: 

508 block: dict[str, Any] = {"provider": seat.provider} 

509 if seat.model: 

510 block["model"] = seat.model 

511 if seat.effort: 

512 block["effort"] = seat.effort 

513 return block 

514 

515 

516def _default_implement(state: State) -> Seat: 

517 """The implementer to start from: the flag, then the policy, then what is available. 

518 

519 A default that names a provider this machine cannot reach is *not* a default — it 

520 would put an unusable option in front of the operator as the recommended one. It 

521 degrades to the first available candidate instead. 

522 """ 

523 if state.delegate: 

524 seat = team.seat_from_token(state.delegate) 

525 if state.catalog.has(seat.provider): 

526 return seat 

527 if state.policy.implement is not None and state.catalog.has(state.policy.implement.provider): 

528 return state.policy.implement 

529 return Seat(provider=state.catalog.names()[0]) 

530 

531 

532def _default_gate(state: State, implementer: str) -> Seat | None: 

533 """The gate seat, dropped when it is unavailable or would be the implementer.""" 

534 gate = state.policy.gate 

535 if gate is None or not state.catalog.has(gate.provider) or gate.provider == implementer: 

536 return None 

537 return gate 

538 

539 

540def _default_bench( 

541 state: State, 

542 tier: str, 

543 implementer: str, 

544 *, 

545 panel_allowed: bool, 

546) -> tuple[Seat, ...] | str: 

547 """The reviewer bench for ``tier``: the policy's, filtered to what is available. 

548 

549 A policy that makes the panel the review keeps it — but only while the jury can 

550 gate. Offering `jury` as the *default* beside an advisory jury would hand the 

551 operator a pre-filled answer `keel validate` refuses. 

552 """ 

553 if _configured_bench(state, tier) == team.JURY_PANEL and panel_allowed: 

554 return team.JURY_PANEL 

555 return _seat_bench(state, tier, implementer) 

556 

557 

558def _configured_bench(state: State, tier: str) -> tuple[Seat, ...] | str | None: 

559 """What the policy says about ``tier``: its own seats, ``review.default``'s, or nothing.""" 

560 configured = state.policy.review_by_tier.get(tier) 

561 return state.policy.review if configured is None else configured 

562 

563 

564def _seat_bench(state: State, tier: str, implementer: str) -> tuple[Seat, ...]: 

565 """The policy's seats for ``tier``, filtered to this machine — **never** the panel. 

566 

567 The run scope's only bench source, and the config scope's whenever the jury cannot 

568 gate: both need seats, and neither may fall back to a sentinel that one of them has 

569 no spelling for and the other would fail validation on. 

570 """ 

571 configured = _configured_bench(state, tier) 

572 if isinstance(configured, tuple): 

573 seats = tuple(seat for seat in configured if state.catalog.has(seat.provider)) 

574 if seats: 

575 return seats 

576 return _spread_bench(state, tier, implementer) 

577 

578 

579def _spread_bench(state: State, tier: str, implementer: str) -> tuple[Seat, ...]: 

580 """The tier-sized bench this machine can staff, preferring seats off the implementer.""" 

581 count = min(DEFAULT_BENCH[tier], len(state.catalog.candidates)) 

582 spread = state.catalog.spread() 

583 preferred = [c for c in spread if c.name != implementer] + [ 

584 c for c in spread if c.name == implementer 

585 ] 

586 return tuple(Seat(provider=candidate.name) for candidate in preferred[:count]) 

587 

588 

589def _bench_answer(bench: tuple[Seat, ...] | str) -> str: 

590 if isinstance(bench, str): 

591 return bench 

592 return ",".join(seat.provider for seat in bench) 

593 

594 

595def _offerable_gate(state: State, name: str, implementer: str) -> bool: 

596 """Is ``name`` a gate this catalogue offers, and not the implementer's own seat?""" 

597 return state.catalog.has(name) and name != implementer 

598 

599 

600def _seats_from( 

601 value: str, 

602 models: Mapping[str, str | None], 

603 *, 

604 catalog: Catalog, 

605 fallback: tuple[Seat, ...], 

606) -> tuple[Seat, ...]: 

607 """Reviewer seats from a bench answer, dropping anything this machine cannot run. 

608 

609 The third guard of the same shape as the implementer's and the gate's: `normalize` 

610 already refuses an off-offer token, but an answer seated straight onto `State` 

611 reaches here unfiltered, and a bench naming a provider the probe never found is a 

612 reviewer slot nobody can staff. An answer that filters down to nothing falls back 

613 to the bench this question offered as its default. 

614 

615 It never returns the jury sentinel. Whether a panel is allowed at all is the 

616 caller's decision, taken structurally rather than only through the offered choices: 

617 a run has no flag that spells one, and a config may name one only beside a *gating* 

618 jury — an advisory panel leaves that tier with no host reviewers and no required 

619 evidence, which `team._review_issues` refuses. An answer of ``jury`` reaching here 

620 is therefore just a name no provider has, and falls back like any other. 

621 """ 

622 seats = tuple( 

623 Seat(provider=name, model=models.get(name)) 

624 for name in value.split(",") 

625 if catalog.has(name) 

626 ) 

627 return seats or fallback 

628 

629 

630def _policy_models(policy: team.TeamPolicy) -> dict[str, str | None]: 

631 """Provider -> the model the policy seats it on, across every reviewer bench. 

632 

633 A reviewer question answers with provider *names* only, so a bench the operator 

634 keeps unchanged would otherwise lose the model its policy pinned. First seat wins; 

635 the walk order is sorted, so the answer does not depend on dict iteration order. 

636 """ 

637 models: dict[str, str | None] = {} 

638 for _, seat in _policy_seats(policy): 

639 if seat.model and seat.provider not in models: 

640 models[seat.provider] = seat.model 

641 return models 

642 

643 

644def _provider_choices(catalog: Catalog, *, skip: str | None = None) -> tuple[Choice, ...]: 

645 return tuple( 

646 Choice(candidate.name, candidate.detail()) 

647 for candidate in catalog.candidates 

648 if candidate.name != skip 

649 ) 

650 

651 

652class WizardError(Exception): 

653 """The wizard cannot run at all: the probe offered nothing to choose between.""" 

654 

655 

656def _walk(state: State) -> tuple[tuple[Question, ...], Resolution]: 

657 """The one traversal: it emits the questions *and* the resolution they resolve to. 

658 

659 Every value is read through the same ``ask`` closure, so a question's default and 

660 the value used when it is unanswered are the same expression — the two can never 

661 drift, which is what makes a quick-start run and an all-defaults customized run 

662 identical by construction. 

663 """ 

664 if not state.catalog.candidates: 

665 raise WizardError( 

666 "no provider is available on this machine — run `keel doctor --providers` " 

667 "to see why; the wizard has nothing it could offer" 

668 ) 

669 questions: list[Question] = [] 

670 answered: set[str] = set() 

671 asking = True 

672 config_scope = state.scope == SCOPE_CONFIG 

673 

674 def ask( 

675 key: str, 

676 prompt: str, 

677 choices: tuple[Choice, ...], 

678 default: str, 

679 *, 

680 help: str = "", 

681 multi: bool = False, 

682 max_values: int = 1, 

683 ) -> str: 

684 question = Question( 

685 key=key, 

686 prompt=prompt, 

687 choices=choices, 

688 default=default, 

689 help=help, 

690 multi=multi, 

691 max_values=max_values, 

692 ) 

693 if asking: 

694 questions.append(question) 

695 if key in state.answers: 

696 answered.add(key) 

697 return state.answers[key] 

698 return default 

699 

700 mode = ask( 

701 "mode", 

702 "Start style", 

703 ( 

704 Choice(QUICK_START, "take every default below and ask nothing else"), 

705 Choice(CUSTOMIZE, "answer each question"), 

706 ), 

707 QUICK_START, 

708 help="quick-start resolves every option to its default", 

709 ) 

710 # Quick-start still *computes* every value below — it just stops asking. The 

711 # resolution is therefore the same object either way, which is the property that 

712 # lets `--wizard` promise it "cannot produce a config the grammar could not". 

713 asking = mode == CUSTOMIZE 

714 

715 implement_default = _default_implement(state) 

716 provider = ask( 

717 "implement.provider", 

718 "Implementer provider", 

719 _provider_choices(state.catalog), 

720 implement_default.provider, 

721 help="who writes the change at s4", 

722 ) 

723 candidate = state.catalog.get(provider) 

724 if candidate is None: 

725 # Only reachable when a caller seats an answer directly on `State` instead of 

726 # going through `normalize`/`apply_answers`. The offer stands: an answer that 

727 # is not on it resolves to the default rather than to an unusable provider. 

728 provider = implement_default.provider 

729 candidate = state.catalog.get(provider) 

730 model = implement_default.model if provider == implement_default.provider else None 

731 if candidate.models: 

732 model_answer = ask( 

733 "implement.model", 

734 "Implementer model", 

735 ( 

736 Choice(NONE, f"the provider's own default for {provider}"), 

737 *(Choice(name) for name in candidate.models), 

738 ), 

739 model if model in candidate.models else NONE, 

740 help=f"models {provider} lists for itself", 

741 ) 

742 model = None if model_answer == NONE else model_answer 

743 # A run carries no effort: `--delegate` splits `provider:model` and there is no 

744 # `--effort` on `keel ship`, so an answer here would change nothing this command 

745 # publishes. It stays a `knobs.team` question (see :data:`CONFIG_ONLY_KEYS`). 

746 effort = ( 

747 implement_default.effort 

748 if config_scope and provider == implement_default.provider 

749 else None 

750 ) 

751 # A vendor that spells effort as a model suffix has nowhere to put one without a 

752 # model, and `keel validate` says so (`team.effort_needs_model`). Asking anyway 

753 # would let the wizard write a config keel then refuses to load — the one thing a 

754 # scaffolder must never do. 

755 if ( 

756 config_scope 

757 and candidate.effort 

758 and not (team.effort_needs_model(candidate.vendor) and model is None) 

759 ): 

760 effort_answer = ask( 

761 "implement.effort", 

762 "Implementer reasoning effort", 

763 (Choice(NONE, "no effort request"), *(Choice(name) for name in _efforts())), 

764 effort if effort in _efforts() else NONE, 

765 help=f"{provider} can express this in its own spelling", 

766 ) 

767 effort = None if effort_answer == NONE else effort_answer 

768 elif config_scope and team.effort_needs_model(candidate.vendor): 

769 # A policy default that carried an effort loses it along with the model it 

770 # was a suffix of; keeping it would write exactly the pair keel rejects. 

771 effort = None 

772 implement = Seat(provider=provider, model=model, effort=effort) 

773 

774 gate = None 

775 if config_scope: 

776 gate_default = _default_gate(state, provider) 

777 gate_answer = ask( 

778 "gate.provider", 

779 "Gate reviewer", 

780 ( 

781 Choice(NONE, "no mandatory second opinion"), 

782 *_provider_choices(state.catalog, skip=provider), 

783 ), 

784 gate_default.provider if gate_default is not None else NONE, 

785 help="one mandatory second opinion, never the seat that wrote the change", 

786 ) 

787 if gate_answer != NONE and not _offerable_gate(state, gate_answer, provider): 

788 # Same guard as the implementer's, for the same reason: an answer seated 

789 # directly on `State` bypasses `normalize`, and a gate naming an unusable 

790 # provider — or the implementer itself — is one `keel validate` refuses. 

791 gate_answer = gate_default.provider if gate_default is not None else NONE 

792 gate = ( 

793 None 

794 if gate_answer == NONE 

795 else Seat(provider=gate_answer, distinct_from=team.IMPLEMENTER) 

796 ) 

797 

798 jury = ask( 

799 "jury", 

800 "Cross-vendor jury", 

801 ( 

802 Choice("gating", "a blocking verdict blocks the merge"), 

803 Choice("advisory", "the panel reports and never gates"), 

804 Choice(JURY_OFF, "no jury on this run"), 

805 ), 

806 state.jury if state.jury in JURY_ANSWERS else JURY_OFF, 

807 help="the cross-vendor panel", 

808 ) 

809 

810 # `keel ship` spells the bench with `--reviewers <1|2|3>` and `--review-delegate`; 

811 # neither can say "the panel *is* the review", so a run-scope `jury` answer emitted 

812 # no flags at all and silently did nothing. It stays a `knobs.team` answer until a 

813 # run flag can express it (#1015 / #1046). 

814 panel_allowed = jury == "gating" and config_scope 

815 bench_choices = _provider_choices(state.catalog) 

816 if panel_allowed: 

817 # Only a *gating* jury may be the review. "The panel is the review" plus "the 

818 # panel does not gate" leaves that tier with nothing enforceable — no host 

819 # reviewer slots to fall back on, and an advisory verdict is not required 

820 # evidence — which `team._review_issues` refuses outright. Offering it for 

821 # `advisory` let the wizard write the one combination `keel validate` rejects. 

822 bench_choices = ( 

823 Choice(team.JURY_PANEL, "the cross-vendor panel *is* the review"), 

824 *bench_choices, 

825 ) 

826 max_seats = min(team.MAX_REVIEW_SEATS, len(state.catalog.candidates)) 

827 seat_models = _policy_models(state.policy) 

828 review: tuple[Seat, ...] = () 

829 review_by_tier: dict[str, tuple[Seat, ...] | str] = {} 

830 if not config_scope: 

831 # Seats, never the sentinel: `_seat_bench` and `_seats_from` cannot produce one. 

832 run_bench = _seat_bench(state, RUN_BENCH_TIER, provider) 

833 review = _seats_from( 

834 ask( 

835 "review", 

836 "Reviewers for this run", 

837 bench_choices, 

838 _bench_answer(run_bench), 

839 help="one seat per slot (A, B, C); leave it to keep the tier's own bench", 

840 multi=True, 

841 max_values=max_seats, 

842 ), 

843 seat_models, 

844 catalog=state.catalog, 

845 fallback=run_bench, 

846 ) 

847 else: 

848 for tier in team.TIERS: 

849 tier_bench = _default_bench(state, tier, provider, panel_allowed=panel_allowed) 

850 answer = ask( 

851 f"review.{tier}", 

852 f"Reviewers for tier {tier}", 

853 bench_choices, 

854 _bench_answer(tier_bench), 

855 help=f"knobs.team.review.by_tier.{quoted(tier)}", 

856 multi=True, 

857 max_values=max_seats, 

858 ) 

859 review_by_tier[tier] = ( 

860 team.JURY_PANEL 

861 if answer == team.JURY_PANEL and panel_allowed 

862 else _seats_from( 

863 answer, 

864 seat_models, 

865 catalog=state.catalog, 

866 fallback=_seat_bench(state, tier, provider), 

867 ) 

868 ) 

869 

870 review_comments = ask( 

871 "review_comments", 

872 "Review comment posting", 

873 ( 

874 Choice("inline", "one comment per finding, on the diff"), 

875 Choice("summary", "one rolled-up review comment"), 

876 ), 

877 state.review_comments, 

878 help="how findings reach the pull request", 

879 ) 

880 resolution = Resolution( 

881 scope=state.scope, 

882 implement=implement, 

883 gate=gate, 

884 review=review, 

885 review_by_tier=review_by_tier, 

886 jury=jury, 

887 review_comments=review_comments, 

888 quick_start=mode == QUICK_START, 

889 answered=frozenset(answered), 

890 ) 

891 return tuple(questions), resolution 

892 

893 

894def quoted(tier: str) -> str: 

895 """A tier key as ``knobs.team`` spells it — quoted, because YAML reads bare ``1:`` as int.""" 

896 return f'"{tier}"' 

897 

898 

899def apply_answers(state: State, answers: Mapping[str, str]) -> tuple[State, tuple[str, ...]]: 

900 """Feed recorded answers in, without prompting. Returns the state and any errors. 

901 

902 This is the non-interactive path: ``--wizard-answer key=value`` on the command 

903 line, or a replayed run. An answer naming a provider the probe did not offer is 

904 reported and *not* applied — the same wall the interactive path puts up. 

905 

906 Supplying any answer other than ``mode`` implies ``mode=customize``. Without that 

907 the first question's own default (quick-start) ends the walk before the second 

908 question exists, so **every** other answer was rejected as "not a question this 

909 wizard asks" — a flag that could only ever set `mode`. An explicit ``mode`` in the 

910 answers still wins, including an explicit ``mode=quick-start``, which then really 

911 does mean "ignore the rest". 

912 """ 

913 errors: list[str] = [] 

914 remaining = dict(answers) 

915 if remaining and "mode" not in remaining: 

916 remaining["mode"] = CUSTOMIZE 

917 while remaining: 

918 question = state.next_question() 

919 if question is None: 

920 break 

921 if question.key not in remaining: 

922 # Asked, not answered: it keeps its default and produces no flag. 

923 state = state.with_default(question.key) 

924 continue 

925 value, error = question.normalize(remaining.pop(question.key)) 

926 if error is not None: 

927 errors.append(error) 

928 state = state.with_default(question.key) 

929 continue 

930 state = state.with_answer(question.key, value) 

931 errors.extend(_unused_answer_issues(state, remaining)) 

932 return state, tuple(errors) 

933 

934 

935def _unused_answer_issues(state: State, remaining: Mapping[str, str]) -> list[str]: 

936 """Why each leftover answer was never consumed — a typo, or an unreachable branch. 

937 

938 The two are different problems with different fixes, and one message for both sent 

939 an operator hunting for a misspelling in a key that was spelled perfectly and simply 

940 never asked (``review.3`` in a run, ``implement.model`` for a provider that lists 

941 none). Say which. 

942 """ 

943 issues = [] 

944 for key in sorted(remaining): 

945 if key not in QUESTION_KEYS: 

946 issues.append( 

947 f"{key}: not a question this wizard asks; valid keys are {', '.join(QUESTION_KEYS)}" 

948 ) 

949 else: 

950 issues.append( 

951 f"{key}: a real wizard question, but this run never reaches it — " 

952 f"{_unreachable_reason(state, key)}" 

953 ) 

954 return issues 

955 

956 

957def _unreachable_reason(state: State, key: str) -> str: 

958 if state.answers.get("mode") == QUICK_START: 

959 return "you passed mode=quick-start, which answers nothing else" 

960 if state.scope == SCOPE_RUN and key in CONFIG_ONLY_KEYS: 

961 return ( 

962 f"{key} is a `keel init --wizard` question: it lands in knobs.team, and " 

963 "`keel ship` has no flag that carries it, so a run could not honour an answer" 

964 ) 

965 if state.scope == SCOPE_RUN and key.startswith("review."): 

966 return ( 

967 f"{key} is a `keel init --wizard` question (one bench per risk tier); a run " 

968 "asks `review` once, because its tier is not classified until s5" 

969 ) 

970 if state.scope == SCOPE_CONFIG and key == "review": 

971 return "a config names one bench per tier, so answer review.1 / review.2 / review.3" 

972 if key == "implement.model": 

973 return "the chosen implementer lists no models for keel to offer" 

974 return ( 

975 "the chosen implementer has no spelling for reasoning effort, or needs a model chosen first" 

976 ) 

977 

978 

979def run( 

980 state: State, 

981 ask: Callable[[str, str], str], 

982 notify: Callable[[str], None], 

983) -> State: 

984 """Ask every remaining question through ``ask``. Pure given its seams. 

985 

986 ``ask(prompt, default)`` returns the operator's answer. **A blank return means "I 

987 accept the default"** — recorded as a default, not as an answer, so it produces no 

988 flag and the command resolves that option exactly as it would have without 

989 ``--wizard``. An answer outside the question's choices is refused through 

990 ``notify`` and asked again, at most :data:`MAX_ATTEMPTS` times — after that the 

991 default stands, because a wizard that argues forever is the hang ``--wizard`` 

992 promises never to be. 

993 """ 

994 question = state.next_question() 

995 while question is not None: 

996 chosen: str | None = None 

997 for _ in range(MAX_ATTEMPTS): 

998 raw = ask(question.text(), question.default) 

999 if not (raw or "").strip(): 

1000 break 

1001 candidate, error = question.normalize(raw) 

1002 if error is None: 

1003 chosen = candidate 

1004 break 

1005 notify(error) 

1006 else: 

1007 notify(f"{question.key}: keeping the default {question.default!r}") 

1008 state = ( 

1009 state.with_default(question.key) 

1010 if chosen is None 

1011 else state.with_answer(question.key, chosen) 

1012 ) 

1013 question = state.next_question() 

1014 return state 

1015 

1016 

1017def render(resolution: Resolution) -> str: 

1018 """The operator-facing echo: the resolved flag set, then the seats behind it.""" 

1019 flags = resolution.flags() 

1020 lines = [ 

1021 f" flags : {' '.join(flags)}" 

1022 if flags 

1023 else " flags : (none — every option kept its default, so nothing is overridden)" 

1024 ] 

1025 seats = [f"implement={seat_token(resolution.implement)}"] 

1026 if resolution.implement.effort: 

1027 seats.append(f"effort={resolution.implement.effort}") 

1028 if resolution.gate is not None: 

1029 seats.append(f"gate={seat_token(resolution.gate)} (distinct from the implementer)") 

1030 if resolution.review: 

1031 seats.append("review=" + ",".join(seat_token(s) for s in resolution.review)) 

1032 for tier, bench in sorted(resolution.review_by_tier.items()): 

1033 rendered = bench if isinstance(bench, str) else ",".join(seat_token(s) for s in bench) 

1034 seats.append(f"review[{tier}]={rendered}") 

1035 lines.append(f" seats : {' · '.join(seats)}") 

1036 return "\n".join(lines) 

1037 

1038 

1039def parse_answer_args(values: Iterable[str]) -> tuple[dict[str, str], tuple[str, ...]]: 

1040 """``KEY=VALUE`` strings -> an answer mapping plus the ones that were not pairs.""" 

1041 answers: dict[str, str] = {} 

1042 errors: list[str] = [] 

1043 for raw in values: 

1044 for item in str(raw).split(";"): 

1045 text = item.strip() 

1046 if not text: 

1047 continue 

1048 key, sep, value = text.partition("=") 

1049 if not sep or not key.strip(): 

1050 errors.append(f"--wizard-answer {text!r} is not KEY=VALUE") 

1051 continue 

1052 answers[key.strip()] = value.strip() 

1053 return answers, tuple(errors) 

1054 

1055 

1056def unavailable(policy: team.TeamPolicy, catalog: Catalog) -> tuple[str, ...]: 

1057 """Providers the policy names that this machine cannot reach, in a stable order. 

1058 

1059 Advisory, never fatal: a shared ``knobs.team`` legitimately names seats other 

1060 machines fill. Saying so once at the top of a wizard run is how an operator learns 

1061 why a configured default is not the offered one. 

1062 """ 

1063 names: list[str] = [] 

1064 for _, seat in _policy_seats(policy): 

1065 if seat.kind != "provider" or catalog.has(seat.provider) or seat.provider in names: 

1066 continue 

1067 names.append(seat.provider) 

1068 return tuple(names) 

1069 

1070 

1071def _policy_seats(policy: team.TeamPolicy) -> Sequence[tuple[str, Seat]]: 

1072 seats: list[tuple[str, Seat]] = [] 

1073 if policy.implement is not None: 

1074 seats.append(("implement.default", policy.implement)) 

1075 seats.extend(sorted(policy.implement_by_role.items())) 

1076 if policy.gate is not None: 

1077 seats.append(("gate", policy.gate)) 

1078 benches: list[tuple[str, tuple[Seat, ...] | str]] = sorted(policy.review_by_tier.items()) 

1079 if policy.review is not None: 

1080 benches.append(("review.default", policy.review)) 

1081 for path, bench in benches: 

1082 if isinstance(bench, tuple): 

1083 seats.extend((path, seat) for seat in bench) 

1084 return seats