Coverage for src/keel/team.py: 100%
498 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""``knobs.team`` — the per-role / per-tier provider policy (#1014).
3Before this module keel could name *an* implementer (``--delegate``,
4``knobs.implementer_agents``) and *one* reviewer vendor for all N reviewers
5(``--review-delegate``). It could not express the team a real engineering org runs:
6*this role implements with provider X at effort E, every implementation gets one gate
7review from a different vendor, tier-2 gets two reviewers from two vendors, tier-3
8convenes the jury as the review panel.*
10``knobs.team`` is that policy, and this module is its pure half:
12* :class:`Seat` — one occupied chair: a provider, optionally a model and an effort.
13* :class:`TeamPolicy` — the parsed ``knobs.team`` block.
14* :func:`parse_team` / :func:`canonical` — YAML in, typed policy out, and back to the
15 canonical dict that feeds ``config_hash`` and the published contract.
16* :func:`team_issues` — the semantic validation ``keel validate`` runs.
17* :func:`resolve_assignment` — the deterministic answer to *who runs this ship*, which
18 ``keel plan``/``keel ship`` render as ``assignment`` so any host runs the same team.
20A ``provider`` names an entry the provider registry resolves — a built-in vendor, a
21``knobs.delegate_profiles`` entry, or a machine-level ``~/.keel/providers.yaml`` entry —
22with two reserved spellings:
24``subagent:<name>`` the pre-#1014 Claude-subagent semantics ``implementer_agents``
25 values carried, kept explicit so a seat can no longer be read as
26 both a vendor and a subagent name;
27``implementer`` valid in ``fix.provider`` and ``gate.distinct_from`` only: *the
28 provider that implemented this change*, whatever it resolved to.
30Pure and deterministic: no wall-clock, no randomness, no I/O, and exactly one keel
31import — the leaf :mod:`keel.vocab`, which owns the provider and effort vocabulary
32:func:`team_issues` validates against. That vocabulary used to live next to dispatch in
33:mod:`keel.agents` / :mod:`keel.delegate`, which both import :mod:`keel.config`, which
34imports this module for the policy type; validation therefore reached it through a
35function-local import. Moving the vocabulary instead of the import is what removed the
36cycle rather than hiding it (#1050).
37"""
39from __future__ import annotations
41from collections.abc import Iterable, Mapping, Sequence
42from dataclasses import dataclass, field
43from typing import Any
45from .vocab import BUILTIN_DELEGATE_VENDORS, EFFORTS, supports_effort
47#: Prefix that keeps the pre-#1014 Claude-subagent semantics of ``implementer_agents``.
48#: ``subagent:backend-developer`` is a subagent of the host agent, not a vendor keel
49#: dispatches to — which is exactly the ambiguity #1014 opened against.
50SUBAGENT_PREFIX = "subagent:"
52#: Reserved provider value: *whoever implemented this change*. Valid in ``fix.provider``
53#: and ``gate.distinct_from``; anywhere else it would name a provider that does not exist.
54IMPLEMENTER = "implementer"
56#: ``review.by_tier.<n>: jury`` — the cross-vendor panel **is** the review for that tier.
57JURY_PANEL = "jury"
59#: Risk tiers ``review.by_tier`` may key on, as **strings**. A JSON-Schema property name
60#: is a string, and YAML reads a bare ``1:`` as an integer key, so the keys are quoted
61#: (``"1":``) and :func:`team_issues` says so when they are not.
62TIERS = ("1", "2", "3")
64#: How an enabled jury gates. ``gating`` blocks the merge on a blocking verdict;
65#: ``advisory`` reports and never gates.
66JURY_MODES = ("gating", "advisory")
68#: What happens on a jury-panel tier when the panel cannot be staffed *here* (#1066).
69#: ``fallback`` seats a host bench of the same size in its place; ``block`` refuses the
70#: run. A **configured allowance, not a flag**: #1014 round 3 settled that an operator's
71#: preference may not take the panel off, and this does not reopen it — availability is
72#: measured by :func:`keel.juryavail.assess` from the probe keel already runs, never
73#: asserted on a command line.
74JURY_ON_UNAVAILABLE = ("fallback", "block")
76#: The allowance a project that never names one gets. ``fallback`` is the sensible answer
77#: for the single-maintainer case the panel would otherwise wall off; a project whose
78#: product claim *is* cross-vendor review says ``block`` and keeps today's strictness.
79JURY_ON_UNAVAILABLE_DEFAULT = "fallback"
81#: ``reviewer_source`` on a bench seated because the panel could not be. Named rather than
82#: derived from the tier's config path so a reader of the published contract can tell a
83#: fallback bench from a tier that simply never had a panel.
84JURY_FALLBACK_SOURCE = "jury-fallback"
86#: The binary s7 dispatches to convene the panel. Here rather than in
87#: :mod:`keel.juryavail` — which re-exports it — only because :func:`refusal_message` names
88#: it and that module imports this one, never the reverse. It is the same string as
89#: :data:`JURY_PANEL` and deliberately a separate name: one is a *review policy value*, the
90#: other is a *command*, and a project could not rename either without the other.
91JURY_RUNNER_COMMAND = "jury"
94class JuryUnavailableError(RuntimeError):
95 """``on_unavailable: block`` and the panel cannot be staffed — the run refuses.
97 Raised from :func:`_review_seats`, which is reached from exactly one function
98 (:func:`resolve_assignment`), and caught centrally in :func:`keel.cli.main`. Every
99 review-aware surface therefore refuses identically rather than each carrying its own
100 near-copy of the check.
102 **Raised where the panel is resolved, not where it is measured** (#1068). The probe
103 (:func:`keel.providerprobe.jury_availability`) answers one question about the
104 *machine*, and a surface may resolve several benches from that one answer: a swarm
105 scores each cluster's tier while it partitions and asks the probe the widest question
106 it can — *could any tier or band make the panel the review* — before any cluster
107 exists. Refusing there refused a plan of entirely non-panel clusters on an unstaffable
108 host, because "some tier could name the panel" is not "this work does". Here the
109 question is already narrowed to the cluster or command being staffed, so the run
110 refuses exactly when the panel really is its review.
112 Re-exported as ``keel.juryavail.JuryUnavailableError``: this module may not import
113 that one (:mod:`keel.juryavail` imports the policy vocabulary from here), and the
114 refusal has to live on the resolver's side of that edge.
115 """
118#: ``by_difficulty`` bands, lightest first. A band names the bench that staffs work of
119#: that weight; :func:`keel.swarm.score_difficulty` decides which band a cluster is, and
120#: the same resolver seats it. Unlike a tier — which is *how risky the change is* and is
121#: read off the files it touches — a band is *how much work it is*, which is what decides
122#: whether the strong implementer is worth spending on it (#1017).
123DIFFICULTY_BANDS = ("easy", "standard", "hard")
125#: Distinct vendors a jury needs before its verdict can gate, when the policy is silent.
126DEFAULT_MIN_VENDORS = 2
128#: Default agent when neither the policy nor a flag names one: the host agent driving
129#: the run. Mirrors :data:`keel.agents.HOST_DEFAULT`, which cannot be imported here.
130HOST_DEFAULT = "claude"
132#: Vendors that spell reasoning effort as a **model suffix** rather than as its own
133#: argument (``gemini-3.8-flash-high``). An effort on such a seat with no ``model``
134#: beside it has nothing to attach to, so :func:`team_issues` rejects the pair — and
135#: :mod:`keel.wizard` reads the same tuple rather than re-deriving the rule, so the
136#: wizard cannot offer an effort it would then be told to take back.
137EFFORT_MODEL_SUFFIX_VENDORS = ("agy",)
140def effort_needs_model(vendor: str | None) -> bool:
141 """True when an ``effort`` on this vendor's seat requires a ``model`` beside it."""
142 return vendor in EFFORT_MODEL_SUFFIX_VENDORS
145@dataclass(frozen=True)
146class Seat:
147 """One chair on the team: who sits in it, on which model, at which effort."""
149 provider: str
150 model: str | None = None
151 effort: str | None = None
152 #: ``gate`` only: the seat this one may not duplicate (``implementer``).
153 distinct_from: str | None = None
155 @property
156 def kind(self) -> str:
157 """``subagent`` | ``alias`` | ``provider`` — how the provider value is read."""
158 if self.provider.startswith(SUBAGENT_PREFIX):
159 return "subagent"
160 if self.provider == IMPLEMENTER:
161 return "alias"
162 return "provider"
164 @property
165 def name(self) -> str:
166 """The bare name: a subagent seat without its prefix, anything else verbatim."""
167 if self.kind == "subagent":
168 return self.provider[len(SUBAGENT_PREFIX) :]
169 return self.provider
171 def as_dict(self, *, source: str | None = None, slot: str | None = None) -> dict[str, Any]:
172 """JSON-stable seat record; ``source``/``slot`` are the assignment's annotations."""
173 record: dict[str, Any] = {
174 "provider": self.provider,
175 "name": self.name,
176 "kind": self.kind,
177 "model": self.model,
178 "effort": self.effort,
179 }
180 if self.distinct_from is not None:
181 record["distinct_from"] = self.distinct_from
182 if source is not None:
183 record["source"] = source
184 if slot is not None:
185 record["slot"] = slot
186 return record
189@dataclass(frozen=True)
190class Bench:
191 """A named bench: who leads it, who implements on it, who reviews, at what effort.
193 One type, two tables. ``team.by_difficulty`` picks a bench from a cluster's *scored
194 difficulty*; ``team.profiles`` lets an operator pick one by name (``--team``). They
195 are the same thing — "staff this piece of work from this bench" — so they resolve
196 through one code path rather than two that can disagree.
197 """
199 lead: Seat | None = None
200 implement: Seat | None = None
201 #: Reviewer seats, or the literal ``"jury"``; ``None`` leaves the tier's policy alone.
202 review: tuple[Seat, ...] | str | None = None
203 #: Effort for this bench's implementer, when the seat does not name one itself.
204 effort: str | None = None
207@dataclass(frozen=True)
208class TeamPolicy:
209 """A parsed ``knobs.team`` block. ``configured`` is False when there is none."""
211 configured: bool = False
212 implement: Seat | None = None
213 implement_by_role: Mapping[str, Seat] = field(default_factory=dict)
214 gate: Seat | None = None
215 #: ``review.default`` — seats, or the literal ``"jury"``.
216 review: tuple[Seat, ...] | str | None = None
217 #: Tier (``"1"``/``"2"``/``"3"``) -> reviewer seats, or the literal ``"jury"``.
218 review_by_tier: Mapping[str, tuple[Seat, ...] | str] = field(default_factory=dict)
219 jury_mode: str | None = None
220 jury_min_vendors: int | None = None
221 #: ``team.jury.on_unavailable`` — what a jury-panel tier does when the panel cannot be
222 #: staffed here. ``None`` is unset and resolves to
223 #: :data:`JURY_ON_UNAVAILABLE_DEFAULT`; kept tri-state at the config boundary so an
224 #: explicit ``fallback`` stays distinguishable from silence for ``config_hash``.
225 jury_on_unavailable: str | None = None
226 fix: Seat | None = None
227 #: The seat that coordinates a batch of ships — the team lead a swarm cluster or a
228 #: work block reports through. Defaults to the host agent driving the run.
229 lead: Seat | None = None
230 #: Difficulty band (:data:`DIFFICULTY_BANDS`) -> the bench that staffs work of that
231 #: weight.
232 by_difficulty: Mapping[str, Bench] = field(default_factory=dict)
233 #: Operator-selectable benches (``--team <profile>``).
234 profiles: Mapping[str, Bench] = field(default_factory=dict)
236 def benches_for(
237 self, *, difficulty: str | None = None, profile: str | None = None
238 ) -> tuple[tuple[Bench, str], ...]:
239 """Benches that apply to this run, most specific first, each with its config path.
241 A named ``--team`` profile is an operator saying *this bench, for this batch*, so
242 it outranks the band the scorer derived. They are not exclusive: each field
243 resolves down the list on its own, so a profile that names only reviewers still
244 lets the band's implementer stand instead of silently dropping it.
245 """
246 found: list[tuple[Bench, str]] = []
247 if profile is not None and profile in self.profiles:
248 found.append((self.profiles[profile], f"team.profiles.{profile}"))
249 if difficulty is not None and difficulty in self.by_difficulty:
250 found.append((self.by_difficulty[difficulty], f"team.by_difficulty.{difficulty}"))
251 return tuple(found)
253 def review_for(self, tier: int | None) -> tuple[tuple[Seat, ...] | str | None, str | None]:
254 """Reviewer seats (or ``"jury"``) for ``tier``, plus the config path they came from."""
255 key = str(tier)
256 if key in self.review_by_tier:
257 return self.review_by_tier[key], f"team.review.by_tier.{key}"
258 if self.review is not None:
259 return self.review, "team.review.default"
260 return None, None
263def _text(value: Any) -> str | None:
264 """A non-blank string, or ``None`` — so a blank field reads as unset."""
265 return value.strip() if isinstance(value, str) and value.strip() else None
268def _seat(raw: Any) -> Seat | None:
269 """One seat mapping -> a :class:`Seat`; ``None`` when it names no provider.
271 Shape errors are the schema's job and meaning is :func:`team_issues`'; this only has
272 to be total, because :func:`keel.config.parse_config` builds the policy *after*
273 validation and a caller must never get a half-parsed seat.
274 """
275 if not isinstance(raw, Mapping):
276 return None
277 provider = _text(raw.get("provider"))
278 if provider is None:
279 return None
280 return Seat(
281 provider=provider,
282 model=_text(raw.get("model")),
283 effort=_text(raw.get("effort")),
284 distinct_from=_text(raw.get("distinct_from")),
285 )
288def _seats(raw: Any) -> tuple[Seat, ...] | str | None:
289 """A reviewer list, the ``"jury"`` literal, or ``None`` when neither."""
290 if isinstance(raw, str):
291 return raw.strip()
292 if not isinstance(raw, Sequence) or isinstance(raw, (bytes, bytearray)):
293 return None
294 seats = tuple(seat for seat in (_seat(entry) for entry in raw) if seat is not None)
295 return seats
298def _bench(raw: Any) -> Bench | None:
299 """One ``by_difficulty``/``profiles`` entry -> a :class:`Bench`, or ``None``.
301 An entry that names nothing at all is ``None`` rather than an empty bench, so an
302 accidental ``hard: {}`` does not read as *a bench that overrides everything with
303 nothing* — it reads as absent, and the role/tier policy stands.
304 """
305 if not isinstance(raw, Mapping):
306 return None
307 bench = Bench(
308 lead=_seat(raw.get("lead")),
309 implement=_seat(raw.get("implement")),
310 review=_seats(raw.get("review")) if "review" in raw else None,
311 effort=_text(raw.get("effort")),
312 )
313 if bench == Bench():
314 return None
315 return bench
318def _benches(raw: Any) -> dict[str, Bench]:
319 """A ``by_difficulty``/``profiles`` mapping -> named benches, in a stable order."""
320 if not isinstance(raw, Mapping):
321 return {}
322 benches: dict[str, Bench] = {}
323 for name in sorted(raw, key=str):
324 bench = _bench(raw[name])
325 if isinstance(name, str) and bench is not None:
326 benches[name] = bench
327 return benches
330def parse_team(raw: Any) -> TeamPolicy:
331 """Parse a ``knobs.team`` block (``None``/malformed -> an unconfigured policy)."""
332 if not isinstance(raw, Mapping):
333 return TeamPolicy()
334 implement = raw.get("implement") if isinstance(raw.get("implement"), Mapping) else {}
335 by_role_raw = implement.get("by_role") if isinstance(implement.get("by_role"), Mapping) else {}
336 by_role = {}
337 for role in sorted(by_role_raw, key=str):
338 seat = _seat(by_role_raw[role])
339 if isinstance(role, str) and seat is not None:
340 by_role[role] = seat
341 review = raw.get("review") if isinstance(raw.get("review"), Mapping) else {}
342 by_tier: dict[str, tuple[Seat, ...] | str] = {}
343 by_tier_raw = review.get("by_tier") if isinstance(review.get("by_tier"), Mapping) else {}
344 for tier in sorted(by_tier_raw, key=str):
345 seats = _seats(by_tier_raw[tier])
346 if isinstance(tier, str) and seats is not None:
347 by_tier[tier] = seats
348 default_review = _seats(review.get("default")) if "default" in review else None
349 jury = raw.get("jury") if isinstance(raw.get("jury"), Mapping) else {}
350 min_vendors = jury.get("min_vendors")
351 return TeamPolicy(
352 configured=True,
353 implement=_seat(implement.get("default")),
354 implement_by_role=by_role,
355 gate=_seat(raw.get("gate")),
356 review=default_review,
357 review_by_tier=by_tier,
358 jury_mode=_text(jury.get("mode")),
359 jury_min_vendors=min_vendors if isinstance(min_vendors, int) else None,
360 jury_on_unavailable=_text(jury.get("on_unavailable")),
361 fix=_seat(raw.get("fix")),
362 lead=_seat(raw.get("lead")),
363 by_difficulty=_benches(raw.get("by_difficulty")),
364 profiles=_benches(raw.get("profiles")),
365 )
368def _seat_canonical(seat: Seat) -> dict[str, Any]:
369 record = {"provider": seat.provider, "model": seat.model, "effort": seat.effort}
370 if seat.distinct_from is not None:
371 record["distinct_from"] = seat.distinct_from
372 return record
375def _bench_canonical(bench: Bench) -> dict[str, Any]:
376 record: dict[str, Any] = {}
377 if bench.lead is not None:
378 record["lead"] = _seat_canonical(bench.lead)
379 if bench.implement is not None:
380 record["implement"] = _seat_canonical(bench.implement)
381 if bench.review is not None:
382 record["review"] = (
383 bench.review
384 if isinstance(bench.review, str)
385 else [_seat_canonical(seat) for seat in bench.review]
386 )
387 if bench.effort is not None:
388 record["effort"] = bench.effort
389 return record
392def canonical(policy: TeamPolicy) -> dict[str, Any]:
393 """``{"team": {...}}`` for a configured policy, ``{}`` otherwise.
395 Empty means **absent**, not ``{}``: an added optional field must not rotate
396 ``config_hash`` for the projects that never used it, which is the same treatment
397 :func:`keel.config.delegate_profiles_dict` gives ``delegate_profiles``. The flip
398 side is the guarantee #1014 asks for — ``config_hash`` changes *iff* ``team`` does.
399 """
400 if not policy.configured:
401 return {}
402 team: dict[str, Any] = {}
403 implement: dict[str, Any] = {}
404 if policy.implement is not None:
405 implement["default"] = _seat_canonical(policy.implement)
406 if policy.implement_by_role:
407 implement["by_role"] = {
408 role: _seat_canonical(seat) for role, seat in sorted(policy.implement_by_role.items())
409 }
410 if implement:
411 team["implement"] = implement
412 if policy.gate is not None:
413 team["gate"] = _seat_canonical(policy.gate)
414 review: dict[str, Any] = {}
415 if policy.review is not None:
416 review["default"] = (
417 policy.review
418 if isinstance(policy.review, str)
419 else [_seat_canonical(seat) for seat in policy.review]
420 )
421 if policy.review_by_tier:
422 review["by_tier"] = {
423 tier: value if isinstance(value, str) else [_seat_canonical(s) for s in value]
424 for tier, value in sorted(policy.review_by_tier.items())
425 }
426 if review:
427 team["review"] = review
428 jury = {}
429 if policy.jury_mode is not None:
430 jury["mode"] = policy.jury_mode
431 if policy.jury_min_vendors is not None:
432 jury["min_vendors"] = policy.jury_min_vendors
433 # Absent when unset, like every other optional field here: a project that never names
434 # `on_unavailable` keeps the `config_hash` it had before the setting existed (#1066).
435 if policy.jury_on_unavailable is not None:
436 jury["on_unavailable"] = policy.jury_on_unavailable
437 if jury:
438 team["jury"] = jury
439 if policy.fix is not None:
440 team["fix"] = _seat_canonical(policy.fix)
441 if policy.lead is not None:
442 team["lead"] = _seat_canonical(policy.lead)
443 for key, benches in (("by_difficulty", policy.by_difficulty), ("profiles", policy.profiles)):
444 if benches:
445 team[key] = {name: _bench_canonical(b) for name, b in sorted(benches.items())}
446 return {"team": team}
449def legacy_seats(
450 implementer_agents: Mapping[str, str],
451 *,
452 provider_names: Iterable[str] = (),
453) -> dict[str, Seat]:
454 """Map the deprecated ``knobs.implementer_agents`` onto ``team.implement.by_role``.
456 ``implementer_agents`` values were documented as vendor strings in
457 ``docs/keel/models.md`` and as Claude subagent names in ``ship.md`` s4, and nothing
458 said which — #1014's opening complaint. The migration reads them the only way that
459 keeps both documented meanings working: a value that names a provider keel can
460 resolve *is* that provider; anything else is a host subagent and gets the explicit
461 ``subagent:`` prefix it always meant.
462 """
463 known = set(provider_names)
464 seats: dict[str, Seat] = {}
465 for role, value in sorted(implementer_agents.items()):
466 name = _text(value)
467 if name is None:
468 continue
469 # Split the same way `--delegate` does before deciding this is a subagent name.
470 # `docs/keel/models.md` documents `frontend: anthropic-api:claude-3-7-sonnet-…`
471 # as a legal value, and treating the whole string as one opaque name turned it
472 # into `subagent:anthropic-api:claude-…` — a host subagent that does not exist,
473 # instead of the hosted-API vendor plus its model.
474 seat = seat_from_token(name)
475 if seat.kind == "subagent" or seat.provider in known:
476 seats[role] = seat
477 else:
478 seats[role] = Seat(provider=f"{SUBAGENT_PREFIX}{name}")
479 return seats
482def seat_from_token(token: str) -> Seat:
483 """A ``--delegate``/``--review-delegate`` token -> a seat.
485 ``vendor:model`` splits on the first colon, exactly as
486 :func:`keel.agents.split_delegate` does — except for ``subagent:<name>``, whose
487 colon separates a *kind* from a name and never a model.
488 """
489 value = (token or "").strip()
490 if value.startswith(SUBAGENT_PREFIX):
491 return Seat(provider=value)
492 provider, sep, model = value.partition(":")
493 return Seat(provider=provider, model=model if (sep and model) else None)
496def _implement_seat(
497 policy: TeamPolicy,
498 *,
499 role: str | None,
500 legacy: Mapping[str, Seat] | None,
501 host_agent: str,
502 benches: Sequence[tuple[Bench, str]] = (),
503) -> tuple[Seat, str]:
504 """The implementer and the config path it came from (bench > policy > legacy > host).
506 A bench outranks ``by_role`` because it is the more specific statement: the role says
507 *what part of the system this is*, the bench says *what this particular piece of work
508 costs*, and "the hard ones go to the strong implementer" is only expressible if the
509 second wins. ``--delegate`` still outranks both — that is the caller's job, above.
510 """
511 for bench, source in benches:
512 if bench.implement is not None:
513 return bench.implement, f"{source}.implement"
514 if role is not None and role in policy.implement_by_role:
515 return policy.implement_by_role[role], f"team.implement.by_role.{role}"
516 if policy.implement is not None:
517 return policy.implement, "team.implement.default"
518 if role is not None and legacy and role in legacy:
519 return legacy[role], f"knobs.implementer_agents.{role} (deprecated)"
520 return Seat(provider=host_agent), "host"
523def jury_on_unavailable(setting: str | None) -> str:
524 """The effective ``knobs.team.jury.on_unavailable`` (#1066).
526 ``None`` — unset — is :data:`JURY_ON_UNAVAILABLE_DEFAULT`. An unrecognised value cannot
527 get past ``keel validate`` (:func:`team_issues` rejects it) but resolves to the default
528 rather than raising: this is read on the resolution path, and a config that reached it
529 must still resolve to *some* policy. The setting stays tri-state at the config boundary
530 so an explicit ``fallback`` remains distinguishable from silence, which is what
531 ``config_hash`` reads.
532 """
533 return setting if setting in JURY_ON_UNAVAILABLE else JURY_ON_UNAVAILABLE_DEFAULT
536def _panel_falls_back(availability: Mapping[str, Any] | None) -> bool:
537 """True when a measured probe says the panel is unstaffable *and* the policy allows it.
539 ``None`` — no probe ran — is False: the panel stands. Nothing here decides the policy;
540 ``decision`` was already resolved by :meth:`keel.juryavail.Availability.decision`, so
541 this module keeps exactly one reading of the operator's configured allowance.
542 """
543 if not isinstance(availability, Mapping):
544 return False
545 return availability.get("decision") == JURY_ON_UNAVAILABLE[0]
548def _panel_refuses(availability: Mapping[str, Any] | None) -> bool:
549 """True when a measured probe says the panel is unstaffable *and* the policy refuses.
551 The mirror of :func:`_panel_falls_back`, written the same way and for the same reason:
552 ``decision`` was already resolved by :meth:`keel.juryavail.Availability.decision`, so
553 the two branches of ``team.jury.on_unavailable`` are read in one module and nowhere
554 else. ``None`` — no probe ran — is False: the panel stands.
555 """
556 if not isinstance(availability, Mapping):
557 return False
558 return availability.get("decision") == JURY_ON_UNAVAILABLE[1]
561def _availability_reason(availability: Mapping[str, Any] | None) -> str:
562 """The probe's own sentence, so the warning names the seats rather than summarising."""
563 reason = availability.get("reason") if isinstance(availability, Mapping) else None
564 return reason if isinstance(reason, str) and reason.strip() else "the probe reported no detail"
567def refusal_message(availability: Mapping[str, Any], *, source: str) -> str:
568 """The message an ``on_unavailable: block`` run refuses with.
570 It names the unavailable seats, because "the panel is unavailable" without them sends
571 the operator to ``keel doctor --providers`` to learn what this run already measured.
573 ``source`` is the config path that made the panel this run's review — the resolver's
574 own reading (:func:`configured_review`), so the overlay that selected the panel is
575 named rather than the tier's policy under it. Re-exported as
576 ``keel.juryavail.refusal_message``.
577 """
578 unavailable = availability.get("unavailable")
579 unavailable = unavailable if isinstance(unavailable, Sequence) else ()
580 seats = [
581 f" - {seat.get('provider')}: {seat.get('reason')}"
582 for seat in unavailable
583 if isinstance(seat, Mapping)
584 ]
585 listed = "\n".join(seats) or " - (no provider was probed)"
586 vendors = availability.get("available_vendors")
587 if not isinstance(vendors, Sequence) or isinstance(vendors, (str, bytes)):
588 vendors = []
589 return (
590 f"{source} makes the cross-vendor jury the review for this tier, and the panel "
591 f"cannot be staffed here: {len(vendors)} vendor(s) available "
592 f"({', '.join(str(v) for v in vendors) or 'none'}), "
593 f"{availability.get('required_vendors')} required.\n"
594 f"Unavailable:\n{listed}\n"
595 "knobs.team.jury.on_unavailable is 'block', so this run refuses rather than "
596 "reviewing with a bench the policy did not ask for. Install or authenticate what "
597 f"is missing — the panel runner answers `{JURY_RUNNER_COMMAND} --doctor` and keel's "
598 "own delegates answer `keel doctor --providers` — or set on_unavailable: fallback "
599 "to let a host bench of the same size review instead."
600 )
603def configured_review(
604 policy: TeamPolicy,
605 *,
606 tier: int | None,
607 benches: Sequence[tuple[Bench, str]] = (),
608) -> tuple[tuple[Seat, ...] | str | None, str | None]:
609 """The review policy in force for this run, and the config path it came from.
611 The tier's own (:meth:`TeamPolicy.review_for`), overlaid by the first bench that names
612 a review — a ``--team`` profile, else the difficulty band — exactly as
613 :meth:`TeamPolicy.benches_for` orders them.
615 **One reading of "what is this run's review", shared by everything that asks.** The
616 resolver that seats the bench (:func:`_review_seats`) and the probe that measures the
617 panel before it (:func:`keel.providerprobe.jury_availability`) had this written twice,
618 and the copies drifted: the probe read ``review_for`` alone, so a project whose
619 ``profiles.strict.review`` or ``by_difficulty.hard.review`` named the panel resolved
620 ``review_panel: jury`` with ``availability: null`` — the panel published on a machine
621 that was never asked whether it could convene one, which is #1066 reached by a second
622 route. A predicate this one depends on may only be written here.
623 """
624 for bench, bench_source in benches:
625 if bench.review is not None:
626 return bench.review, f"{bench_source}.review"
627 return policy.review_for(tier)
630def panel_review_source(
631 policy: TeamPolicy,
632 *,
633 tier: int | None,
634 difficulty: str | None = None,
635 profile: str | None = None,
636 any_difficulty: bool = False,
637) -> str | None:
638 """The config path that makes the cross-vendor panel this run's review, or ``None``.
640 ``None`` means no route to the panel exists for these coordinates, so the panel probe
641 has nothing to measure and a project that never convenes one pays nothing.
643 ``any_difficulty`` is for a caller that cannot name the band yet:
644 :func:`keel.swarm.build_swarm_plan` scores each cluster's difficulty *while* it
645 partitions, so the probe runs before any band exists. It then asks the wider question —
646 *could* any band this policy configures make the panel the review — and the answer is a
647 deliberate superset. That matches what the record already is: availability is a fact
648 about the machine, not about a cluster, which is why
649 :func:`keel.providerprobe.jury_availability_for_any_tier` sweeps tiers the same way.
650 """
651 bands = (None, *DIFFICULTY_BANDS) if any_difficulty else (difficulty,)
652 for band in bands:
653 configured, source = configured_review(
654 policy, tier=tier, benches=policy.benches_for(difficulty=band, profile=profile)
655 )
656 if configured == JURY_PANEL:
657 return source
658 return None
661def _review_seats(
662 policy: TeamPolicy,
663 *,
664 tier: int | None,
665 default_count: int,
666 reviewer_override: int | None,
667 host_agent: str,
668 jury_disabled: bool = False,
669 jury_advisory: bool = False,
670 benches: Sequence[tuple[Bench, str]] = (),
671 jury_availability: Mapping[str, Any] | None = None,
672) -> tuple[tuple[Seat, ...], str, str, list[str]]:
673 """Reviewer seats, the panel, the source, and any warnings.
675 Precedence: a tier whose policy is ``jury`` empties the reviewer bench (the panel
676 *is* the review); otherwise ``--reviewers`` wins over the policy's seat count, which
677 wins over the tier-derived default.
679 ``jury_availability`` is the one exception, and it is not a preference (#1066). It is
680 :func:`keel.juryavail.assess`'s verdict on whether this machine can convene the panel
681 at all, measured from the same probe ``keel doctor --providers`` prints. When it says
682 the panel is unstaffable and ``team.jury.on_unavailable`` is ``fallback``, the tier
683 resolves onto the **tier's own** seat count, staffed from the host, and the record says
684 so. The seat count and the evidence requirement do not move; only who sits does — which
685 is why ``--reviewers`` stays ignored on a fallback bench exactly as it is ignored while
686 the panel sits. A flag that was inert on a staffable panel and *lowered* the tier's
687 requirement the moment the probe failed would make a failed probe a policy change.
688 Under ``block`` the panel stays the panel and the run refuses — *here*, on the first
689 line, with :class:`JuryUnavailableError` (#1068). The probe only measures; it is asked
690 once per surface and may staff many benches, so it cannot know whether this particular
691 cluster's review is the panel. This can, and it is the only place that resolves a
692 bench, so it is the only place the check can neither be forgotten nor over-reach.
694 **The bench is a pure function of config + tier + role + the explicit ``--reviewers``
695 and ``--review-delegate`` overrides, and of nothing else.** In particular it does not
696 depend on the jury flags. It cannot: every surface accepts them since #1043, but
697 nothing makes a *run* pass them uniformly — keel's CI passes ``--no-jury`` to
698 ``evidence-verify`` on every run while passing it to neither ``ship`` nor ``plan``.
699 A bench that moved with that flag would have ``plan`` requiring
700 a jury verdict from zero reviewers while ``evidence-verify`` demanded three host
701 verdicts of the same PR, which is the same contract disagreement in a new place.
703 So ``jury_disabled`` / ``jury_advisory`` are recorded, never applied: on a tier whose
704 review policy is the panel, the panel **is** the review, and a per-run flag does not
705 get to remove the only review that tier has. :func:`keel.ship.resolve_jury` keeps the
706 verdict required for the same reason.
707 """
708 configured, source = configured_review(policy, tier=tier, benches=benches)
709 warnings: list[str] = []
710 if configured == JURY_PANEL and _panel_refuses(jury_availability):
711 # The one place the run refuses (#1068). Not at the probe: the probe answers a
712 # question about the *machine*, and a caller may resolve many benches from one
713 # answer. `keel swarm plan` measures before it has scored a single cluster, asking
714 # the widest question there is — could *any* tier or band name the panel — so a
715 # refusal there refused a swarm of entirely non-panel clusters on an unstaffable
716 # host. By here `configured` is this cluster's own resolved review, so the run
717 # refuses when the panel really is what it would have dispatched, and `source`
718 # names the config path that made it so.
719 raise JuryUnavailableError(refusal_message(jury_availability, source=source))
720 fell_back = configured == JURY_PANEL and _panel_falls_back(jury_availability)
721 if fell_back:
722 # The panel is this tier's review and this machine cannot convene it. Fall through
723 # to the ordinary path with no configured seats, which *is* "a tier without a
724 # panel": the tier's own count, staffed from the host. Reached only from a
725 # measured probe, and the reason travels in `warnings` and in the assignment's
726 # `jury.availability` block so no reader has to re-derive it.
727 warnings.append(
728 f"{source} makes the jury the review for this tier, but the panel cannot be "
729 f"staffed here; knobs.team.jury.on_unavailable is 'fallback', so a host bench "
730 f"of {default_count} seat(s) reviews instead — the same count and the same "
731 f"evidence, different reviewers, and they share one vendor so this review "
732 f"carries no cross-vendor independence claim. "
733 f"{_availability_reason(jury_availability)}"
734 )
735 if reviewer_override is not None:
736 # The same flag, ignored the same way, whether or not the panel could sit.
737 # `--reviewers` is inert on a panel tier — the panel *is* the review — and a
738 # fallback may not turn that inert flag into a live one: a bench sized by
739 # `--reviewers 2` would publish a two-verdict evidence requirement where the
740 # tier asks for three, so a probe failure would have *lowered* the tier's
741 # policy. The fallback changes who sat, never how many.
742 warnings.append(
743 f"--reviewers {reviewer_override} ignored: {source} makes the jury the "
744 f"review panel, and the host bench standing in for it is the tier's own "
745 f"{default_count} seat(s) — a fallback changes who reviews, not how many"
746 )
747 reviewer_override = None
748 configured = ()
749 if configured == JURY_PANEL:
750 if reviewer_override is not None:
751 warnings.append(
752 f"--reviewers {reviewer_override} ignored: {source} makes the jury the "
753 "review panel, so there are no host reviewer slots to size"
754 )
755 ignored = [
756 flag
757 for flag, passed in (("--no-jury", jury_disabled), ("--jury-advisory", jury_advisory))
758 if passed
759 ]
760 if ignored:
761 warnings.append(
762 f"{' and '.join(ignored)} does not apply: this tier's review is the jury "
763 f"panel ({source}). The panel is the review, so its verdict stays required"
764 )
765 return (), JURY_PANEL, source, warnings
766 seats = configured if isinstance(configured, tuple) else ()
767 tier_source = "risk-tier" if tier is not None else "unresolved"
768 if reviewer_override is not None:
769 count, source = reviewer_override, "override"
770 elif seats:
771 count = len(seats)
772 else:
773 count, source = default_count, tier_source
774 if fell_back:
775 # Named, not inherited from the tier's config path: a reader of the published
776 # contract must be able to tell a bench seated because the panel could not be
777 # from a tier that simply never had one.
778 source = JURY_FALLBACK_SOURCE
779 resolved = tuple(
780 seats[index] if index < len(seats) else Seat(provider=host_agent) for index in range(count)
781 )
782 padded = max(0, count - len(seats))
783 if fell_back:
784 # The pad warning's advice ("name the extra seats in knobs.team.review") is wrong
785 # here: this tier *did* name its reviewers — it named the panel. The fallback
786 # warning above already carries the one fact that advice was protecting.
787 padded = 0
788 if padded > 1 or (padded and any(seat.provider == host_agent for seat in seats)):
789 # Two conditions, because there are two ways the pad duplicates. The host may
790 # already be a configured seat (`[claude, codex]` + `--reviewers 3`), or it may
791 # not be and simply get seated twice (`[codex]` + `--reviewers 3` ->
792 # `[codex, claude, claude]`): the second is a duplicate between two *padded*
793 # slots, which a check against the configured seats alone never sees.
794 #
795 # Comparing provider names, not vendors: resolving a name to its vendor needs the
796 # registry, which this module deliberately cannot reach. A repeated *name* is
797 # already a repeated vendor, and `require_distinct_vendors` rejects it at the
798 # evidence gate long after the run.
799 seated = any(seat.provider == host_agent for seat in seats)
800 where = "which is already seated" if seated else "filling more than one slot"
801 warnings.append(
802 f"{padded} reviewer slot(s) padded with the host agent {host_agent!r}, "
803 f"{where}; those reviewers cannot return distinct vendor provenance — name "
804 "the extra seats in knobs.team.review, or lower --reviewers"
805 )
806 if seats and len(seats) > count:
807 warnings.append(
808 f"{len(seats)} reviewer seat(s) configured but only {count} slot(s) are "
809 "staffed; the surplus seats are not dispatched"
810 )
811 return resolved, "reviewers", source or tier_source, warnings
814#: The most reviewer seats a tier may name — keel's reviewer vocabulary is A/B/C, and
815#: :func:`keel.ship.reviewer_focuses` has focus coverage for exactly those three.
816MAX_REVIEW_SEATS = 3
818#: Slot letters, by seat count. Deliberately not ``A``/``B``/``C`` for two seats: keel
819#: merges the B focus into A at that count, so the second reviewer is slot **C**. Mirrors
820#: :func:`keel.ship.reviewer_focuses`, which cannot be imported here (``ship`` imports
821#: this module); ``tests/test_team.py`` asserts the two agree for every valid count.
822_SLOT_LABELS = {0: (), 1: ("A",), 2: ("A", "C"), 3: ("A", "B", "C")}
825def slot_labels(count: int) -> tuple[str, ...]:
826 """Slot letters for ``count`` reviewer seats, matching ``ship.reviewer_focuses``.
828 Total for any non-negative count, including counts keel's own vocabulary does not
829 have a focus for. It is a labelling function: running short here turned an
830 out-of-range reviewer count into an ``IndexError`` from the middle of the resolver
831 rather than the documented ``ValueError`` the caller raises for it.
832 """
833 if count in _SLOT_LABELS:
834 return _SLOT_LABELS[count]
835 return tuple(chr(ord("A") + index) for index in range(max(0, count)))
838def _effective_effort(
839 seat: Seat, *, bench_effort: str | None, flag_effort: str | None
840) -> str | None:
841 """The effort this seat actually runs at: ``--effort`` > the seat's own > the bench's.
843 The flag wins because it is the operator speaking about *this run*; the seat wins over
844 the bench because a seat that names a provider **and** an effort is one statement, and
845 the bench's ``effort`` is the default for seats that did not bother.
846 """
847 return flag_effort or seat.effort or bench_effort
850def _seat_at_effort(seat: Seat, effort: str | None) -> Seat:
851 """``seat`` running at ``effort`` — the same object when nothing changes."""
852 if effort == seat.effort:
853 return seat
854 return Seat(
855 provider=seat.provider,
856 model=seat.model,
857 effort=effort,
858 distinct_from=seat.distinct_from,
859 )
862def _lead_seat(
863 policy: TeamPolicy,
864 *,
865 benches: Sequence[tuple[Bench, str]],
866 host_agent: str,
867) -> tuple[Seat, str]:
868 """Who coordinates this batch of work: bench > ``team.lead`` > the host agent.
870 Always a seat, never ``None``: an unconfigured project still has a lead — the agent
871 driving the run — and a swarm coordinator that had to special-case "no lead" would
872 grow a second answer to a question this one already answers.
873 """
874 for bench, source in benches:
875 if bench.lead is not None:
876 return bench.lead, f"{source}.lead"
877 if policy.lead is not None:
878 return policy.lead, "team.lead"
879 return Seat(provider=host_agent), "host"
882def resolve_assignment(
883 policy: TeamPolicy,
884 *,
885 tier: int | None = None,
886 role: str | None = None,
887 default_count: int = 2,
888 reviewer_override: int | None = None,
889 delegate: str | None = None,
890 review_delegates: Sequence[str] = (),
891 host_agent: str = HOST_DEFAULT,
892 legacy: Mapping[str, Seat] | None = None,
893 jury_disabled: bool = False,
894 jury_advisory: bool = False,
895 difficulty: str | None = None,
896 team_profile: str | None = None,
897 effort: str | None = None,
898 jury_availability: Mapping[str, Any] | None = None,
899) -> dict[str, Any]:
900 """Who runs this ship: implementer, gate, reviewer slots, jury, fix.
902 Deterministic for identical inputs, which is what lets ``keel plan`` and
903 ``keel ship`` render the same team and any host run it. Per-run flags stay per-run
904 overrides: ``--delegate`` replaces the implementer, and each ``--review-delegate``
905 replaces one reviewer slot **positionally** — the first flag is slot A, the second
906 slot B — so a two-vendor panel is expressible from the command line without a config
907 change. A flag past the last slot is reported in ``warnings`` rather than silently
908 dropped or silently growing the panel.
910 ``difficulty`` and ``team_profile`` select a :class:`Bench` (see
911 :meth:`TeamPolicy.benches_for`), which is how a batch runner says *this cluster is
912 hard, give it the strong implementer at high effort; the easy ones go to the cheap
913 model*. They are inputs to this one resolver rather than a second one beside it, so
914 ``keel ship``, ``keel plan`` and ``keel swarm-plan`` cannot disagree about who runs a
915 given issue.
916 """
917 benches = policy.benches_for(difficulty=difficulty, profile=team_profile)
918 implementer, implementer_source = _implement_seat(
919 policy, role=role, legacy=legacy, host_agent=host_agent, benches=benches
920 )
921 if delegate:
922 implementer, implementer_source = seat_from_token(delegate), "flag:--delegate"
923 bench_effort = next((b.effort for b, _ in benches if b.effort is not None), None)
924 implementer = _seat_at_effort(
925 implementer, _effective_effort(implementer, bench_effort=bench_effort, flag_effort=effort)
926 )
927 lead, lead_source = _lead_seat(policy, benches=benches, host_agent=host_agent)
928 seats, panel, review_source, warnings = _review_seats(
929 policy,
930 tier=tier,
931 default_count=default_count,
932 reviewer_override=reviewer_override,
933 host_agent=host_agent,
934 jury_disabled=jury_disabled,
935 jury_advisory=jury_advisory,
936 benches=benches,
937 jury_availability=jury_availability,
938 )
939 if team_profile is not None and team_profile not in policy.profiles:
940 warnings.append(
941 f"--team {team_profile!r} names no knobs.team.profiles entry; this run is "
942 f"staffed from the configured policy instead. Known: "
943 f"{', '.join(sorted(policy.profiles)) or 'none'}"
944 )
945 labels = slot_labels(len(seats))
946 sources = [review_source] * len(seats)
947 overrides = [token for token in review_delegates if (token or "").strip()]
948 for index, token in enumerate(overrides):
949 if index >= len(seats):
950 warnings.append(
951 f"--review-delegate {token!r} names reviewer slot {index + 1}, but only "
952 f"{len(seats)} reviewer slot(s) are staffed; it is not dispatched"
953 )
954 continue
955 seats = seats[:index] + (seat_from_token(token),) + seats[index + 1 :]
956 sources[index] = "flag:--review-delegate"
957 gate = policy.gate
958 gate_record = None
959 if gate is not None:
960 distinct_ok = gate.distinct_from != IMPLEMENTER or gate.name != implementer.name
961 gate_record = gate.as_dict(source="team.gate")
962 gate_record["distinct_ok"] = distinct_ok
963 if not distinct_ok:
964 warnings.append(
965 f"team.gate.provider {gate.provider!r} is the resolved implementer and "
966 "gate.distinct_from is 'implementer'; the gate review would be a second "
967 "opinion from the first opinion"
968 )
969 fix_seat = policy.fix if policy.fix is not None else Seat(provider=IMPLEMENTER)
970 fix_source = "team.fix" if policy.fix is not None else "default"
971 resolved_fix = implementer if fix_seat.kind == "alias" else fix_seat
972 fix_record = resolved_fix.as_dict(source=fix_source)
973 fix_record["alias"] = IMPLEMENTER if fix_seat.kind == "alias" else None
974 return {
975 "configured": policy.configured,
976 "role": role,
977 "tier": tier,
978 "difficulty": difficulty,
979 "team_profile": team_profile,
980 "bench": [source for _bench, source in benches],
981 "lead": lead.as_dict(source=lead_source),
982 "implementer": implementer.as_dict(source=implementer_source),
983 "effort": implementer.effort,
984 "gate": gate_record,
985 "review_panel": panel,
986 "reviewers": [
987 seat.as_dict(source=sources[index], slot=labels[index])
988 for index, seat in enumerate(seats)
989 ],
990 "reviewer_count": len(seats),
991 "reviewer_source": review_source,
992 "jury": {
993 "mode": policy.jury_mode,
994 "min_vendors": policy.jury_min_vendors or DEFAULT_MIN_VENDORS,
995 "panel_is_review": panel == JURY_PANEL,
996 # What the tier's policy asked for, before availability had its say. Without
997 # it a fallback run's assignment is indistinguishable from a tier that never
998 # configured a panel — which is the silent downgrade #1066 exists to refuse.
999 "panel_configured": review_source == JURY_FALLBACK_SOURCE or panel == JURY_PANEL,
1000 "on_unavailable": jury_on_unavailable(policy.jury_on_unavailable),
1001 # `None` until something measured it: `keel plan --no-probe`-shaped callers
1002 # and every non-panel tier resolve without a probe, and an absent measurement
1003 # must not read as "we checked and it was fine".
1004 "availability": (
1005 dict(jury_availability) if isinstance(jury_availability, Mapping) else None
1006 ),
1007 },
1008 "fix": fix_record,
1009 "warnings": warnings,
1010 }
1013def require_distinct_vendors(setting: bool | None) -> bool:
1014 """The effective ``evidence_require_distinct_vendors``.
1016 **Opt-in** (#1065). ``None`` is *unset* and resolves to ``False``: the knob asserts
1017 that the required verdicts came from *independent* opinions, and that is a claim only
1018 the project can make. It is a property a cross-vendor panel provides, not one every
1019 high-tier review has to carry — a person with a single agent CLI installed must still
1020 be able to land a TIER-3 change without configuring anything. A project that wants
1021 the independence claim enforced says so, and the requirement then lives in a file a
1022 reviewer can read rather than in a default nobody chose.
1024 The setting is still tri-state at the config boundary (``None`` unset, ``True``,
1025 ``False``) so an explicit ``false`` stays distinguishable from silence, which is what
1026 ``config_hash`` and the wizard read.
1027 """
1028 return bool(setting)
1031def _seat_paths(policy: TeamPolicy) -> list[tuple[str, Seat]]:
1032 """Every seat in the policy with the config path it sits at, in a stable order."""
1033 paths: list[tuple[str, Seat]] = []
1034 if policy.implement is not None:
1035 paths.append(("implement.default", policy.implement))
1036 paths.extend(
1037 (f"implement.by_role.{role}", seat)
1038 for role, seat in sorted(policy.implement_by_role.items())
1039 )
1040 if policy.gate is not None:
1041 paths.append(("gate", policy.gate))
1042 if isinstance(policy.review, tuple):
1043 paths.extend((f"review.default[{i}]", seat) for i, seat in enumerate(policy.review))
1044 for tier, value in sorted(policy.review_by_tier.items()):
1045 if isinstance(value, tuple):
1046 paths.extend((f"review.by_tier.{tier}[{i}]", seat) for i, seat in enumerate(value))
1047 if policy.fix is not None:
1048 paths.append(("fix", policy.fix))
1049 if policy.lead is not None:
1050 paths.append(("lead", policy.lead))
1051 paths.extend(_bench_seat_paths("by_difficulty", policy.by_difficulty))
1052 paths.extend(_bench_seat_paths("profiles", policy.profiles))
1053 return paths
1056def _bench_seat_paths(key: str, benches: Mapping[str, Bench]) -> list[tuple[str, Seat]]:
1057 """Every seat of every bench in one table, with its config path."""
1058 paths: list[tuple[str, Seat]] = []
1059 for name, bench in sorted(benches.items()):
1060 where = f"{key}.{name}"
1061 if bench.lead is not None:
1062 paths.append((f"{where}.lead", bench.lead))
1063 if bench.implement is not None:
1064 paths.append((f"{where}.implement", bench.implement))
1065 if isinstance(bench.review, tuple):
1066 paths.extend((f"{where}.review[{i}]", seat) for i, seat in enumerate(bench.review))
1067 return paths
1070def team_issues(
1071 raw: Any,
1072 *,
1073 source: str,
1074 profiles: Mapping[str, str] | None = None,
1075 implementer_agents: Mapping[str, str] | None = None,
1076) -> list[str]:
1077 """Semantic errors for ``knobs.team`` (empty == valid).
1079 The schema owns the *shape*; this owns the *meaning* — which providers exist, which
1080 of them can honour an ``effort``, that the mandatory gate review really is a second
1081 opinion, and that a ``review.by_tier`` entry is either reviewer seats or the jury.
1083 ``implementer_agents`` is the deprecated per-role knob. It is read here because it
1084 still resolves implementers (:func:`legacy_seats`), so a ``gate`` declared
1085 ``distinct_from: implementer`` has to be checked against those seats too — otherwise
1086 ``implementer_agents: {core: codex}`` beside ``gate: {provider: codex}`` passes
1087 validation and the "mandatory second opinion" is the first opinion again.
1089 ``profiles`` maps a ``knobs.delegate_profiles`` name to its vendor. A machine-level
1090 ``~/.keel/providers.yaml`` entry is deliberately **not** consulted: validation has to
1091 give the same answer on every machine, and a policy that only validates where its
1092 author's home directory does is a policy the next operator cannot read.
1093 """
1094 if raw is None:
1095 return []
1096 if not isinstance(raw, Mapping):
1097 return [] # the schema already reported the wrong shape
1098 profiles = dict(profiles or {})
1099 known = {**{vendor: vendor for vendor in BUILTIN_DELEGATE_VENDORS}, **profiles}
1100 policy = parse_team(raw)
1101 errors: list[str] = []
1102 errors.extend(_tier_key_issues(raw, source=source))
1103 errors.extend(_band_key_issues(raw, source=source))
1104 for path, seat in _seat_paths(policy):
1105 where = f"{source}.{path}"
1106 vendor = _seat_vendor(seat, path=path, known=known, where=where, errors=errors)
1107 if seat.effort is None:
1108 continue
1109 errors.extend(_effort_issues(seat.effort, seat=seat, vendor=vendor, where=where))
1110 legacy = legacy_seats(implementer_agents or {}, provider_names=known)
1111 errors.extend(_bench_effort_issues(policy, source=source, known=known, legacy=legacy))
1112 errors.extend(_gate_issues(policy, source=source, legacy=legacy))
1113 errors.extend(_review_issues(policy, source=source))
1114 return errors
1117def _effort_issues(
1118 effort: str,
1119 *,
1120 seat: Seat,
1121 vendor: str | None,
1122 where: str,
1123 applies_to: str | None = None,
1124) -> list[str]:
1125 """The ways an ``effort`` is not honourable by ``seat``, worded the same everywhere.
1127 Extracted so a bench-level ``effort`` (:func:`_bench_effort_issues`) is judged by
1128 exactly the rules a seat-level one is, in exactly the same words. Two copies of these
1129 three sentences is how ``by_difficulty.hard.effort`` came to bypass the checks
1130 ``implement.default.effort`` has passed since #1014.
1132 ``applies_to`` names the seat the effort would land on, for the bench case where the
1133 effort and the seat it breaks are written in different places.
1135 These are #1014's three rules and only those, so the two callers cannot drift. The
1136 subagent rule is deliberately *not* here: it applies to a bench effort alone, and
1137 :func:`_bench_effort_issues` says why.
1138 """
1139 tail = "" if applies_to is None else f" (applied to the implementer at {applies_to})"
1140 if effort not in EFFORTS:
1141 return [f"{where}: unknown effort {effort!r}; valid: {', '.join(EFFORTS)}{tail}"]
1142 if vendor is None:
1143 return [] # an unresolvable provider is already reported; do not pile on
1144 if not supports_effort(vendor):
1145 return [
1146 f"{where}: provider {seat.provider!r} ({vendor}) has no spelling for "
1147 f"reasoning effort, so effort {effort!r} would be silently dropped "
1148 f"— drop the field, or name a provider that can honour it{tail}"
1149 ]
1150 if effort_needs_model(vendor) and seat.model is None:
1151 return [
1152 f"{where}: agy spells reasoning effort as a model suffix "
1153 f"(e.g. gemini-3.8-flash-high), so effort {effort!r} needs a "
1154 f"'model' beside it{tail}"
1155 ]
1156 return []
1159def _bench_effort_issues(
1160 policy: TeamPolicy,
1161 *,
1162 source: str,
1163 known: Mapping[str, str],
1164 legacy: Mapping[str, Seat],
1165) -> list[str]:
1166 """A bench ``effort`` has to be honourable by every implementer it could land on.
1168 A seat's own ``effort`` sits next to the provider it applies to, so an operator
1169 reading one line sees both halves. A bench's does not: ``by_difficulty.hard.effort``
1170 lands on whichever implementer resolves for that band, which may be written in
1171 another file's worth of config — action at a distance, and exactly the case where a
1172 silently-dropped effort is invisible. So it is checked against every seat it could
1173 reach, by the same rules and in the same words.
1175 Which seats those are follows the resolution order. A bench naming its own
1176 ``implement`` seat can only land there. One that does not falls through to the
1177 role/default/legacy implementers, and any of them is reachable. The host-agent
1178 fallback is deliberately not checked: the host is a per-run flag, and a policy that
1179 only validates against one operator's default is the kind of rule #1014 refused.
1181 A **subagent** target is an error here and not at seat level, and the difference is
1182 the point. ``fix: {provider: "subagent:x", effort: high}`` pairs the two on one line:
1183 the operator saw both halves and #1014 chose to tolerate it. A bench effort lands on
1184 a seat written somewhere else entirely, so the same pairing is one nobody ever read.
1185 """
1186 errors: list[str] = []
1187 for table, benches in (("by_difficulty", policy.by_difficulty), ("profiles", policy.profiles)):
1188 for name, bench in sorted(benches.items()):
1189 if bench.effort is None:
1190 continue
1191 where = f"{source}.{table}.{name}.effort"
1192 for seat_path, seat in _bench_effort_targets(
1193 policy, table=table, bench=bench, legacy=legacy
1194 ):
1195 # A seat naming its own effort never receives the bench's, so a bench
1196 # effort it could not honour is not a defect — the seat's wins.
1197 if seat.effort is not None:
1198 continue
1199 if seat.kind == "subagent":
1200 errors.append(
1201 f"{where}: {seat.provider!r} is a host subagent, which has no "
1202 f"reasoning-effort dial keel can set, so effort {bench.effort!r} "
1203 "would be silently dropped — drop the field, or name a provider "
1204 f"that can honour it (applied to the implementer at {seat_path})"
1205 )
1206 continue
1207 errors.extend(
1208 _effort_issues(
1209 bench.effort,
1210 seat=seat,
1211 vendor=known.get(seat.provider),
1212 where=where,
1213 applies_to=seat_path,
1214 )
1215 )
1216 return errors
1219def _bench_effort_targets(
1220 policy: TeamPolicy,
1221 *,
1222 table: str,
1223 bench: Bench,
1224 legacy: Mapping[str, Seat],
1225) -> list[tuple[str, Seat]]:
1226 """Implementer seats a bench's ``effort`` could land on, with their config paths.
1228 *Reachable*, not merely *present*. One run resolves exactly one difficulty band and
1229 at most one ``--team`` profile, so two entries of the same table never apply
1230 together: a ``by_difficulty.hard`` effort can never meet ``by_difficulty.easy``'s
1231 implementer. Checking against siblings reported errors for combinations no run can
1232 produce, which is how a validator teaches people to ignore it.
1234 What is left is genuinely reachable. A bench naming its own ``implement`` seat can
1235 only land there. Otherwise the implementer comes from the *other* table (a ``--team``
1236 profile supplying the seat while the band supplies the effort, or the reverse), or
1237 from the role/default/legacy seats every run falls through to.
1238 """
1239 if bench.implement is not None:
1240 return [("this bench's own implement seat", bench.implement)]
1241 other = "profiles" if table == "by_difficulty" else "by_difficulty"
1242 targets = [
1243 (path, seat)
1244 for path, seat in _seat_paths(policy)
1245 if path.startswith("implement")
1246 or (path.startswith(f"{other}.") and path.endswith(".implement"))
1247 ]
1248 targets.extend(
1249 (f"knobs.implementer_agents.{role}", seat) for role, seat in sorted(legacy.items())
1250 )
1251 return targets
1254def _seat_vendor(
1255 seat: Seat,
1256 *,
1257 path: str,
1258 known: Mapping[str, str],
1259 where: str,
1260 errors: list[str],
1261) -> str | None:
1262 """The vendor behind a seat, appending the error when the provider is not resolvable."""
1263 if seat.kind == "subagent":
1264 if not seat.name:
1265 errors.append(
1266 f"{where}: {SUBAGENT_PREFIX!r} needs a subagent name after it, "
1267 "e.g. 'subagent:backend-developer'"
1268 )
1269 return None
1270 if seat.kind == "alias":
1271 if path != "fix":
1272 errors.append(
1273 f"{where}: provider {IMPLEMENTER!r} means 'whoever implemented this "
1274 "change' and is only valid at team.fix.provider"
1275 )
1276 return None
1277 if seat.provider not in known:
1278 errors.append(
1279 f"{where}: unknown provider {seat.provider!r}; name a built-in vendor, a "
1280 f"knobs.delegate_profiles entry, or a host subagent as "
1281 f"'{SUBAGENT_PREFIX}{seat.provider}'. Known: {', '.join(sorted(known))}"
1282 )
1283 return None
1284 return known[seat.provider]
1287def _tier_key_issues(raw: Mapping[str, Any], *, source: str) -> list[str]:
1288 """``review.by_tier`` keys must be the quoted tier strings ``"1"``/``"2"``/``"3"``."""
1289 review = raw.get("review")
1290 by_tier = review.get("by_tier") if isinstance(review, Mapping) else None
1291 if not isinstance(by_tier, Mapping):
1292 return []
1293 errors = []
1294 for key in by_tier:
1295 if key in TIERS:
1296 continue
1297 errors.append(
1298 f"{source}.review.by_tier: {key!r} is not a risk tier; quote the key as "
1299 f'"1", "2" or "3" (YAML reads a bare 1: as an integer key, which a JSON '
1300 "schema cannot describe)"
1301 )
1302 return errors
1305def _band_key_issues(raw: Mapping[str, Any], *, source: str) -> list[str]:
1306 """``by_difficulty`` keys must be difficulty bands the scorer can produce.
1308 A typo here is silent otherwise: ``medium:`` beside ``easy:``/``hard:`` never matches
1309 anything the scorer emits, so the bench an operator wrote is simply never staffed and
1310 the run looks like the table was ignored.
1311 """
1312 by_difficulty = raw.get("by_difficulty")
1313 if not isinstance(by_difficulty, Mapping):
1314 return []
1315 return [
1316 f"{source}.by_difficulty: {key!r} is not a difficulty band; valid: "
1317 f"{', '.join(DIFFICULTY_BANDS)}"
1318 for key in by_difficulty
1319 if key not in DIFFICULTY_BANDS
1320 ]
1323def _gate_issues(
1324 policy: TeamPolicy,
1325 *,
1326 source: str,
1327 legacy: Mapping[str, Seat] | None = None,
1328) -> list[str]:
1329 """The mandatory second opinion must be able to *be* a second opinion."""
1330 gate = policy.gate
1331 if gate is None:
1332 return []
1333 if gate.distinct_from is not None and gate.distinct_from != IMPLEMENTER:
1334 return [
1335 f"{source}.gate: distinct_from {gate.distinct_from!r} is not a seat; the only "
1336 f"supported value is {IMPLEMENTER!r}"
1337 ]
1338 if gate.distinct_from != IMPLEMENTER:
1339 return []
1340 implementers = {
1341 seat.name: f"{source}.{path}"
1342 for path, seat in _seat_paths(policy)
1343 if path.startswith("implement") or path.endswith(".implement")
1344 }
1345 # A role the policy does not name still resolves through the deprecated knob, so a
1346 # gate matching one of those seats is the same defect wearing an older spelling.
1347 for role, seat in sorted((legacy or {}).items()):
1348 implementers.setdefault(seat.name, f"knobs.implementer_agents.{role}")
1349 clash = implementers.get(gate.name)
1350 if clash is None:
1351 return []
1352 return [
1353 f"{source}.gate: provider {gate.provider!r} is also the implementer at "
1354 f"{clash}, and gate.distinct_from is {IMPLEMENTER!r} — a gate review "
1355 "from the vendor that wrote the change is not a second opinion"
1356 ]
1359def _review_issues(policy: TeamPolicy, *, source: str) -> list[str]:
1360 """A tier's review policy is reviewer seats or the ``jury`` literal, nothing else."""
1361 errors = []
1362 entries: list[tuple[str, tuple[Seat, ...] | str]] = [
1363 (f"review.by_tier.{tier}", value) for tier, value in sorted(policy.review_by_tier.items())
1364 ]
1365 if policy.review is not None:
1366 entries.append(("review.default", policy.review)) # seats or the jury literal
1367 for key, benches in (("by_difficulty", policy.by_difficulty), ("profiles", policy.profiles)):
1368 entries.extend(
1369 (f"{key}.{name}.review", bench.review)
1370 for name, bench in sorted(benches.items())
1371 if bench.review is not None
1372 )
1373 for path, value in entries:
1374 if isinstance(value, str) and value != JURY_PANEL:
1375 errors.append(
1376 f"{source}.{path}: {value!r} is neither a list of reviewer seats nor "
1377 f"{JURY_PANEL!r} (the cross-vendor panel as the review)"
1378 )
1379 elif isinstance(value, tuple) and not value:
1380 errors.append(
1381 f"{source}.{path}: an empty reviewer list would leave the change with no "
1382 f"review; use {JURY_PANEL!r} for the panel, or name at least one seat"
1383 )
1384 elif isinstance(value, tuple) and len(value) > MAX_REVIEW_SEATS:
1385 errors.append(
1386 f"{source}.{path}: {len(value)} reviewer seats, but keel dispatches at "
1387 f"most {MAX_REVIEW_SEATS} (slots A/B/C, one focus each); use "
1388 f"{JURY_PANEL!r} for a wider panel"
1389 )
1390 if policy.jury_mode is not None and policy.jury_mode not in JURY_MODES:
1391 errors.append(
1392 f"{source}.jury.mode: unknown mode {policy.jury_mode!r}; valid: {', '.join(JURY_MODES)}"
1393 )
1394 elif policy.jury_mode == "advisory":
1395 # "The panel is the review" plus "the panel does not gate" is a tier with no
1396 # enforceable review at all: there are no host reviewer slots to fall back on, and
1397 # an advisory verdict is not required evidence. Refused here rather than
1398 # discovered as a merge that sailed through the tier the project marked strictest.
1399 panels = sorted(path for path, value in entries if value == JURY_PANEL)
1400 if panels:
1401 errors.append(
1402 f"{source}.jury.mode: 'advisory' leaves {', '.join(panels)} with no "
1403 "enforceable review — that tier has no host reviewers, so an advisory "
1404 "panel requires nothing. Use 'gating' for a jury panel, or name reviewer "
1405 "seats for that tier"
1406 )
1407 if policy.jury_on_unavailable is not None and policy.jury_on_unavailable not in (
1408 JURY_ON_UNAVAILABLE
1409 ):
1410 errors.append(
1411 f"{source}.jury.on_unavailable: unknown policy "
1412 f"{policy.jury_on_unavailable!r}; valid: {', '.join(JURY_ON_UNAVAILABLE)}"
1413 )
1414 return errors