Coverage for src/keel/providerprobe.py: 100%
167 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-02 20:26 +0000
1"""Thin I/O: probe every provider keel can dispatch to (#1011).
3The pure half — what a provider *is*, where the registry lives, what may clash with
4what — is :mod:`keel.providers`. This module answers the machine-dependent question:
5**is it usable here, right now?** Every edge is injectable (``_which``, ``_run``,
6``_env``, ``_opener``) so the whole surface is unit-testable offline.
8Three rules hold for every probe:
10* **Time-boxed.** A subprocess gets :data:`PROBE_TIMEOUT_S`, the Ollama HTTP call
11 :data:`HTTP_TIMEOUT_S`. ``keel doctor --providers`` must answer in seconds, not
12 hang behind a CLI waiting on a login prompt.
13* **Fail-soft.** Nothing here raises. A missing binary, a non-zero exit, a timeout,
14 an unreachable server, a malformed response — each becomes ``available: False``
15 with a reason an operator can act on.
16* **Names, never values.** An API-key probe reports that ``ANTHROPIC_API_KEY`` is
17 set or not. The key itself is never read into a result, printed, or logged.
19The only URL dialed is :data:`keel.providers.OLLAMA_TAGS_URL`, a hardcoded loopback
20constant, through the shared non-redirecting opener. A registry- or config-supplied
21endpoint is checked for *key presence* only: keel does not make outbound requests to
22an address a file names just because ``doctor`` was run.
23"""
25from __future__ import annotations
27import json
28import os
29import re
30import shutil
31import urllib.request
32from collections.abc import Callable, Iterable, Mapping
33from dataclasses import dataclass
35from . import api_delegate, juryavail, providers, runner, team
36from .providers import Provider, Registry
38#: Wall-clock seconds a single provider subprocess may take.
39PROBE_TIMEOUT_S = 5
40#: Wall-clock seconds the Ollama tag listing may take.
41HTTP_TIMEOUT_S = 3
43SCHEMA_VERSION = "keel.providers.v1"
46@dataclass(frozen=True)
47class ProbeResult:
48 """One provider's probe outcome (JSON-stable via :meth:`as_dict`)."""
50 provider: Provider
51 available: bool
52 reason: str
53 #: Models the provider itself reported (``agy models``, Ollama ``/api/tags``).
54 #: Empty when the provider exposes no listing or the listing failed.
55 models: tuple[str, ...] = ()
57 def as_dict(self) -> dict[str, object]:
58 record = self.provider.as_dict()
59 record.update(
60 {
61 "available": self.available,
62 "reason": self.reason,
63 "models": list(self.models),
64 }
65 )
66 return record
69def probe_providers(
70 plan: Iterable[Provider],
71 *,
72 _which: Callable[[str], str | None] = shutil.which,
73 _run: Callable[..., runner.CommandResult] = runner.run_argv,
74 _env: Mapping[str, str] | None = None,
75 _opener=None,
76) -> tuple[ProbeResult, ...]:
77 """Probe each planned provider in order. Never raises; order is the plan's."""
78 env = os.environ if _env is None else _env
79 results = []
80 for provider in plan:
81 try:
82 available, reason, models = _probe(
83 provider, which=_which, run=_run, env=env, opener=_opener
84 )
85 except Exception as exc:
86 # One provider must never take the report down. Everything below is
87 # already fail-soft, so reaching here means an injected seam or a stdlib
88 # call surprised us — which is a row that says so, not a traceback out of
89 # `keel doctor`.
90 available, reason, models = False, f"probe failed: {exc}", ()
91 results.append(ProbeResult(provider, available, reason, models))
92 return tuple(results)
95def _probe(
96 provider: Provider,
97 *,
98 which: Callable[[str], str | None],
99 run: Callable[..., runner.CommandResult],
100 env: Mapping[str, str],
101 opener,
102) -> tuple[bool, str, tuple[str, ...]]:
103 if provider.transport == "api":
104 return _probe_api(provider, env=env)
105 if provider.transport == "local":
106 return _probe_local(provider, which=which, opener=opener)
107 return _probe_cli(provider, which=which, run=run)
110#: A Windows drive prefix (``C:``). ``C:evil`` is *drive-relative* — ``os.path.isabs`` is
111#: False and it holds no separator, yet ``CreateProcess`` resolves it against the current
112#: directory on that drive, i.e. the checkout when that is cwd.
113_DRIVE_PREFIX = re.compile(r"^[A-Za-z]:")
116def _repo_relative(command: str | None) -> bool:
117 """Does ``command`` name a path that could resolve inside the working tree? (#1247)
119 A probe runs ``<command> --version`` (and ``<command> models``) to report whether a
120 coding-agent CLI is reachable, and for a profile or registry seat that command is read
121 from the project's own ``.keel/project.yaml``. When keel inspects a project it did not
122 write — a clone, a fork's pull-request branch — a relative path there resolves against the
123 checkout, so probing it would execute a program the untrusted project ships. Only a bare
124 name (resolved through ``PATH`` alone) and an absolute path (an explicit operator choice an
125 attacker cannot predict) are safe to probe; a separator or a Windows drive prefix that is
126 not absolute is refused.
127 """
128 if not command or os.path.isabs(command):
129 return False
130 return "/" in command or "\\" in command or bool(_DRIVE_PREFIX.match(command))
133def _probe_cli(
134 provider: Provider,
135 *,
136 which: Callable[[str], str | None],
137 run: Callable[..., runner.CommandResult],
138) -> tuple[bool, str, tuple[str, ...]]:
139 """A coding-agent CLI: on ``PATH``, and answering ``--version``.
141 Both halves matter. A binary that is present but broken (a half-finished install,
142 a wrapper whose runtime is gone) would otherwise be reported as a usable
143 implementer and fail at s4, which is the expensive place to find out.
144 """
145 command = provider.command
146 if _repo_relative(command):
147 # Refuse before any execution: a relative path resolves against the inspected
148 # checkout, and a probe is a read-only readiness check (#1247).
149 return (
150 False,
151 f"{command} names a path inside the project; not probed — configure a PATH command",
152 (),
153 )
154 found, reason, _ = _probe_command_only(command, which=which)
155 if not found:
156 return False, reason, ()
157 result = run([command, "--version"], timeout=PROBE_TIMEOUT_S)
158 if getattr(result, "timed_out", False):
159 return False, f"{command} --version timed out after {PROBE_TIMEOUT_S}s", ()
160 if not result.ok:
161 return False, f"{command} --version failed (exit {result.code})", ()
162 version = _first_line(result.stdout or result.output)
163 if version:
164 reason = f"{reason} ({version})"
165 return True, reason, _cli_models(provider, command, run)
168def _cli_models(
169 provider: Provider,
170 command: str,
171 run: Callable[..., runner.CommandResult],
172) -> tuple[str, ...]:
173 """Models a CLI lists for itself. Only ``agy`` exposes one keel can read."""
174 if not (provider.source == "builtin" and provider.name == "agy"):
175 return ()
176 result = run([command, "models"], timeout=PROBE_TIMEOUT_S)
177 if not result.ok:
178 return ()
179 return providers.parse_model_lines(result.stdout or result.output)
182def _probe_local(
183 provider: Provider,
184 *,
185 which: Callable[[str], str | None],
186 opener,
187) -> tuple[bool, str, tuple[str, ...]]:
188 """A model served on this machine.
190 The built-in ``ollama`` vendor is both a CLI and a server, and keel needs both:
191 the binary to dispatch through, the server to hold the model. A registry ``local``
192 entry is probed by its command alone — keel does not dial an address a file names.
193 """
194 if provider.source != "builtin":
195 return _probe_command_only(provider.command, which=which)
196 ok, models, error = _ollama_tags(opener)
197 if not ok:
198 return False, f"{providers.OLLAMA_TAGS_URL} unreachable: {error}", ()
199 found, reason, _ = _probe_command_only(provider.command, which=which)
200 if not found:
201 return False, f"{reason} (the server at {providers.OLLAMA_TAGS_URL} answers)", ()
202 return True, f"{reason}; {len(models)} model(s) served locally", models
205def _probe_command_only(
206 command: str | None,
207 *,
208 which: Callable[[str], str | None],
209) -> tuple[bool, str, tuple[str, ...]]:
210 """Is ``command`` on ``PATH``? The half every transport that runs a binary shares."""
211 if not command:
212 return False, "no command configured", ()
213 path = which(command)
214 if not path:
215 return False, f"{command} not found on PATH", ()
216 return True, path, ()
219def _probe_api(
220 provider: Provider,
221 *,
222 env: Mapping[str, str],
223) -> tuple[bool, str, tuple[str, ...]]:
224 """A hosted endpoint: is its key present? Names only — the value is never read.
226 No request is made. Whether the key *works* is a question only the vendor can
227 answer, and asking costs a billable call plus a round trip on every ``doctor``
228 run; whether one is configured is the fact an operator is missing.
229 """
230 key_env = provider.api_key_env
231 if not key_env:
232 return False, "no api_key_env configured", ()
233 if provider.source == "builtin":
234 present = api_delegate.has_api_token(provider.vendor, _env=env)
235 else:
236 present = bool(env.get(key_env, "").strip())
237 where = f" for {provider.endpoint}" if provider.endpoint else ""
238 if not present:
239 return False, f"{key_env} is not set in the environment{where}", ()
240 return True, f"{key_env} is set{where}", ()
243def _ollama_tags(opener) -> tuple[bool, tuple[str, ...], str]:
244 """GET the local tag listing. Returns ``(ok, models, error)``; never raises."""
245 client = opener if opener is not None else api_delegate.build_http_only_opener()
246 request = urllib.request.Request(providers.OLLAMA_TAGS_URL, method="GET") # nosec B310
247 try:
248 with client.open(request, timeout=HTTP_TIMEOUT_S) as response:
249 raw = response.read(5 * 1024 * 1024).decode("utf-8", errors="replace")
250 except Exception as exc:
251 # Deliberately broad: urllib raises URLError, http.client exceptions (not
252 # OSError subclasses), socket timeouts and the address guard's own error.
253 # A doctor run reports them all the same way — as "not reachable".
254 return False, (), str(exc)
255 try:
256 data = json.loads(raw)
257 except ValueError:
258 return False, (), "response is not valid JSON"
259 return True, providers.parse_tag_payload(data), ""
262def _first_line(text: str) -> str:
263 for line in (text or "").splitlines():
264 if line.strip():
265 return line.strip()
266 return ""
269def build_report(
270 results: Iterable[ProbeResult],
271 *,
272 registry: Registry,
273 errors: Iterable[str] = (),
274) -> dict[str, object]:
275 """Assemble the JSON-stable provider document ``keel doctor --providers`` prints."""
276 rows = [result.as_dict() for result in results]
277 return {
278 "schema_version": SCHEMA_VERSION,
279 "providers": rows,
280 "registry_path": registry.path,
281 "registry_present": registry.present,
282 "warnings": list(registry.warnings),
283 "errors": list(errors),
284 "available": sum(1 for row in rows if row["available"]),
285 "total": len(rows),
286 }
289#: Wall-clock seconds ``jury --doctor`` may take. Longer than :data:`PROBE_TIMEOUT_S`
290#: because the runner probes its *own* panel behind that one call — one ``--version`` per
291#: configured agent — where keel's per-provider probes each get their own budget.
292JURY_DOCTOR_TIMEOUT_S = 30
294#: ai-jury's readiness document announces itself with this. Checked rather than assumed, so
295#: some other ``jury`` on ``PATH`` printing JSON cannot be read as a panel report.
296JURY_DOCTOR_SCHEMA_PREFIX = "ai-jury.doctor."
299def probe_jury_runner(
300 *,
301 _which: Callable[[str], str | None] = shutil.which,
302 _run: Callable[..., runner.CommandResult] = runner.run_argv,
303) -> juryavail.Runner:
304 """Is the ``jury`` binary s7 dispatches usable here, and what panel does it hold?
306 s7 does not convene a panel out of keel's delegate inventory — it runs ``jury``
307 (``src/keel/adapters/commands/ship.md``), which carries its own configured panel. A
308 probe that only counted keel's providers answered a different question than the one it
309 was asked: on a host with ``claude`` and ``codex`` installed and no ``jury``, it
310 reported the panel available, published the panel bench, and left s7 to fail at the
311 invocation instead of taking the project's configured fallback or block path.
313 So ask the runner. ``jury --doctor --json`` is ai-jury's own readiness document
314 (``schema_version: ai-jury.doctor.v1``): it establishes that the binary is present and
315 runnable, and reports which of *its* agents are usable — the panel that would actually
316 sit. Fail-soft in the same way every other probe here is: a missing binary, a timeout,
317 a crash and unreadable output are each a :class:`keel.juryavail.Runner` saying so, never
318 an exception out of a resolver.
320 **The document is how the binary identifies itself, so no document is not usable**
321 (#1068). This branch previously read a clean exit with unreadable output as a usable
322 runner, on the reasoning that an ai-jury too old for ``--doctor --json`` still convenes
323 panels and should not be called broken. That reasoning does not survive contact with
324 the older runner: ai-jury parses its arguments strictly, so a version that predates the
325 flag exits **2** on ``--doctor --json`` (``unrecognized arguments: --json``) and is
326 already reported unusable by the exit-code branch below. What the fallback actually
327 granted was the one case it was not written for — a binary that accepts the flag, exits
328 0, and then declines to say it is ai-jury. That is a stranger named ``jury`` on
329 ``PATH``, and calling it usable reintroduced exactly the failure this probe exists to
330 prevent: the panel bench published, s7 failing at the invocation, past the point where
331 ``on_unavailable`` could still choose fallback or block.
333 No second identity probe is taken. ``jury --version`` prints ``%(prog)s <version>`` —
334 the program name it was invoked as — so it identifies the file, not the tool, and a
335 check built on it would pass for any binary of that name. :func:`_doctor_report`'s
336 ``schema_version`` prefix is the identity claim ai-jury actually makes, and one probe
337 that asks for it is better than two that guess.
339 Identity and *inventory* stay separate, and the fallback that matters is the second
340 one: a runner that identified itself but reported no ``agents`` is usable, and
341 :func:`keel.juryavail.assess` then reads keel's own delegate inventory
342 (:func:`collect`) for the vendor count. So a quiet-but-real ai-jury can still staff a
343 panel through keel's proxy; what can no longer staff one is a binary that never said
344 what it was.
345 """
346 command = juryavail.JURY_RUNNER_COMMAND
347 found, reason, _ = _probe_command_only(command, which=_which)
348 if not found:
349 return juryavail.Runner(False, reason)
350 return _probe_jury_doctor(command, reason, run=_run)
353def _probe_jury_doctor(
354 command: str,
355 path: str,
356 *,
357 run: Callable[..., runner.CommandResult],
358) -> juryavail.Runner:
359 """Run ``jury --doctor --json`` and read what came back. Never raises.
361 Through :func:`keel.runner.run_argv` like every other probe in this module, which
362 closes the child's standard input: a readiness check that stopped on a login prompt
363 would hang the resolver it is called from.
365 Usable **only** with a readable ai-jury doctor document; see
366 :func:`probe_jury_runner` for why a clean exit is not on its own an identity.
367 """
368 result = run([command, "--doctor", "--json"], timeout=JURY_DOCTOR_TIMEOUT_S)
369 if getattr(result, "timed_out", False):
370 return juryavail.Runner(
371 False, f"{command} --doctor timed out after {JURY_DOCTOR_TIMEOUT_S}s"
372 )
373 doctor = _doctor_report(result.stdout or result.output)
374 if doctor is not None:
375 version = doctor.get("tool_version")
376 version = version if isinstance(version, str) and version.strip() else "unknown version"
377 return juryavail.Runner(True, f"{path} (ai-jury {version})", doctor)
378 if not result.ok:
379 return juryavail.Runner(False, f"{command} --doctor --json failed (exit {result.code})")
380 return juryavail.Runner(
381 False,
382 f"{path} exited 0 but printed no {JURY_DOCTOR_SCHEMA_PREFIX}* report, so it did "
383 f"not identify itself as the ai-jury panel runner",
384 )
387def _doctor_report(text: str) -> dict[str, object] | None:
388 """ai-jury's ``--doctor --json`` document, or ``None`` when the output is not one.
390 ``raw_decode`` rather than ``json.loads`` for the same reason
391 :func:`keel.jury.parse_report` uses it: :func:`keel.runner.run_argv` hands back stdout
392 and stderr concatenated, so a real report can be followed by log lines.
393 """
394 try:
395 data, _end = json.JSONDecoder().raw_decode((text or "").lstrip())
396 except ValueError:
397 return None
398 if not isinstance(data, dict):
399 return None
400 schema = data.get("schema_version")
401 if isinstance(schema, str) and schema.startswith(JURY_DOCTOR_SCHEMA_PREFIX):
402 return data
403 return None
406def jury_availability(
407 config,
408 *,
409 tier: int | None,
410 difficulty: str | None = None,
411 profile: str | None = None,
412 any_difficulty: bool = False,
413 _probe=None,
414 _runner_probe=None,
415) -> dict[str, object] | None:
416 """Can this machine convene the panel this run's review policy names? (#1066)
418 ``None`` when the question does not arise — this run's review is a host bench, so
419 nothing about the panel can change the answer and nothing is spent asking. That keeps
420 the whole feature inert for every project that has not made the panel its review,
421 keel's own ``projects/keel.yaml`` included.
423 *This run's* review, not the tier's: ``team.by_difficulty.<band>.review`` and
424 ``team.profiles.<name>.review`` may each name the panel, and the resolver applies them
425 over the tier's policy. So the predicate is :func:`keel.team.panel_review_source` — the
426 same overlay :func:`keel.team._review_seats` resolves the bench with, asked once and in
427 one place. Read from ``review.by_tier`` alone it was narrower than the resolver it
428 guards: ``keel ship --team-profile strict`` on an unstaffable host published
429 ``review_panel: jury`` with ``availability: null`` and was stuck exactly as the issue
430 describes, and the call-site sweep could not see it because that site *was* handed a
431 measurement — a ``None`` one.
433 ``difficulty`` and ``profile`` are the run's own coordinates, the ones handed to
434 :func:`keel.team.resolve_assignment`. ``any_difficulty`` is for a caller that cannot
435 name the band yet; see :func:`jury_availability_for_any_tier`.
437 On a panel tier it asks the panel runner first (:func:`probe_jury_runner`), because the
438 runner is what s7 dispatches and it holds the configured panel. Its own report is the
439 inventory when it produced one; otherwise keel falls back to :func:`collect` — the
440 machinery ``keel doctor --providers`` already prints, reused rather than
441 re-implemented — and :func:`keel.juryavail.assess` reads whichever answered. Both probes
442 are local: ``PATH`` lookups and ``--version``-shaped calls, an env-var *name* check per
443 hosted API, and one loopback request for Ollama. No key value is read, and no address a
444 config names is dialled.
446 **This is the one machine-dependent input to the reviewer bench, and it is
447 deliberate.** Every other input is config; this one is a fact about the world, which is
448 why it is allowed to move the outcome — #1014 round 3 closed the *flag* route, not this
449 one — and why :meth:`keel.juryavail.Availability.as_dict` travels with it into the
450 assignment, the review contract, the run ledger and the closure comment. Two machines
451 can resolve the same tier differently: a runner with no ``jury`` installed falls back
452 where a workstation convenes the panel, and each says which it did rather than either
453 quietly claiming the other's provenance. What a *verification* surface does with that
454 is :func:`keel.juryavail.pin`'s question, not this one's: it pins to what the ship
455 measured rather than re-measuring on a different machine.
457 **This function measures; it does not refuse** (#1068). Under ``on_unavailable:
458 block`` the ``block`` decision travels in the record, and
459 :func:`keel.team._review_seats` raises :class:`keel.juryavail.JuryUnavailableError` on
460 the cluster or command whose review really *is* the panel. Refusing here was narrower
461 than it looked: one measurement staffs many benches, and
462 :func:`jury_availability_for_any_tier` takes it before any of them is known — so a
463 ``block`` project could not plan a swarm of entirely non-panel work on an unstaffable
464 host, the panel never entering into it. Nothing is lost by deferring the refusal:
465 ``_review_seats`` is reached from exactly one function,
466 :func:`keel.team.resolve_assignment`, which is every place a bench is resolved.
467 """
468 if (
469 team.panel_review_source(
470 config.knobs.team,
471 tier=tier,
472 difficulty=difficulty,
473 profile=profile,
474 any_difficulty=any_difficulty,
475 )
476 is None
477 ):
478 return None
479 jury_runner = (probe_jury_runner if _runner_probe is None else _runner_probe)()
480 probe = collect if _probe is None else _probe
481 # Only when the runner could not name its own panel: two sweeps of the same agent CLIs
482 # is twice the subprocess cost for a second opinion keel would then have to reconcile.
483 report = None if jury_runner.panel_rows is not None else probe(config)
484 return juryavail.assess(
485 report,
486 runner=jury_runner,
487 min_vendors=config.knobs.team.jury_min_vendors or team.DEFAULT_MIN_VENDORS,
488 policy=config.knobs.team.jury_on_unavailable,
489 ).as_dict()
492def jury_availability_for_any_tier(
493 config, *, profile: str | None = None, **kwargs
494) -> dict[str, object] | None:
495 """The panel probe for a surface that resolves *several* tiers in one call (#1066).
497 :func:`keel.swarm.build_swarm_plan` scores each cluster's risk tier while it partitions,
498 so its caller cannot name the tier before the plan exists — but every cluster in that
499 plan resolves a bench, and a swarm that skipped the probe published ``review_panel:
500 jury`` for a tier-3 cluster while the child ``keel ship`` it launched on the same machine
501 seated three host reviewers. That is the in-process disagreement #1066 exists to close,
502 one layer up.
504 Availability is a fact about the *machine*, not about a tier: the tier only decides
505 whether the question arises. So this asks it once, for the first tier whose review policy
506 names the panel, and hands the one record to every cluster. A project with no panel at
507 any tier probes nothing, exactly as before — :func:`jury_availability` returns ``None``
508 without measuring when the run's review is a host bench, so the loop below costs a
509 config lookup per tier and no subprocess at all.
511 The difficulty band is unknowable here for the same reason the tier is — the scorer runs
512 inside the partition this record is being measured *for* — so the sweep widens the same
513 way: ``any_difficulty`` asks whether any band this policy configures could make the panel
514 the review. ``profile`` is the operator's ``--team``, which *is* known, and is passed so
515 the profile's own precedence over the band holds here exactly as it does in the resolver.
517 Being a superset is exactly why the ``block`` refusal does not live on this path
518 (#1068). "Some tier or band of this project could name the panel" is not "this swarm's
519 work does": a project with ``by_tier.3: jury`` and ``on_unavailable: block`` may
520 legitimately plan a wave of tier-1 docs clusters on a host with no panel, and a refusal
521 taken here — before the partition has scored a single cluster — refused it. The record
522 carries ``decision: block`` to every cluster instead, and
523 :func:`keel.team._review_seats` refuses on the ones whose review really is the panel.
524 """
525 for tier in (None, *(int(name) for name in team.TIERS)):
526 record = jury_availability(
527 config, tier=tier, profile=profile, any_difficulty=True, **kwargs
528 )
529 if record is not None:
530 return record
531 return None
534def collect(
535 config,
536 *,
537 registry_path: str | None = None,
538 _which: Callable[[str], str | None] = shutil.which,
539 _run: Callable[..., runner.CommandResult] = runner.run_argv,
540 _env: Mapping[str, str] | None = None,
541 _opener=None,
542 _read=None,
543) -> dict[str, object]:
544 """Load the registry, plan, probe, and return the report. The one call the CLI makes."""
545 env = os.environ if _env is None else _env
546 kwargs = {} if _read is None else {"_read": _read}
547 registry = providers.load_registry(registry_path, env=env, **kwargs)
548 plan = providers.plan_probes(config, registry)
549 results = probe_providers(plan, _which=_which, _run=_run, _env=env, _opener=_opener)
550 return build_report(
551 results,
552 registry=registry,
553 errors=providers.registry_clashes(registry, config),
554 )