Coverage for src/ai_jury/doctor.py: 99%
366 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Local diagnostics for the agent review jury (``jury --doctor``).
3The ``--doctor`` command reports local readiness and common configuration
4problems. Its output is intentionally SAFE to share:
6- It includes tool/Python/OS versions, a redacted config summary, agent
7 availability (which agent CLIs are on PATH), each agent's detected CLI
8 version and capability summary, and detected config warnings.
9- It NEVER includes the raw diff under review or any agent output.
10- Secret-like values in the config summary are redacted via
11 :func:`ai_jury.redaction.redact`.
13This project collects and transmits NO telemetry. Diagnostics are built
14locally and only written where you explicitly ask (stdout, or ``--write``).
15"""
17from __future__ import annotations
19import platform
20import shutil
21import sys
22import tomllib
23from pathlib import Path
25from . import __version__
26from .adapters import effort_supported, local_model_listing, make_adapter
27from .config import (
28 ConfigError,
29 agy_opt_in_hint,
30 is_commandless_vendor,
31 load_config,
32 normalise_vendor,
33 spec_adapter,
34 vendor_identity,
35)
36from .metadata import claimed_vendors, resolve_min_vendors, vendor_guard_fails
37from .panel import shortfall
38from .redaction import redact, redact_url_userinfo
39from .scaffold import zero_config_local_seat
41#: Version of the machine-readable export emitted by ``jury --doctor --json``.
42#: Bump this (and ``tests/test_doctor.py``'s schema test) on any breaking change
43#: to the shape produced by :func:`doctor_report_dict`.
44DOCTOR_SCHEMA_VERSION = "ai-jury.doctor.v1"
46#: Default endpoint assumed for a ``vendor = "local"`` agent with none configured.
47_DEFAULT_LOCAL_ENDPOINT = "http://localhost:11434/v1"
50def _redact_value(value):
51 """Redact a single config value if it looks secret-like.
53 ``redact`` operates on text and returns ``(text, count)``; non-string
54 values are returned unchanged.
55 """
56 if isinstance(value, str):
57 return redact(value)[0]
58 return value
61def _detect_capabilities(spec):
62 """Best-effort capability/version probe for one agent spec.
64 Uses the real adapter (NOT the mock) so doctor reports actual installed
65 versions, but guards against any failure: an unavailable CLI just reports
66 ``status="unavailable"`` and a crashing probe degrades to ``unknown_version``.
67 This must stay fast (short subprocess timeout) and never crash doctor.
68 """
69 try:
70 adapter = make_adapter(spec)
71 return adapter.detect_capabilities()
72 except Exception as exc: # noqa: BLE001 - diagnostics must never crash
73 return {
74 "version": None,
75 "supports_headless": None,
76 "supports_model_selection": None,
77 "raw_version_output": "",
78 "status": "unknown_version",
79 "warnings": [f"capability probe raised: {redact(str(exc))[0]}"],
80 }
83def _is_available(spec) -> bool:
84 """Whether an agent is reachable, via its adapter's own check.
86 Uses ``adapter.available()`` rather than ``shutil.which`` so a local/HTTP
87 agent (issue #43), which has no ``command`` and probes its endpoint instead,
88 is reported correctly. Guarded — any failure reads as unavailable.
89 """
90 try:
91 return make_adapter(spec).available()
92 except Exception: # noqa: BLE001 - diagnostics must never crash
93 return False
96def _resolved_command(spec):
97 """Absolute path a CLI agent's command resolves to on PATH (issue #296).
99 Lets an operator verify *which* binary will run (a poisoned PATH could
100 resolve a bare name to a shim). None for a local/HTTP agent (no command) or
101 when nothing is found on PATH.
102 """
103 command = getattr(spec, "command", "") or ""
104 has_endpoint = bool(getattr(spec, "endpoint", None))
105 if not command or is_commandless_vendor(spec_adapter(spec)) or has_endpoint:
106 return None
107 try:
108 return shutil.which(command)
109 except Exception: # noqa: BLE001 - diagnostics must never crash
110 return None
113def _probe_models(spec):
114 """Model ids this agent could be pointed at, or None (issue #662).
116 Delegates to the adapter's own ``list_models`` seam — ``agy models`` for
117 Antigravity, the OpenAI-compatible ``/models`` listing for a local server —
118 and, like every other doctor probe, swallows any failure.
119 """
120 try:
121 models = make_adapter(spec).list_models()
122 except Exception: # noqa: BLE001 - diagnostics must never crash
123 return None
124 if not models:
125 return None
126 return [_redact_value(m) for m in models]
129#: The model `ollama pull` is suggested for when a seat names none (it is the one
130#: `jury init --preset offline` writes).
131_SUGGESTED_LOCAL_MODEL = "qwen2.5-coder:7b"
134def _model_is_listed(model: str, listing: list[str]) -> bool:
135 """Whether a local server lists ``model``. Ollama resolves an untagged name to
136 ``:latest``, so ``qwen2.5-coder`` is served by ``qwen2.5-coder:latest``."""
137 return model in listing or (":" not in model and f"{model}:latest" in listing)
140def _local_model_gap(spec):
141 """What a reachable local seat's server says about its model (issue #849).
143 A local seat was reported ready whenever its server answered, so Ollama with
144 nothing pulled read as a working reviewer and every run through it came back
145 `HTTP 404: model … not found`. Returns:
147 * ``("unusable", why)`` — the server lists **no** model. Ollama, vLLM,
148 LM Studio and llama.cpp all list at least the one they would serve, so an
149 empty list means nothing can answer a review, whatever the seat names.
150 * ``("unlisted", why)`` — models are listed but not the configured one. Ollama
151 answers that with a 404, but a llama.cpp server ignores the name and serves
152 the model it loaded, so this is a warning, never a verdict.
153 * ``None`` — the model is listed, or the listing itself failed. No evidence is
154 not evidence of a fault, so a failed listing changes nothing.
155 """
156 if spec_adapter(spec) != "local":
157 return None
158 endpoint = getattr(spec, "endpoint", None) or _DEFAULT_LOCAL_ENDPOINT
159 try:
160 listing = local_model_listing(endpoint)
161 except Exception: # noqa: BLE001 - diagnostics must never crash
162 return None
163 if listing is None:
164 return None
165 shown = redact_url_userinfo(endpoint)
166 model = getattr(spec, "model", "") or ""
167 if not listing:
168 pull = _redact_value(model) if model else _SUGGESTED_LOCAL_MODEL
169 return (
170 "unusable",
171 f"the local server at '{shown}' lists no models — Ollama, vLLM, LM Studio and "
172 f"llama.cpp all list the one they serve, so this seat has nothing to answer "
173 f"with; pull one first (for Ollama: `ollama pull {pull}`). A proxy that serves "
174 f"models it does not list is unaffected: a run still calls it",
175 )
176 if model and not _model_is_listed(model, listing):
177 served = ", ".join(_redact_value(m) for m in listing[:5])
178 more = ", …" if len(listing) > 5 else ""
179 return (
180 "unlisted",
181 f"model '{_redact_value(model)}' is not among the models the local server at "
182 f"'{shown}' lists ({served}{more}); pull it (for Ollama: `ollama pull "
183 f"{_redact_value(model)}`) or set 'model' to one it serves — a llama.cpp "
184 f"server ignores the name and uses the model it loaded, so there it is harmless",
185 )
186 return None
189def _endpoint_for(spec):
190 """The HTTP endpoint an agent talks to, or None for a CLI agent.
192 A configured ``endpoint`` wins (with any userinfo credentials stripped); a
193 hosted-API vendor reports its adapter's fixed vendor URL, so the export
194 answers "where would this actually go?" for every non-CLI transport.
195 """
196 if getattr(spec, "endpoint", None):
197 return redact_url_userinfo(spec.endpoint)
198 vendor = spec_adapter(spec)
199 if vendor == "local":
200 return _DEFAULT_LOCAL_ENDPOINT
201 try:
202 api_url = getattr(make_adapter(spec), "_api_url", None)
203 return api_url() if callable(api_url) else None
204 except Exception: # noqa: BLE001 - diagnostics must never crash
205 return None
208def _unavailable_transport(spec) -> str:
209 """Which transport failed for an unavailable seat: local, hosted-api or cli.
211 The single classifier behind both unavailability messages — the per-agent
212 ``reason`` in the export and the corresponding line in ``warnings`` (issue
213 #701, review round 3). Two readers asking the same question of the same seat
214 must get the same answer, so they ask it here, once, of the normalised
215 vendor: ``vendor = "XAI-API"`` used to be a hosted API to one reader and a
216 CLI with a missing ``command`` to the other.
217 """
218 # Asked of the adapter (#705): which transport failed is a fact about how
219 # the seat is reached, not about whose model was on the other end.
220 vendor = spec_adapter(spec)
221 if vendor == "local":
222 return "local"
223 if vendor in _HOSTED_API_VENDORS or vendor.endswith("-api"):
224 return "hosted-api"
225 return "cli"
228def _unavailable_reason(spec, capability_warnings) -> str:
229 """Why an agent is not usable, in one line (pure given its inputs).
231 Prefers the adapter's own capability warning (e.g. "ANTHROPIC_API_KEY is not
232 set") so the export cannot drift from what the adapter reports, and falls
233 back to a transport-appropriate message when the probe said nothing.
234 """
235 if capability_warnings:
236 return "; ".join(capability_warnings)
237 transport = _unavailable_transport(spec)
238 if transport == "local":
239 endpoint = redact_url_userinfo(spec.endpoint or _DEFAULT_LOCAL_ENDPOINT)
240 return f"endpoint '{endpoint}' is not reachable"
241 if transport == "hosted-api":
242 return "the hosted API is not reachable"
243 if getattr(spec, "command", ""):
244 return f"command '{_redact_value(spec.command)}' is not on PATH"
245 return "not available"
248def _agent_entry(spec, probe_models: bool = False, local_gaps=None):
249 caps = _detect_capabilities(spec)
250 available = _is_available(spec)
251 # A server that answers but lists no model cannot review (#849): the seat is
252 # reported unavailable, with that as its reason, so `ready to run` stops saying
253 # yes for a panel whose only reviewer has nothing to run. The doctor passes a
254 # dict: an available seat's server is listed once, here, and the result kept for
255 # the warnings. A run's metadata passes nothing, so recording a run never makes
256 # a listing request, and an unavailable seat is never listed.
257 gap = None
258 if available and local_gaps is not None:
259 gap = local_gaps[spec.name] = _local_model_gap(spec)
260 unusable = gap[1] if gap and gap[0] == "unusable" else None
261 if unusable:
262 available = False
263 capability_warnings = [_redact_value(w) for w in caps.get("warnings", [])]
264 return {
265 "name": _redact_value(spec.name),
266 "command": _redact_value(spec.command),
267 "endpoint": _endpoint_for(spec),
268 "resolved": _resolved_command(spec),
269 # Two vendor fields, because they answer two different questions and a
270 # row that showed only one was read as answering both (#701, round 2).
271 # `vendor` is provenance: the vendor the operator named, in the one
272 # normalised spelling every rule reads (round 3 — the spec normalises on
273 # construction, so this is the same string the doctor reasons about).
274 # `vendor_identity` is the gate's view: what this seat counts as under
275 # `min_vendors`, which is `cli` for anything the build cannot identify.
276 "vendor": _redact_value(spec.vendor),
277 "vendor_identity": vendor_identity(getattr(spec, "vendor", "")),
278 # The third field answers the third question (#705): `vendor` is what the
279 # operator called the seat, `vendor_identity` is what the gate counts it
280 # as, and `adapter` is the protocol that builds its command line. A
281 # Codex seat and a GPT-through-Cursor seat differ in nothing else.
282 "adapter": spec_adapter(spec),
283 "available": available,
284 "reason": None
285 if available
286 else (unusable or _unavailable_reason(spec, capability_warnings)),
287 "version": _redact_value(caps.get("version")),
288 "capabilities": {
289 "supports_headless": caps.get("supports_headless"),
290 "supports_model_selection": caps.get("supports_model_selection"),
291 "status": caps.get("status"),
292 },
293 # Only the JSON export renders a model listing, and discovering one
294 # costs a subprocess (`agy models`) or an HTTP round trip per agent. The
295 # text report would pay that for nothing, so the probe is opt-in.
296 "models": _probe_models(spec) if probe_models else None,
297 "effort": _redact_value(getattr(spec, "effort", None)),
298 "effort_supported": effort_supported(spec_adapter(spec)),
299 # Cost tier (issue #714): which seats tiered routing may bench on a
300 # routine diff, and which one it may anchor with.
301 "tier": getattr(spec, "tier", "frontier"),
302 "capability_warnings": capability_warnings,
303 }
306def _config_summary(cfg):
307 """Build a redacted, secret-free summary of the loaded config."""
308 return {
309 "rounds": cfg.rounds,
310 "chair": _redact_value(cfg.chair),
311 "context_mode": _redact_value(cfg.context.mode),
312 "enabled_agents": [_redact_value(a.name) for a in cfg.enabled_agents],
313 }
316# Hosted-API vendors (issue #430/#432): no `command`/`endpoint`, so neither
317# the "local" nor the "CLI on PATH" branch below is the right diagnosis when
318# one is unavailable.
319_HOSTED_API_VENDORS = ("anthropic-api", "openai-api", "google-api", "xai-api")
322def _detect_warnings(cfg, local_gaps=None) -> list[str]:
323 """Best-effort config sanity checks reported to the user."""
324 warnings: list[str] = []
325 if not cfg.agents:
326 warnings.append("no agents are configured")
327 enabled = cfg.enabled_agents
328 if cfg.agents and not enabled:
329 warnings.append("all configured agents are disabled")
330 names = {a.name for a in cfg.agents}
331 if cfg.chair not in names:
332 warnings.append(f"chair '{_redact_value(cfg.chair)}' does not match any configured agent")
333 for agent in enabled:
334 if _is_available(agent):
335 # Both cases are warnings as well as the seat's reason: the text report
336 # prints warnings, not reasons, and `ollama pull <model>` is the line a
337 # user needs to see (#850 third seat).
338 gap = (local_gaps or {}).get(agent.name)
339 if gap:
340 warnings.append(f"agent '{_redact_value(agent.name)}' (local): {gap[1]}")
341 # An available hosted-API seat (no command, no endpoint) with no model is still
342 # reported ready, but the API call fails at request time — `--config-validate`
343 # warns while `--doctor` did not (#831). Flag it here too, in the same words.
344 if not (
345 (getattr(agent, "command", "") or "")
346 or (getattr(agent, "endpoint", "") or "")
347 or (getattr(agent, "model", "") or "")
348 ):
349 warnings.append(
350 f"agent '{_redact_value(agent.name)}' "
351 f"({_redact_value(vendor_identity(getattr(agent, 'vendor', '')))}) "
352 "has no 'model'; the server or API call will likely reject the request"
353 )
354 continue
355 # Classified by the SAME predicate `_unavailable_reason` uses, so the
356 # warning list and the per-agent `reason` cannot diagnose one seat two
357 # different ways (issue #701, review round 3). They differ only in
358 # wording; disagreeing about *which* transport failed was the bug.
359 transport = _unavailable_transport(agent)
360 if transport == "local":
361 warnings.append(
362 f"agent '{_redact_value(agent.name)}' (local) endpoint "
363 f"'{redact_url_userinfo(agent.endpoint or _DEFAULT_LOCAL_ENDPOINT)}' "
364 f"is not reachable"
365 )
366 elif transport == "hosted-api":
367 # Reuse the adapter's own capability warning (issue #430) instead
368 # of re-deriving the vendor -> env-var mapping here, so the
369 # message can't drift from what the adapter actually reports.
370 caps = _detect_capabilities(agent)
371 reason = "; ".join(caps.get("warnings", [])) or "the hosted API is not reachable"
372 warnings.append(f"agent '{_redact_value(agent.name)}' (hosted API): {reason}")
373 else:
374 warnings.append(
375 f"agent '{_redact_value(agent.name)}' command "
376 f"'{_redact_value(agent.command)}' is not on PATH"
377 )
378 return warnings
381def _panel_readiness(cfg, agents, min_vendors=None, local_fallback=None) -> dict:
382 """How close this machine is to being able to form cross-vendor consensus.
384 Doctor is offline and runs no review, so it can only report what it can
385 see: how many distinct vendors are ENABLED, and how many of those are
386 reachable. ``contributing_vendors`` is therefore always ``None`` here — the
387 contributed-vendor count is a property of a run, and lives in the run
388 metadata's ``panel.vendors``. Saying so in the export is the point: an
389 available CLI that returns nothing is exactly the failure #635 was, and a
390 green doctor is not evidence against it.
392 Both counts are of :func:`config.vendor_identity`, the same arithmetic
393 :func:`metadata.distinct_vendors` does at the gate, so
394 ``vendors_configured`` equals what a run would count for the same config
395 (#701, round 2). Counting raw strings here meant two seats on the generic
396 fallback read as two vendors in ``--doctor`` and as one vendor in the run:
397 doctor called the bench cross-vendor ready and the run then refused it.
399 ``min_vendors`` is the run's ``--min-vendors`` / ``--no-min-vendors`` value
400 (``None`` when neither was named), resolved as the run resolves it. With
401 ``local_fallback`` — the seat :func:`scaffold.zero_config_local_seat` says a
402 run with no config would review with — the panel is that one seat, as the
403 run forms it (#863): the built-in seats it replaces are all unreachable, and
404 its server has just listed the model, which is the run's reason to seat it.
405 """
406 entries = {entry["name"]: entry for entry in agents}
407 enabled = list(getattr(cfg, "enabled_agents", []) or [])
408 if local_fallback is not None:
409 enabled = [local_fallback]
410 entries = {local_fallback.name: {"available": True}}
411 available = {
412 vendor_identity(a.vendor) for a in enabled if entries.get(a.name, {}).get("available")
413 } - {""}
414 minimum, explicit = resolve_min_vendors(min_vendors, cfg)
415 # The number a downstream consumer counts, which doctor never reported and
416 # which is not the vendor count (#699): one review per agent that answers.
417 # The chair's synthesis record is not added here — a consumer reads it as the
418 # panel's consensus, not as a review, and adding it is how a bench with
419 # nothing reachable came to advertise one review. Doctor can only see
420 # reachability, so this is the CEILING — an agent that runs and returns
421 # nothing, or names nothing checkable, casts no review (#700) — which is why
422 # it is labelled "at most" below.
423 seats = sum(1 for a in enabled if entries.get(a.name, {}).get("available"))
424 panel = {
425 "vendors_configured": claimed_vendors(enabled),
426 "vendors_available": len(available),
427 "min_vendors": minimum,
428 "contributing_vendors": None,
429 "panelists_available": seats,
430 "reviews_supplied_max": seats,
431 "min_reviews": int(getattr(cfg.ci, "min_reviews", 0) or 0),
432 }
433 # Derived from the same predicate the warning uses, so the field and the
434 # warning cannot disagree about the same machine (#682, round 3).
435 panel["multi_vendor_ready"] = not _gate_would_fail(panel, explicit)
436 return panel
439#: What a machine with no loadable config can prove about its panel: nothing.
440#: ``multi_vendor_ready`` is ``False`` here for a different reason than below —
441#: not "the gate would fail" but "there is no config to satisfy", which is not a
442#: readiness anyone should build on.
443_NO_PANEL = {
444 "vendors_configured": 0,
445 "vendors_available": 0,
446 "min_vendors": 0,
447 "contributing_vendors": None,
448 "panelists_available": 0,
449 "reviews_supplied_max": 0,
450 "min_reviews": 0,
451 "multi_vendor_ready": False,
452}
455def _gate_would_fail(panel, explicit: bool = False) -> bool:
456 """Would a run on this machine fail the cross-vendor gate?
458 The single place doctor decides that, so the ``multi_vendor_ready`` field
459 and the warning below are the same judgement rendered twice. It mirrors
460 :func:`metadata.collapse_reason`'s scoping term for term: the gate is off at
461 ``0``; a config with fewer distinct vendors enabled than the threshold never
462 claimed that consensus and is left alone; otherwise every configured vendor
463 short of the threshold is a vendor the run cannot hear from.
465 Doctor can only see reachability, so this is a *lower* bound on failure: a
466 reachable CLI that returns nothing (#635) fails the gate too, and no offline
467 check can predict it. That is why the report says so in as many words.
469 It is :func:`metadata.vendor_guard_fails`, the run's own predicate, with the
470 reachable vendors standing in for the contributing ones; an ``explicit``
471 threshold (named on the command line) is enforced unscoped, as in the run.
472 """
473 return vendor_guard_fails(
474 panel["vendors_available"],
475 panel["min_vendors"],
476 None if explicit else panel["vendors_configured"],
477 )
480def _fallback_ready_reason(panel, explicit: bool) -> str | None:
481 """Why the zero-config fallback's one seat reads `cross-vendor ready: yes` (pure).
483 One seat is not a cross-vendor panel, so the text report says which of three
484 things makes the gate pass (#863): the guard is off; the default threshold
485 was scoped away because one seat claims no cross-vendor consensus; or a
486 threshold is met. ``None`` when the gate would fail — there is nothing to
487 explain, and the warning says the rest.
488 """
489 if not panel["multi_vendor_ready"]:
490 return None
491 required = panel["min_vendors"]
492 if required <= 0:
493 return "the gate is off"
494 if not explicit and panel["vendors_configured"] < required:
495 return "the gate would not fail: one seat claims no cross-vendor consensus"
496 return (
497 f"the gate would not fail: {panel['vendors_available']} vendor(s) reachable, "
498 f"{required} required"
499 )
502def _panel_warning(panel, explicit: bool = False) -> str | None:
503 """The one actionable thing offline diagnostics can say about the panel.
505 Fires only when a run on this machine would actually fail the gate, because
506 then it exits 3 and the operator would rather know now. A warning that does
507 not predict the run is worse than none — it is the kind people learn to
508 scroll past — which is also why a single-vendor config under the shipped
509 ``min_vendors = 2`` is silent: :func:`metadata.collapse_reason` leaves that
510 run alone, so there is nothing to warn about.
511 """
512 if not _gate_would_fail(panel, explicit):
513 return None
514 asked = (
515 f"--min-vendors {panel['min_vendors']} asks for {panel['min_vendors']} vendors"
516 if explicit
517 else f"{panel['vendors_configured']} vendors are enabled"
518 )
519 return (
520 f"{asked} but only "
521 f"{panel['vendors_available']} is/are reachable; a run would fail the "
522 f"cross-vendor guard (min_vendors = {panel['min_vendors']}, exit 3). "
523 f"Install the missing CLI, or opt out with `--no-min-vendors` "
524 f"(`[jury.ci] min_vendors = 0`)."
525 )
528def _recommendations(
529 config_path,
530 config_summary,
531 agents,
532 config_error: bool = False,
533 local_fallback=None,
534 local_models=None,
535) -> dict:
536 """Build actionable next-steps from the diagnostics (issue: doctor UX).
538 Returns ``{"ready": bool, "steps": [str, ...]}``. ``ready`` is true when at
539 least one agent is reachable. Steps point the user at the cheapest fix:
540 scaffold a config, install a CLI, or use a reachable local model.
542 *config_error* says the config could not be loaded at all, so no seat was
543 inspected — and the only honest next step is the error itself (issue #708).
545 *local_fallback* is the seat a run with no config would review with (#863):
546 that run has a reviewer, so the machine is ready, and the step says what the
547 run will do. *local_models* is a listing already made, reused rather than
548 asked for twice.
549 """
550 steps: list[str] = []
551 available = [a for a in agents if a.get("available")]
552 ready = bool(available)
554 # No config file in play -> suggest scaffolding one.
555 if config_path is None and not Path("jury.toml").exists():
556 steps.append("No jury.toml found — run `jury init` to create one.")
558 # Nothing was loaded, so nothing below is knowable: "install an agent CLI"
559 # would name the wrong fault for a config the RUN also refuses, and it costs
560 # a local-model probe to say it. The verdict is the config error (#708).
561 if config_error:
562 steps.append(
563 "The config could not be loaded, so no agent was inspected — fix the "
564 "config error above. A run refuses this file for the same reason."
565 )
566 return {"ready": False, "steps": steps}
568 if local_fallback is not None:
569 ready = True
570 steps.append(
571 f"No agent CLI is available, so a run with no jury.toml reviews with the "
572 f"local model '{local_fallback.model}' alone: a single-vendor panel, which "
573 f"the default cross-vendor guard does not fail. For a panel you choose: "
574 f"`jury init --preset offline` (or `--list-models`)."
575 )
576 hint = agy_opt_in_hint((a.get("adapter") for a in agents), shutil.which)
577 if hint:
578 steps.append(f"Note: {hint}.")
579 elif not ready:
580 from .adapters import list_local_models
582 models = local_models if local_models is not None else list_local_models()
583 if models:
584 steps.append(
585 f"No agent CLI is available, but a local model server is reachable "
586 f"({len(models)} model(s): {', '.join(models[:3])}). Add a free local "
587 f"reviewer: `jury init --preset offline` (or `--list-models`)."
588 )
589 else:
590 steps.append(
591 "No reviewer is available. Install an agent CLI (claude / codex); "
592 "run a local model (e.g. `ollama serve` + `ollama pull "
593 'qwen2.5-coder:7b`) and add a `vendor = "local"` agent; or, with no '
594 "install at all, set a hosted-API key (e.g. ANTHROPIC_API_KEY) and add "
595 "that seat with `jury init --agents claude-api` — or use `--mock` for an "
596 "offline demo."
597 )
598 # agy on PATH but in no seat: say why the one CLI here was not used.
599 hint = agy_opt_in_hint((a.get("adapter") for a in agents), shutil.which)
600 if hint:
601 steps.append(f"Note: {hint}.")
602 else:
603 missing = [
604 a["name"]
605 for a in agents
606 if not a.get("available")
607 and config_summary
608 and a["name"] in config_summary.get("enabled_agents", [])
609 ]
610 if missing:
611 steps.append(
612 f"Enabled but unavailable (will be skipped): {', '.join(missing)}. "
613 f"Install them or run with `--strict` to fail instead."
614 )
616 return {"ready": ready, "steps": steps}
619def build_diagnostics(config_path=None, probe_models: bool = False, min_vendors=None):
620 """Build a SAFE diagnostics dict for the given config path.
622 Best-effort: if the config cannot be loaded, the error is captured as a
623 string under ``config_warnings`` and ``config`` is left ``None``. Never
624 raises for a bad/missing config. The returned dict never contains the raw
625 diff or any agent output.
627 The config is loaded WITH validation — the same call a run makes (issue
628 #708). Best-effort is about not crashing, not about being more permissive
629 than the run: this report's whole promise is that its arithmetic equals what
630 a run counts, and it cannot keep that promise while accepting a file the run
631 rejects. It used to. An `adapter` name this build does not have is a hard
632 config error, but with validation off the seat kept the name, `make_adapter`
633 missed the registry and fell through to the generic CLI adapter, and three
634 such seats reported `[available]`, `cross-vendor ready: yes` and `ready to
635 run: yes` for a config the run refused before its first round. A hard error
636 is now the report's verdict: it lands in ``config_warnings`` naming the seat
637 and the adapter, no seat is described, and ``ready to run`` is ``no``.
638 Warnings are still warnings — ``strict`` is not passed, so a soft issue
639 (unknown vendor, unknown key) is reported, not fatal, exactly as before.
641 ``probe_models`` opts into discovering each agent's available model ids —
642 a subprocess (``agy models``) or an HTTP round trip *per agent*, each
643 time-boxed but not free. Only ``--doctor --json`` renders that listing, so
644 it defaults off: the human report used to pay ~2 s (and up to the probe
645 timeout if a CLI hangs) for a field it never printed.
647 ``min_vendors`` is the run's ``--min-vendors`` / ``--no-min-vendors`` value,
648 so the cross-vendor prediction is made for the threshold the run would use.
649 With no config file and no reachable seat, the local server is listed once
650 and :func:`scaffold.zero_config_local_seat` — the run's own decision — says
651 whether a run would review with a local model; if so, the panel and the
652 prediction are that run's (#863).
653 """
654 config_summary = None
655 config_warnings: list[str] = []
656 agents: list = []
657 panel: dict = dict(_NO_PANEL)
658 # Distinct from `config_summary is None`, which a caller may also hand
659 # `_recommendations` directly: this says a load was ATTEMPTED and failed.
660 # Cleared only on the success path, so a new `except` arm cannot forget it.
661 config_error = True
662 local_fallback = None
663 ready_reason = None
664 listed: list = []
666 def _list_once():
667 from .adapters import list_local_models
669 listed.append(list_local_models())
670 return listed[-1]
672 try:
673 cfg = load_config(config_path, validate=True)
674 except FileNotFoundError as exc:
675 config_warnings.append(f"config error: {redact(str(exc))[0]}")
676 except OSError as exc:
677 # Present but unreadable — a chmod-0 jury.toml, or --config naming a directory —
678 # is reported like the CLI reports it, not raised through the report (#893).
679 path = exc.filename or config_path or "jury.toml"
680 why = exc.strerror or str(exc)
681 config_warnings.append(redact(f"config error: cannot read config {path}: {why}")[0])
682 except tomllib.TOMLDecodeError as exc:
683 config_warnings.append(f"config error: invalid TOML: {redact(str(exc))[0]}")
684 except ConfigError as exc:
685 config_warnings.append(f"config error: {redact(str(exc))[0]}")
686 except (KeyError, ValueError, TypeError) as exc:
687 config_warnings.append(f"config error: {redact(str(exc))[0]}")
688 else:
689 config_error = False
690 config_summary = _config_summary(cfg)
691 # One listing per available local server, shared by the entry and the warnings.
692 local_gaps: dict = {}
693 agents = [
694 _agent_entry(spec, probe_models=probe_models, local_gaps=local_gaps)
695 for spec in cfg.agents
696 ]
697 config_warnings = _detect_warnings(cfg, local_gaps)
698 # Fold capability/version probe warnings (e.g. an available CLI whose
699 # version could not be detected) into the user-facing warnings list.
700 # Probes already ran while building the agent entries above.
701 enabled_names = {a.name for a in cfg.enabled_agents}
702 for spec, entry in zip(cfg.agents, agents, strict=False):
703 if spec.name not in enabled_names:
704 continue
705 for warning in entry.get("capability_warnings", []):
706 config_warnings.append(f"agent '{entry['name']}': {warning}")
707 # Cross-vendor readiness (issue #682), after the availability probes the
708 # agent entries already ran — this adds no probe of its own.
709 local_fallback = zero_config_local_seat(
710 config_path,
711 False,
712 Path("jury.toml").exists(),
713 lambda: any(e.get("available") for e in agents if e.get("name") in enabled_names),
714 _list_once,
715 )
716 panel = _panel_readiness(cfg, agents, min_vendors, local_fallback)
717 explicit = resolve_min_vendors(min_vendors, cfg)[1]
718 panel_warning = _panel_warning(panel, explicit)
719 if local_fallback is not None:
720 ready_reason = _fallback_ready_reason(panel, explicit)
721 if panel_warning:
722 config_warnings.append(panel_warning)
723 # A bench that cannot reach the consumer's minimum is a shortfall this
724 # machine can prove offline (#699) — worth saying here rather than after
725 # the run, which is where it used to surface.
726 short = shortfall(
727 panel["panelists_available"],
728 panel["min_reviews"],
729 stage="on this machine",
730 )
731 if short:
732 config_warnings.append(short)
734 return {
735 "tool_version": __version__,
736 "python_version": platform.python_version(),
737 "python_implementation": platform.python_implementation(),
738 "python_executable": sys.executable,
739 "os": platform.platform(),
740 "config_path": str(config_path) if config_path else "(default)",
741 "agents": agents,
742 "config": config_summary,
743 "config_warnings": config_warnings,
744 "panel": panel,
745 # The model a run with no config would review with alone, or None (#863).
746 "local_fallback": local_fallback.model if local_fallback is not None else None,
747 # Why that one seat reads as cross-vendor ready, for the text report.
748 "cross_vendor_ready_reason": ready_reason,
749 "recommendations": _recommendations(
750 config_path,
751 config_summary,
752 agents,
753 config_error=config_error,
754 local_fallback=local_fallback,
755 local_models=listed[-1] if listed else None,
756 ),
757 }
760# Capability flags -> the short labels shown in both renderers, in a fixed order.
761_CAPABILITY_LABELS = (
762 ("supports_headless", "headless"),
763 ("supports_model_selection", "model-selection"),
764)
767def capability_labels(capabilities) -> list[str]:
768 """Short capability labels for one agent's capability dict (pure)."""
769 caps = capabilities or {}
770 return [label for key, label in _CAPABILITY_LABELS if caps.get(key)]
773def _transport(vendor: str, command: str, endpoint) -> str:
774 """Classify how an agent is reached: ``cli``, ``api`` or ``local`` (pure)."""
775 name = normalise_vendor(vendor)
776 if name == "local":
777 return "local"
778 if name.endswith("-api") or name == "openai-compatible":
779 return "api"
780 if command:
781 return "cli"
782 return "api" if endpoint else "cli"
785def doctor_report_dict(diagnostics) -> dict:
786 """Project a diagnostics dict onto the stable ``ai-jury.doctor.v1`` export.
788 PURE: it runs no probes and touches no network, PATH or filesystem — every
789 fact comes from the dict :func:`build_diagnostics` already built, so the
790 probes run exactly once no matter how many renderers consume them.
792 Both renderers consume this: ``jury --doctor --json`` serializes it, and
793 :func:`render_report` renders its agent rows from it, so the human report
794 and the machine export cannot describe the panel differently.
796 Secrets are never included — only environment *variable names*, which reach
797 this dict through the adapters' own capability warnings.
798 """
799 agents = []
800 for entry in diagnostics.get("agents", []):
801 vendor = entry.get("vendor") or ""
802 adapter = entry.get("adapter") or ""
803 command = entry.get("command") or ""
804 endpoint = entry.get("endpoint")
805 # The transport follows the adapter (#705): `vendor = "openai",
806 # adapter = "cli"` is reached over a CLI, not over OpenAI's API.
807 transport = _transport(adapter or vendor, command, endpoint)
808 item = {
809 "name": entry.get("name"),
810 "vendor": vendor,
811 "vendor_identity": entry.get("vendor_identity") or "",
812 "adapter": adapter,
813 "transport": transport,
814 "available": bool(entry.get("available")),
815 "reason": entry.get("reason"),
816 }
817 # One address key per agent, named for the transport that reaches it.
818 if transport == "cli":
819 item["command"] = command
820 else:
821 item["endpoint"] = endpoint
822 models = entry.get("models")
823 item["resolved"] = entry.get("resolved")
824 item["version"] = entry.get("version")
825 item["capabilities"] = capability_labels(entry.get("capabilities"))
826 item["models"] = list(models) if models else None
827 item["effort_supported"] = bool(entry.get("effort_supported"))
828 item["effort"] = entry.get("effort")
829 # Cost tier (#714): what tiered routing reads; `frontier` unless said.
830 item["tier"] = entry.get("tier") or "frontier"
831 agents.append(item)
833 recommendations = diagnostics.get("recommendations") or {}
834 return {
835 "schema_version": DOCTOR_SCHEMA_VERSION,
836 "tool_version": diagnostics.get("tool_version"),
837 "python": diagnostics.get("python_version"),
838 "config_path": diagnostics.get("config_path"),
839 "ready": bool(recommendations.get("ready")),
840 # Cross-vendor readiness (issue #682). ``contributing_vendors`` is null
841 # by construction: doctor runs no review, so the contributed count is
842 # only ever known from a run's metadata (``panel.vendors``). A consumer
843 # reading this must not treat availability as contribution.
844 "panel": dict(diagnostics.get("panel") or _NO_PANEL),
845 "agents": agents,
846 "warnings": list(diagnostics.get("config_warnings") or []),
847 }
850def render_report(diagnostics) -> str:
851 """Render a human-readable text report from a diagnostics dict.
853 The Agents section and the readiness line are rendered from
854 :func:`doctor_report_dict`, the same projection ``--json`` serializes, so
855 the two views cannot drift; the capability *probe status* is read alongside
856 it from the raw diagnostics (it is a diagnostic detail, not part of the
857 exported schema).
858 """
859 report = doctor_report_dict(diagnostics)
860 lines = []
861 lines.append("jury doctor")
862 lines.append("=" * 40)
863 lines.append(f"tool version: {diagnostics['tool_version']}")
864 lines.append(
865 f"python: {diagnostics['python_version']} ({diagnostics['python_implementation']})"
866 )
867 lines.append(f"python exe: {diagnostics['python_executable']}")
868 lines.append(f"os: {diagnostics['os']}")
869 lines.append(f"config path: {diagnostics['config_path']}")
870 lines.append("")
872 lines.append("Agents")
873 lines.append("-" * 40)
874 agents = report["agents"]
875 if not agents:
876 lines.append(" (no agents loaded)")
877 else:
878 for agent, probe in zip(agents, diagnostics["agents"], strict=False):
879 status = "available" if agent["available"] else "MISSING"
880 # A CLI agent is identified by its command; an api/local agent has
881 # none, so name the endpoint it actually talks to rather than
882 # printing an empty `command=`.
883 if agent["transport"] == "cli":
884 address = f"command={agent.get('command', '')}"
885 else:
886 address = f"endpoint={agent.get('endpoint') or '(unknown)'}"
887 # A seat whose configured vendor is not the identity it carries
888 # says so on its own row, so the row and the panel count below can
889 # be read together without arithmetic (#701, round 2).
890 identity = agent.get("vendor_identity") or ""
891 vendor_field = agent["vendor"]
892 if identity and identity != normalise_vendor(vendor_field):
893 vendor_field = f"{vendor_field} -> counts as {identity}"
894 # Vendor AND adapter on every row (#705), never one standing in for
895 # both: a reader must be able to tell a Codex seat (openai/openai)
896 # from a GPT-through-Cursor seat (openai/cli) at a glance, and the
897 # `--doctor` row is where that difference becomes visible.
898 lines.append(
899 f" [{status:>9}] {agent['name']} (vendor={vendor_field}, "
900 f"adapter={agent.get('adapter') or '(default)'}, {address})"
901 )
902 if agent.get("command"):
903 resolved = agent.get("resolved") or "(not found on PATH)"
904 lines.append(f" resolved: {resolved}")
905 version = agent.get("version") or "unknown"
906 cap_summary = ", ".join(agent["capabilities"]) or "none"
907 cap_status = (probe.get("capabilities") or {}).get("status") or "unknown"
908 summary = (
909 f" version={version}, capabilities=[{cap_summary}] "
910 f"(probe: {cap_status})"
911 )
912 if agent.get("effort"):
913 summary += f", effort={agent['effort']}"
914 lines.append(summary)
915 lines.append("")
917 lines.append("Config summary")
918 lines.append("-" * 40)
919 config = diagnostics["config"]
920 if config is None:
921 lines.append(" (config could not be loaded)")
922 else:
923 lines.append(f" rounds: {config['rounds']}")
924 lines.append(f" chair: {config['chair']}")
925 lines.append(f" context mode: {config['context_mode']}")
926 enabled = ", ".join(config["enabled_agents"]) or "(none)"
927 lines.append(f" enabled: {enabled}")
928 lines.append("")
930 panel = report["panel"]
931 lines.append("Cross-vendor readiness")
932 lines.append("-" * 40)
933 if diagnostics.get("local_fallback"):
934 lines.append(
935 f" panel of a run: the local model '{diagnostics['local_fallback']}' alone "
936 f"(no jury.toml and no agent CLI: the zero-config fallback)"
937 )
938 lines.append(f" vendors enabled: {panel['vendors_configured']} (by vendor identity)")
939 lines.append(f" vendors reachable: {panel['vendors_available']} (by vendor identity)")
940 lines.append(f" min_vendors gate: {panel['min_vendors'] or 'off'}")
941 ready_text = "yes" if panel["multi_vendor_ready"] else "no"
942 if diagnostics.get("cross_vendor_ready_reason"):
943 # One seat is not a cross-vendor panel: "ready" means the gate would not
944 # fail it, and the reason says why (#863).
945 ready_text += f" ({diagnostics['cross_vendor_ready_reason']})"
946 lines.append(f" cross-vendor ready: {ready_text}")
947 # The number a consumer counts, said in the same breath as readiness (#699).
948 # "cross-vendor ready: yes" on a bench that cannot supply the reviews a gate
949 # requires is a true statement that answers the wrong question.
950 lines.append(
951 f" reviews for a consumer: at most {panel['reviews_supplied_max']} "
952 f"({panel['panelists_available']} panel ballot(s); the chairing agent "
953 f"reviews too, so its ballot is one of them, and the chair's synthesis "
954 f"record is not a review)"
955 )
956 lines.append(f" min_reviews gate: {panel['min_reviews'] or 'off'}")
957 lines.append(
958 " note: this checks availability, not contribution. A reachable CLI "
959 "can still return no review (#635) — only a run can prove the panel."
960 )
961 lines.append(
962 " note: counted by vendor identity, the same arithmetic min_vendors "
963 "uses — a vendor this build does not recognise counts as 'cli', so two "
964 "such seats are one vendor here and in the run. The Agents rows above "
965 "show each seat's configured vendor string."
966 )
967 lines.append("")
969 lines.append("Warnings")
970 lines.append("-" * 40)
971 warnings = diagnostics["config_warnings"]
972 if not warnings:
973 lines.append(" (none)")
974 else:
975 for warning in warnings:
976 lines.append(f" - {warning}")
977 lines.append("")
979 rec = diagnostics.get("recommendations") or {}
980 lines.append("Next steps")
981 lines.append("-" * 40)
982 lines.append(f" ready to run: {'yes' if report['ready'] else 'no'}")
983 for step in rec.get("steps", []):
984 lines.append(f" - {step}")
985 lines.append("")
987 lines.append(
988 "Privacy: no telemetry is collected or sent. This report is "
989 "local-only and redacts secret-like values."
990 )
992 return "\n".join(lines)