Coverage for src/ai_jury/doctor.py: 99%

366 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-30 06:29 +0000

1"""Local diagnostics for the agent review jury (``jury --doctor``). 

2 

3The ``--doctor`` command reports local readiness and common configuration 

4problems. Its output is intentionally SAFE to share: 

5 

6- It includes tool/Python/OS versions, a redacted config summary, agent 

7 availability (which agent CLIs are on PATH), each agent's detected CLI 

8 version and capability summary, and detected config warnings. 

9- It NEVER includes the raw diff under review or any agent output. 

10- Secret-like values in the config summary are redacted via 

11 :func:`ai_jury.redaction.redact`. 

12 

13This project collects and transmits NO telemetry. Diagnostics are built 

14locally and only written where you explicitly ask (stdout, or ``--write``). 

15""" 

16 

17from __future__ import annotations 

18 

19import platform 

20import shutil 

21import sys 

22import tomllib 

23from pathlib import Path 

24 

25from . import __version__ 

26from .adapters import effort_supported, local_model_listing, make_adapter 

27from .config import ( 

28 ConfigError, 

29 agy_opt_in_hint, 

30 is_commandless_vendor, 

31 load_config, 

32 normalise_vendor, 

33 spec_adapter, 

34 vendor_identity, 

35) 

36from .metadata import claimed_vendors, resolve_min_vendors, vendor_guard_fails 

37from .panel import shortfall 

38from .redaction import redact, redact_url_userinfo 

39from .scaffold import zero_config_local_seat 

40 

41#: Version of the machine-readable export emitted by ``jury --doctor --json``. 

42#: Bump this (and ``tests/test_doctor.py``'s schema test) on any breaking change 

43#: to the shape produced by :func:`doctor_report_dict`. 

44DOCTOR_SCHEMA_VERSION = "ai-jury.doctor.v1" 

45 

46#: Default endpoint assumed for a ``vendor = "local"`` agent with none configured. 

47_DEFAULT_LOCAL_ENDPOINT = "http://localhost:11434/v1" 

48 

49 

50def _redact_value(value): 

51 """Redact a single config value if it looks secret-like. 

52 

53 ``redact`` operates on text and returns ``(text, count)``; non-string 

54 values are returned unchanged. 

55 """ 

56 if isinstance(value, str): 

57 return redact(value)[0] 

58 return value 

59 

60 

61def _detect_capabilities(spec): 

62 """Best-effort capability/version probe for one agent spec. 

63 

64 Uses the real adapter (NOT the mock) so doctor reports actual installed 

65 versions, but guards against any failure: an unavailable CLI just reports 

66 ``status="unavailable"`` and a crashing probe degrades to ``unknown_version``. 

67 This must stay fast (short subprocess timeout) and never crash doctor. 

68 """ 

69 try: 

70 adapter = make_adapter(spec) 

71 return adapter.detect_capabilities() 

72 except Exception as exc: # noqa: BLE001 - diagnostics must never crash 

73 return { 

74 "version": None, 

75 "supports_headless": None, 

76 "supports_model_selection": None, 

77 "raw_version_output": "", 

78 "status": "unknown_version", 

79 "warnings": [f"capability probe raised: {redact(str(exc))[0]}"], 

80 } 

81 

82 

83def _is_available(spec) -> bool: 

84 """Whether an agent is reachable, via its adapter's own check. 

85 

86 Uses ``adapter.available()`` rather than ``shutil.which`` so a local/HTTP 

87 agent (issue #43), which has no ``command`` and probes its endpoint instead, 

88 is reported correctly. Guarded — any failure reads as unavailable. 

89 """ 

90 try: 

91 return make_adapter(spec).available() 

92 except Exception: # noqa: BLE001 - diagnostics must never crash 

93 return False 

94 

95 

96def _resolved_command(spec): 

97 """Absolute path a CLI agent's command resolves to on PATH (issue #296). 

98 

99 Lets an operator verify *which* binary will run (a poisoned PATH could 

100 resolve a bare name to a shim). None for a local/HTTP agent (no command) or 

101 when nothing is found on PATH. 

102 """ 

103 command = getattr(spec, "command", "") or "" 

104 has_endpoint = bool(getattr(spec, "endpoint", None)) 

105 if not command or is_commandless_vendor(spec_adapter(spec)) or has_endpoint: 

106 return None 

107 try: 

108 return shutil.which(command) 

109 except Exception: # noqa: BLE001 - diagnostics must never crash 

110 return None 

111 

112 

113def _probe_models(spec): 

114 """Model ids this agent could be pointed at, or None (issue #662). 

115 

116 Delegates to the adapter's own ``list_models`` seam — ``agy models`` for 

117 Antigravity, the OpenAI-compatible ``/models`` listing for a local server — 

118 and, like every other doctor probe, swallows any failure. 

119 """ 

120 try: 

121 models = make_adapter(spec).list_models() 

122 except Exception: # noqa: BLE001 - diagnostics must never crash 

123 return None 

124 if not models: 

125 return None 

126 return [_redact_value(m) for m in models] 

127 

128 

129#: The model `ollama pull` is suggested for when a seat names none (it is the one 

130#: `jury init --preset offline` writes). 

131_SUGGESTED_LOCAL_MODEL = "qwen2.5-coder:7b" 

132 

133 

134def _model_is_listed(model: str, listing: list[str]) -> bool: 

135 """Whether a local server lists ``model``. Ollama resolves an untagged name to 

136 ``:latest``, so ``qwen2.5-coder`` is served by ``qwen2.5-coder:latest``.""" 

137 return model in listing or (":" not in model and f"{model}:latest" in listing) 

138 

139 

140def _local_model_gap(spec): 

141 """What a reachable local seat's server says about its model (issue #849). 

142 

143 A local seat was reported ready whenever its server answered, so Ollama with 

144 nothing pulled read as a working reviewer and every run through it came back 

145 `HTTP 404: model … not found`. Returns: 

146 

147 * ``("unusable", why)`` — the server lists **no** model. Ollama, vLLM, 

148 LM Studio and llama.cpp all list at least the one they would serve, so an 

149 empty list means nothing can answer a review, whatever the seat names. 

150 * ``("unlisted", why)`` — models are listed but not the configured one. Ollama 

151 answers that with a 404, but a llama.cpp server ignores the name and serves 

152 the model it loaded, so this is a warning, never a verdict. 

153 * ``None`` — the model is listed, or the listing itself failed. No evidence is 

154 not evidence of a fault, so a failed listing changes nothing. 

155 """ 

156 if spec_adapter(spec) != "local": 

157 return None 

158 endpoint = getattr(spec, "endpoint", None) or _DEFAULT_LOCAL_ENDPOINT 

159 try: 

160 listing = local_model_listing(endpoint) 

161 except Exception: # noqa: BLE001 - diagnostics must never crash 

162 return None 

163 if listing is None: 

164 return None 

165 shown = redact_url_userinfo(endpoint) 

166 model = getattr(spec, "model", "") or "" 

167 if not listing: 

168 pull = _redact_value(model) if model else _SUGGESTED_LOCAL_MODEL 

169 return ( 

170 "unusable", 

171 f"the local server at '{shown}' lists no models — Ollama, vLLM, LM Studio and " 

172 f"llama.cpp all list the one they serve, so this seat has nothing to answer " 

173 f"with; pull one first (for Ollama: `ollama pull {pull}`). A proxy that serves " 

174 f"models it does not list is unaffected: a run still calls it", 

175 ) 

176 if model and not _model_is_listed(model, listing): 

177 served = ", ".join(_redact_value(m) for m in listing[:5]) 

178 more = ", …" if len(listing) > 5 else "" 

179 return ( 

180 "unlisted", 

181 f"model '{_redact_value(model)}' is not among the models the local server at " 

182 f"'{shown}' lists ({served}{more}); pull it (for Ollama: `ollama pull " 

183 f"{_redact_value(model)}`) or set 'model' to one it serves — a llama.cpp " 

184 f"server ignores the name and uses the model it loaded, so there it is harmless", 

185 ) 

186 return None 

187 

188 

189def _endpoint_for(spec): 

190 """The HTTP endpoint an agent talks to, or None for a CLI agent. 

191 

192 A configured ``endpoint`` wins (with any userinfo credentials stripped); a 

193 hosted-API vendor reports its adapter's fixed vendor URL, so the export 

194 answers "where would this actually go?" for every non-CLI transport. 

195 """ 

196 if getattr(spec, "endpoint", None): 

197 return redact_url_userinfo(spec.endpoint) 

198 vendor = spec_adapter(spec) 

199 if vendor == "local": 

200 return _DEFAULT_LOCAL_ENDPOINT 

201 try: 

202 api_url = getattr(make_adapter(spec), "_api_url", None) 

203 return api_url() if callable(api_url) else None 

204 except Exception: # noqa: BLE001 - diagnostics must never crash 

205 return None 

206 

207 

208def _unavailable_transport(spec) -> str: 

209 """Which transport failed for an unavailable seat: local, hosted-api or cli. 

210 

211 The single classifier behind both unavailability messages — the per-agent 

212 ``reason`` in the export and the corresponding line in ``warnings`` (issue 

213 #701, review round 3). Two readers asking the same question of the same seat 

214 must get the same answer, so they ask it here, once, of the normalised 

215 vendor: ``vendor = "XAI-API"`` used to be a hosted API to one reader and a 

216 CLI with a missing ``command`` to the other. 

217 """ 

218 # Asked of the adapter (#705): which transport failed is a fact about how 

219 # the seat is reached, not about whose model was on the other end. 

220 vendor = spec_adapter(spec) 

221 if vendor == "local": 

222 return "local" 

223 if vendor in _HOSTED_API_VENDORS or vendor.endswith("-api"): 

224 return "hosted-api" 

225 return "cli" 

226 

227 

228def _unavailable_reason(spec, capability_warnings) -> str: 

229 """Why an agent is not usable, in one line (pure given its inputs). 

230 

231 Prefers the adapter's own capability warning (e.g. "ANTHROPIC_API_KEY is not 

232 set") so the export cannot drift from what the adapter reports, and falls 

233 back to a transport-appropriate message when the probe said nothing. 

234 """ 

235 if capability_warnings: 

236 return "; ".join(capability_warnings) 

237 transport = _unavailable_transport(spec) 

238 if transport == "local": 

239 endpoint = redact_url_userinfo(spec.endpoint or _DEFAULT_LOCAL_ENDPOINT) 

240 return f"endpoint '{endpoint}' is not reachable" 

241 if transport == "hosted-api": 

242 return "the hosted API is not reachable" 

243 if getattr(spec, "command", ""): 

244 return f"command '{_redact_value(spec.command)}' is not on PATH" 

245 return "not available" 

246 

247 

248def _agent_entry(spec, probe_models: bool = False, local_gaps=None): 

249 caps = _detect_capabilities(spec) 

250 available = _is_available(spec) 

251 # A server that answers but lists no model cannot review (#849): the seat is 

252 # reported unavailable, with that as its reason, so `ready to run` stops saying 

253 # yes for a panel whose only reviewer has nothing to run. The doctor passes a 

254 # dict: an available seat's server is listed once, here, and the result kept for 

255 # the warnings. A run's metadata passes nothing, so recording a run never makes 

256 # a listing request, and an unavailable seat is never listed. 

257 gap = None 

258 if available and local_gaps is not None: 

259 gap = local_gaps[spec.name] = _local_model_gap(spec) 

260 unusable = gap[1] if gap and gap[0] == "unusable" else None 

261 if unusable: 

262 available = False 

263 capability_warnings = [_redact_value(w) for w in caps.get("warnings", [])] 

264 return { 

265 "name": _redact_value(spec.name), 

266 "command": _redact_value(spec.command), 

267 "endpoint": _endpoint_for(spec), 

268 "resolved": _resolved_command(spec), 

269 # Two vendor fields, because they answer two different questions and a 

270 # row that showed only one was read as answering both (#701, round 2). 

271 # `vendor` is provenance: the vendor the operator named, in the one 

272 # normalised spelling every rule reads (round 3 — the spec normalises on 

273 # construction, so this is the same string the doctor reasons about). 

274 # `vendor_identity` is the gate's view: what this seat counts as under 

275 # `min_vendors`, which is `cli` for anything the build cannot identify. 

276 "vendor": _redact_value(spec.vendor), 

277 "vendor_identity": vendor_identity(getattr(spec, "vendor", "")), 

278 # The third field answers the third question (#705): `vendor` is what the 

279 # operator called the seat, `vendor_identity` is what the gate counts it 

280 # as, and `adapter` is the protocol that builds its command line. A 

281 # Codex seat and a GPT-through-Cursor seat differ in nothing else. 

282 "adapter": spec_adapter(spec), 

283 "available": available, 

284 "reason": None 

285 if available 

286 else (unusable or _unavailable_reason(spec, capability_warnings)), 

287 "version": _redact_value(caps.get("version")), 

288 "capabilities": { 

289 "supports_headless": caps.get("supports_headless"), 

290 "supports_model_selection": caps.get("supports_model_selection"), 

291 "status": caps.get("status"), 

292 }, 

293 # Only the JSON export renders a model listing, and discovering one 

294 # costs a subprocess (`agy models`) or an HTTP round trip per agent. The 

295 # text report would pay that for nothing, so the probe is opt-in. 

296 "models": _probe_models(spec) if probe_models else None, 

297 "effort": _redact_value(getattr(spec, "effort", None)), 

298 "effort_supported": effort_supported(spec_adapter(spec)), 

299 # Cost tier (issue #714): which seats tiered routing may bench on a 

300 # routine diff, and which one it may anchor with. 

301 "tier": getattr(spec, "tier", "frontier"), 

302 "capability_warnings": capability_warnings, 

303 } 

304 

305 

306def _config_summary(cfg): 

307 """Build a redacted, secret-free summary of the loaded config.""" 

308 return { 

309 "rounds": cfg.rounds, 

310 "chair": _redact_value(cfg.chair), 

311 "context_mode": _redact_value(cfg.context.mode), 

312 "enabled_agents": [_redact_value(a.name) for a in cfg.enabled_agents], 

313 } 

314 

315 

316# Hosted-API vendors (issue #430/#432): no `command`/`endpoint`, so neither 

317# the "local" nor the "CLI on PATH" branch below is the right diagnosis when 

318# one is unavailable. 

319_HOSTED_API_VENDORS = ("anthropic-api", "openai-api", "google-api", "xai-api") 

320 

321 

322def _detect_warnings(cfg, local_gaps=None) -> list[str]: 

323 """Best-effort config sanity checks reported to the user.""" 

324 warnings: list[str] = [] 

325 if not cfg.agents: 

326 warnings.append("no agents are configured") 

327 enabled = cfg.enabled_agents 

328 if cfg.agents and not enabled: 

329 warnings.append("all configured agents are disabled") 

330 names = {a.name for a in cfg.agents} 

331 if cfg.chair not in names: 

332 warnings.append(f"chair '{_redact_value(cfg.chair)}' does not match any configured agent") 

333 for agent in enabled: 

334 if _is_available(agent): 

335 # Both cases are warnings as well as the seat's reason: the text report 

336 # prints warnings, not reasons, and `ollama pull <model>` is the line a 

337 # user needs to see (#850 third seat). 

338 gap = (local_gaps or {}).get(agent.name) 

339 if gap: 

340 warnings.append(f"agent '{_redact_value(agent.name)}' (local): {gap[1]}") 

341 # An available hosted-API seat (no command, no endpoint) with no model is still 

342 # reported ready, but the API call fails at request time — `--config-validate` 

343 # warns while `--doctor` did not (#831). Flag it here too, in the same words. 

344 if not ( 

345 (getattr(agent, "command", "") or "") 

346 or (getattr(agent, "endpoint", "") or "") 

347 or (getattr(agent, "model", "") or "") 

348 ): 

349 warnings.append( 

350 f"agent '{_redact_value(agent.name)}' " 

351 f"({_redact_value(vendor_identity(getattr(agent, 'vendor', '')))}) " 

352 "has no 'model'; the server or API call will likely reject the request" 

353 ) 

354 continue 

355 # Classified by the SAME predicate `_unavailable_reason` uses, so the 

356 # warning list and the per-agent `reason` cannot diagnose one seat two 

357 # different ways (issue #701, review round 3). They differ only in 

358 # wording; disagreeing about *which* transport failed was the bug. 

359 transport = _unavailable_transport(agent) 

360 if transport == "local": 

361 warnings.append( 

362 f"agent '{_redact_value(agent.name)}' (local) endpoint " 

363 f"'{redact_url_userinfo(agent.endpoint or _DEFAULT_LOCAL_ENDPOINT)}' " 

364 f"is not reachable" 

365 ) 

366 elif transport == "hosted-api": 

367 # Reuse the adapter's own capability warning (issue #430) instead 

368 # of re-deriving the vendor -> env-var mapping here, so the 

369 # message can't drift from what the adapter actually reports. 

370 caps = _detect_capabilities(agent) 

371 reason = "; ".join(caps.get("warnings", [])) or "the hosted API is not reachable" 

372 warnings.append(f"agent '{_redact_value(agent.name)}' (hosted API): {reason}") 

373 else: 

374 warnings.append( 

375 f"agent '{_redact_value(agent.name)}' command " 

376 f"'{_redact_value(agent.command)}' is not on PATH" 

377 ) 

378 return warnings 

379 

380 

381def _panel_readiness(cfg, agents, min_vendors=None, local_fallback=None) -> dict: 

382 """How close this machine is to being able to form cross-vendor consensus. 

383 

384 Doctor is offline and runs no review, so it can only report what it can 

385 see: how many distinct vendors are ENABLED, and how many of those are 

386 reachable. ``contributing_vendors`` is therefore always ``None`` here — the 

387 contributed-vendor count is a property of a run, and lives in the run 

388 metadata's ``panel.vendors``. Saying so in the export is the point: an 

389 available CLI that returns nothing is exactly the failure #635 was, and a 

390 green doctor is not evidence against it. 

391 

392 Both counts are of :func:`config.vendor_identity`, the same arithmetic 

393 :func:`metadata.distinct_vendors` does at the gate, so 

394 ``vendors_configured`` equals what a run would count for the same config 

395 (#701, round 2). Counting raw strings here meant two seats on the generic 

396 fallback read as two vendors in ``--doctor`` and as one vendor in the run: 

397 doctor called the bench cross-vendor ready and the run then refused it. 

398 

399 ``min_vendors`` is the run's ``--min-vendors`` / ``--no-min-vendors`` value 

400 (``None`` when neither was named), resolved as the run resolves it. With 

401 ``local_fallback`` — the seat :func:`scaffold.zero_config_local_seat` says a 

402 run with no config would review with — the panel is that one seat, as the 

403 run forms it (#863): the built-in seats it replaces are all unreachable, and 

404 its server has just listed the model, which is the run's reason to seat it. 

405 """ 

406 entries = {entry["name"]: entry for entry in agents} 

407 enabled = list(getattr(cfg, "enabled_agents", []) or []) 

408 if local_fallback is not None: 

409 enabled = [local_fallback] 

410 entries = {local_fallback.name: {"available": True}} 

411 available = { 

412 vendor_identity(a.vendor) for a in enabled if entries.get(a.name, {}).get("available") 

413 } - {""} 

414 minimum, explicit = resolve_min_vendors(min_vendors, cfg) 

415 # The number a downstream consumer counts, which doctor never reported and 

416 # which is not the vendor count (#699): one review per agent that answers. 

417 # The chair's synthesis record is not added here — a consumer reads it as the 

418 # panel's consensus, not as a review, and adding it is how a bench with 

419 # nothing reachable came to advertise one review. Doctor can only see 

420 # reachability, so this is the CEILING — an agent that runs and returns 

421 # nothing, or names nothing checkable, casts no review (#700) — which is why 

422 # it is labelled "at most" below. 

423 seats = sum(1 for a in enabled if entries.get(a.name, {}).get("available")) 

424 panel = { 

425 "vendors_configured": claimed_vendors(enabled), 

426 "vendors_available": len(available), 

427 "min_vendors": minimum, 

428 "contributing_vendors": None, 

429 "panelists_available": seats, 

430 "reviews_supplied_max": seats, 

431 "min_reviews": int(getattr(cfg.ci, "min_reviews", 0) or 0), 

432 } 

433 # Derived from the same predicate the warning uses, so the field and the 

434 # warning cannot disagree about the same machine (#682, round 3). 

435 panel["multi_vendor_ready"] = not _gate_would_fail(panel, explicit) 

436 return panel 

437 

438 

439#: What a machine with no loadable config can prove about its panel: nothing. 

440#: ``multi_vendor_ready`` is ``False`` here for a different reason than below — 

441#: not "the gate would fail" but "there is no config to satisfy", which is not a 

442#: readiness anyone should build on. 

443_NO_PANEL = { 

444 "vendors_configured": 0, 

445 "vendors_available": 0, 

446 "min_vendors": 0, 

447 "contributing_vendors": None, 

448 "panelists_available": 0, 

449 "reviews_supplied_max": 0, 

450 "min_reviews": 0, 

451 "multi_vendor_ready": False, 

452} 

453 

454 

455def _gate_would_fail(panel, explicit: bool = False) -> bool: 

456 """Would a run on this machine fail the cross-vendor gate? 

457 

458 The single place doctor decides that, so the ``multi_vendor_ready`` field 

459 and the warning below are the same judgement rendered twice. It mirrors 

460 :func:`metadata.collapse_reason`'s scoping term for term: the gate is off at 

461 ``0``; a config with fewer distinct vendors enabled than the threshold never 

462 claimed that consensus and is left alone; otherwise every configured vendor 

463 short of the threshold is a vendor the run cannot hear from. 

464 

465 Doctor can only see reachability, so this is a *lower* bound on failure: a 

466 reachable CLI that returns nothing (#635) fails the gate too, and no offline 

467 check can predict it. That is why the report says so in as many words. 

468 

469 It is :func:`metadata.vendor_guard_fails`, the run's own predicate, with the 

470 reachable vendors standing in for the contributing ones; an ``explicit`` 

471 threshold (named on the command line) is enforced unscoped, as in the run. 

472 """ 

473 return vendor_guard_fails( 

474 panel["vendors_available"], 

475 panel["min_vendors"], 

476 None if explicit else panel["vendors_configured"], 

477 ) 

478 

479 

480def _fallback_ready_reason(panel, explicit: bool) -> str | None: 

481 """Why the zero-config fallback's one seat reads `cross-vendor ready: yes` (pure). 

482 

483 One seat is not a cross-vendor panel, so the text report says which of three 

484 things makes the gate pass (#863): the guard is off; the default threshold 

485 was scoped away because one seat claims no cross-vendor consensus; or a 

486 threshold is met. ``None`` when the gate would fail — there is nothing to 

487 explain, and the warning says the rest. 

488 """ 

489 if not panel["multi_vendor_ready"]: 

490 return None 

491 required = panel["min_vendors"] 

492 if required <= 0: 

493 return "the gate is off" 

494 if not explicit and panel["vendors_configured"] < required: 

495 return "the gate would not fail: one seat claims no cross-vendor consensus" 

496 return ( 

497 f"the gate would not fail: {panel['vendors_available']} vendor(s) reachable, " 

498 f"{required} required" 

499 ) 

500 

501 

502def _panel_warning(panel, explicit: bool = False) -> str | None: 

503 """The one actionable thing offline diagnostics can say about the panel. 

504 

505 Fires only when a run on this machine would actually fail the gate, because 

506 then it exits 3 and the operator would rather know now. A warning that does 

507 not predict the run is worse than none — it is the kind people learn to 

508 scroll past — which is also why a single-vendor config under the shipped 

509 ``min_vendors = 2`` is silent: :func:`metadata.collapse_reason` leaves that 

510 run alone, so there is nothing to warn about. 

511 """ 

512 if not _gate_would_fail(panel, explicit): 

513 return None 

514 asked = ( 

515 f"--min-vendors {panel['min_vendors']} asks for {panel['min_vendors']} vendors" 

516 if explicit 

517 else f"{panel['vendors_configured']} vendors are enabled" 

518 ) 

519 return ( 

520 f"{asked} but only " 

521 f"{panel['vendors_available']} is/are reachable; a run would fail the " 

522 f"cross-vendor guard (min_vendors = {panel['min_vendors']}, exit 3). " 

523 f"Install the missing CLI, or opt out with `--no-min-vendors` " 

524 f"(`[jury.ci] min_vendors = 0`)." 

525 ) 

526 

527 

528def _recommendations( 

529 config_path, 

530 config_summary, 

531 agents, 

532 config_error: bool = False, 

533 local_fallback=None, 

534 local_models=None, 

535) -> dict: 

536 """Build actionable next-steps from the diagnostics (issue: doctor UX). 

537 

538 Returns ``{"ready": bool, "steps": [str, ...]}``. ``ready`` is true when at 

539 least one agent is reachable. Steps point the user at the cheapest fix: 

540 scaffold a config, install a CLI, or use a reachable local model. 

541 

542 *config_error* says the config could not be loaded at all, so no seat was 

543 inspected — and the only honest next step is the error itself (issue #708). 

544 

545 *local_fallback* is the seat a run with no config would review with (#863): 

546 that run has a reviewer, so the machine is ready, and the step says what the 

547 run will do. *local_models* is a listing already made, reused rather than 

548 asked for twice. 

549 """ 

550 steps: list[str] = [] 

551 available = [a for a in agents if a.get("available")] 

552 ready = bool(available) 

553 

554 # No config file in play -> suggest scaffolding one. 

555 if config_path is None and not Path("jury.toml").exists(): 

556 steps.append("No jury.toml found — run `jury init` to create one.") 

557 

558 # Nothing was loaded, so nothing below is knowable: "install an agent CLI" 

559 # would name the wrong fault for a config the RUN also refuses, and it costs 

560 # a local-model probe to say it. The verdict is the config error (#708). 

561 if config_error: 

562 steps.append( 

563 "The config could not be loaded, so no agent was inspected — fix the " 

564 "config error above. A run refuses this file for the same reason." 

565 ) 

566 return {"ready": False, "steps": steps} 

567 

568 if local_fallback is not None: 

569 ready = True 

570 steps.append( 

571 f"No agent CLI is available, so a run with no jury.toml reviews with the " 

572 f"local model '{local_fallback.model}' alone: a single-vendor panel, which " 

573 f"the default cross-vendor guard does not fail. For a panel you choose: " 

574 f"`jury init --preset offline` (or `--list-models`)." 

575 ) 

576 hint = agy_opt_in_hint((a.get("adapter") for a in agents), shutil.which) 

577 if hint: 

578 steps.append(f"Note: {hint}.") 

579 elif not ready: 

580 from .adapters import list_local_models 

581 

582 models = local_models if local_models is not None else list_local_models() 

583 if models: 

584 steps.append( 

585 f"No agent CLI is available, but a local model server is reachable " 

586 f"({len(models)} model(s): {', '.join(models[:3])}). Add a free local " 

587 f"reviewer: `jury init --preset offline` (or `--list-models`)." 

588 ) 

589 else: 

590 steps.append( 

591 "No reviewer is available. Install an agent CLI (claude / codex); " 

592 "run a local model (e.g. `ollama serve` + `ollama pull " 

593 'qwen2.5-coder:7b`) and add a `vendor = "local"` agent; or, with no ' 

594 "install at all, set a hosted-API key (e.g. ANTHROPIC_API_KEY) and add " 

595 "that seat with `jury init --agents claude-api` — or use `--mock` for an " 

596 "offline demo." 

597 ) 

598 # agy on PATH but in no seat: say why the one CLI here was not used. 

599 hint = agy_opt_in_hint((a.get("adapter") for a in agents), shutil.which) 

600 if hint: 

601 steps.append(f"Note: {hint}.") 

602 else: 

603 missing = [ 

604 a["name"] 

605 for a in agents 

606 if not a.get("available") 

607 and config_summary 

608 and a["name"] in config_summary.get("enabled_agents", []) 

609 ] 

610 if missing: 

611 steps.append( 

612 f"Enabled but unavailable (will be skipped): {', '.join(missing)}. " 

613 f"Install them or run with `--strict` to fail instead." 

614 ) 

615 

616 return {"ready": ready, "steps": steps} 

617 

618 

619def build_diagnostics(config_path=None, probe_models: bool = False, min_vendors=None): 

620 """Build a SAFE diagnostics dict for the given config path. 

621 

622 Best-effort: if the config cannot be loaded, the error is captured as a 

623 string under ``config_warnings`` and ``config`` is left ``None``. Never 

624 raises for a bad/missing config. The returned dict never contains the raw 

625 diff or any agent output. 

626 

627 The config is loaded WITH validation — the same call a run makes (issue 

628 #708). Best-effort is about not crashing, not about being more permissive 

629 than the run: this report's whole promise is that its arithmetic equals what 

630 a run counts, and it cannot keep that promise while accepting a file the run 

631 rejects. It used to. An `adapter` name this build does not have is a hard 

632 config error, but with validation off the seat kept the name, `make_adapter` 

633 missed the registry and fell through to the generic CLI adapter, and three 

634 such seats reported `[available]`, `cross-vendor ready: yes` and `ready to 

635 run: yes` for a config the run refused before its first round. A hard error 

636 is now the report's verdict: it lands in ``config_warnings`` naming the seat 

637 and the adapter, no seat is described, and ``ready to run`` is ``no``. 

638 Warnings are still warnings — ``strict`` is not passed, so a soft issue 

639 (unknown vendor, unknown key) is reported, not fatal, exactly as before. 

640 

641 ``probe_models`` opts into discovering each agent's available model ids — 

642 a subprocess (``agy models``) or an HTTP round trip *per agent*, each 

643 time-boxed but not free. Only ``--doctor --json`` renders that listing, so 

644 it defaults off: the human report used to pay ~2 s (and up to the probe 

645 timeout if a CLI hangs) for a field it never printed. 

646 

647 ``min_vendors`` is the run's ``--min-vendors`` / ``--no-min-vendors`` value, 

648 so the cross-vendor prediction is made for the threshold the run would use. 

649 With no config file and no reachable seat, the local server is listed once 

650 and :func:`scaffold.zero_config_local_seat` — the run's own decision — says 

651 whether a run would review with a local model; if so, the panel and the 

652 prediction are that run's (#863). 

653 """ 

654 config_summary = None 

655 config_warnings: list[str] = [] 

656 agents: list = [] 

657 panel: dict = dict(_NO_PANEL) 

658 # Distinct from `config_summary is None`, which a caller may also hand 

659 # `_recommendations` directly: this says a load was ATTEMPTED and failed. 

660 # Cleared only on the success path, so a new `except` arm cannot forget it. 

661 config_error = True 

662 local_fallback = None 

663 ready_reason = None 

664 listed: list = [] 

665 

666 def _list_once(): 

667 from .adapters import list_local_models 

668 

669 listed.append(list_local_models()) 

670 return listed[-1] 

671 

672 try: 

673 cfg = load_config(config_path, validate=True) 

674 except FileNotFoundError as exc: 

675 config_warnings.append(f"config error: {redact(str(exc))[0]}") 

676 except OSError as exc: 

677 # Present but unreadable — a chmod-0 jury.toml, or --config naming a directory — 

678 # is reported like the CLI reports it, not raised through the report (#893). 

679 path = exc.filename or config_path or "jury.toml" 

680 why = exc.strerror or str(exc) 

681 config_warnings.append(redact(f"config error: cannot read config {path}: {why}")[0]) 

682 except tomllib.TOMLDecodeError as exc: 

683 config_warnings.append(f"config error: invalid TOML: {redact(str(exc))[0]}") 

684 except ConfigError as exc: 

685 config_warnings.append(f"config error: {redact(str(exc))[0]}") 

686 except (KeyError, ValueError, TypeError) as exc: 

687 config_warnings.append(f"config error: {redact(str(exc))[0]}") 

688 else: 

689 config_error = False 

690 config_summary = _config_summary(cfg) 

691 # One listing per available local server, shared by the entry and the warnings. 

692 local_gaps: dict = {} 

693 agents = [ 

694 _agent_entry(spec, probe_models=probe_models, local_gaps=local_gaps) 

695 for spec in cfg.agents 

696 ] 

697 config_warnings = _detect_warnings(cfg, local_gaps) 

698 # Fold capability/version probe warnings (e.g. an available CLI whose 

699 # version could not be detected) into the user-facing warnings list. 

700 # Probes already ran while building the agent entries above. 

701 enabled_names = {a.name for a in cfg.enabled_agents} 

702 for spec, entry in zip(cfg.agents, agents, strict=False): 

703 if spec.name not in enabled_names: 

704 continue 

705 for warning in entry.get("capability_warnings", []): 

706 config_warnings.append(f"agent '{entry['name']}': {warning}") 

707 # Cross-vendor readiness (issue #682), after the availability probes the 

708 # agent entries already ran — this adds no probe of its own. 

709 local_fallback = zero_config_local_seat( 

710 config_path, 

711 False, 

712 Path("jury.toml").exists(), 

713 lambda: any(e.get("available") for e in agents if e.get("name") in enabled_names), 

714 _list_once, 

715 ) 

716 panel = _panel_readiness(cfg, agents, min_vendors, local_fallback) 

717 explicit = resolve_min_vendors(min_vendors, cfg)[1] 

718 panel_warning = _panel_warning(panel, explicit) 

719 if local_fallback is not None: 

720 ready_reason = _fallback_ready_reason(panel, explicit) 

721 if panel_warning: 

722 config_warnings.append(panel_warning) 

723 # A bench that cannot reach the consumer's minimum is a shortfall this 

724 # machine can prove offline (#699) — worth saying here rather than after 

725 # the run, which is where it used to surface. 

726 short = shortfall( 

727 panel["panelists_available"], 

728 panel["min_reviews"], 

729 stage="on this machine", 

730 ) 

731 if short: 

732 config_warnings.append(short) 

733 

734 return { 

735 "tool_version": __version__, 

736 "python_version": platform.python_version(), 

737 "python_implementation": platform.python_implementation(), 

738 "python_executable": sys.executable, 

739 "os": platform.platform(), 

740 "config_path": str(config_path) if config_path else "(default)", 

741 "agents": agents, 

742 "config": config_summary, 

743 "config_warnings": config_warnings, 

744 "panel": panel, 

745 # The model a run with no config would review with alone, or None (#863). 

746 "local_fallback": local_fallback.model if local_fallback is not None else None, 

747 # Why that one seat reads as cross-vendor ready, for the text report. 

748 "cross_vendor_ready_reason": ready_reason, 

749 "recommendations": _recommendations( 

750 config_path, 

751 config_summary, 

752 agents, 

753 config_error=config_error, 

754 local_fallback=local_fallback, 

755 local_models=listed[-1] if listed else None, 

756 ), 

757 } 

758 

759 

760# Capability flags -> the short labels shown in both renderers, in a fixed order. 

761_CAPABILITY_LABELS = ( 

762 ("supports_headless", "headless"), 

763 ("supports_model_selection", "model-selection"), 

764) 

765 

766 

767def capability_labels(capabilities) -> list[str]: 

768 """Short capability labels for one agent's capability dict (pure).""" 

769 caps = capabilities or {} 

770 return [label for key, label in _CAPABILITY_LABELS if caps.get(key)] 

771 

772 

773def _transport(vendor: str, command: str, endpoint) -> str: 

774 """Classify how an agent is reached: ``cli``, ``api`` or ``local`` (pure).""" 

775 name = normalise_vendor(vendor) 

776 if name == "local": 

777 return "local" 

778 if name.endswith("-api") or name == "openai-compatible": 

779 return "api" 

780 if command: 

781 return "cli" 

782 return "api" if endpoint else "cli" 

783 

784 

785def doctor_report_dict(diagnostics) -> dict: 

786 """Project a diagnostics dict onto the stable ``ai-jury.doctor.v1`` export. 

787 

788 PURE: it runs no probes and touches no network, PATH or filesystem — every 

789 fact comes from the dict :func:`build_diagnostics` already built, so the 

790 probes run exactly once no matter how many renderers consume them. 

791 

792 Both renderers consume this: ``jury --doctor --json`` serializes it, and 

793 :func:`render_report` renders its agent rows from it, so the human report 

794 and the machine export cannot describe the panel differently. 

795 

796 Secrets are never included — only environment *variable names*, which reach 

797 this dict through the adapters' own capability warnings. 

798 """ 

799 agents = [] 

800 for entry in diagnostics.get("agents", []): 

801 vendor = entry.get("vendor") or "" 

802 adapter = entry.get("adapter") or "" 

803 command = entry.get("command") or "" 

804 endpoint = entry.get("endpoint") 

805 # The transport follows the adapter (#705): `vendor = "openai", 

806 # adapter = "cli"` is reached over a CLI, not over OpenAI's API. 

807 transport = _transport(adapter or vendor, command, endpoint) 

808 item = { 

809 "name": entry.get("name"), 

810 "vendor": vendor, 

811 "vendor_identity": entry.get("vendor_identity") or "", 

812 "adapter": adapter, 

813 "transport": transport, 

814 "available": bool(entry.get("available")), 

815 "reason": entry.get("reason"), 

816 } 

817 # One address key per agent, named for the transport that reaches it. 

818 if transport == "cli": 

819 item["command"] = command 

820 else: 

821 item["endpoint"] = endpoint 

822 models = entry.get("models") 

823 item["resolved"] = entry.get("resolved") 

824 item["version"] = entry.get("version") 

825 item["capabilities"] = capability_labels(entry.get("capabilities")) 

826 item["models"] = list(models) if models else None 

827 item["effort_supported"] = bool(entry.get("effort_supported")) 

828 item["effort"] = entry.get("effort") 

829 # Cost tier (#714): what tiered routing reads; `frontier` unless said. 

830 item["tier"] = entry.get("tier") or "frontier" 

831 agents.append(item) 

832 

833 recommendations = diagnostics.get("recommendations") or {} 

834 return { 

835 "schema_version": DOCTOR_SCHEMA_VERSION, 

836 "tool_version": diagnostics.get("tool_version"), 

837 "python": diagnostics.get("python_version"), 

838 "config_path": diagnostics.get("config_path"), 

839 "ready": bool(recommendations.get("ready")), 

840 # Cross-vendor readiness (issue #682). ``contributing_vendors`` is null 

841 # by construction: doctor runs no review, so the contributed count is 

842 # only ever known from a run's metadata (``panel.vendors``). A consumer 

843 # reading this must not treat availability as contribution. 

844 "panel": dict(diagnostics.get("panel") or _NO_PANEL), 

845 "agents": agents, 

846 "warnings": list(diagnostics.get("config_warnings") or []), 

847 } 

848 

849 

850def render_report(diagnostics) -> str: 

851 """Render a human-readable text report from a diagnostics dict. 

852 

853 The Agents section and the readiness line are rendered from 

854 :func:`doctor_report_dict`, the same projection ``--json`` serializes, so 

855 the two views cannot drift; the capability *probe status* is read alongside 

856 it from the raw diagnostics (it is a diagnostic detail, not part of the 

857 exported schema). 

858 """ 

859 report = doctor_report_dict(diagnostics) 

860 lines = [] 

861 lines.append("jury doctor") 

862 lines.append("=" * 40) 

863 lines.append(f"tool version: {diagnostics['tool_version']}") 

864 lines.append( 

865 f"python: {diagnostics['python_version']} ({diagnostics['python_implementation']})" 

866 ) 

867 lines.append(f"python exe: {diagnostics['python_executable']}") 

868 lines.append(f"os: {diagnostics['os']}") 

869 lines.append(f"config path: {diagnostics['config_path']}") 

870 lines.append("") 

871 

872 lines.append("Agents") 

873 lines.append("-" * 40) 

874 agents = report["agents"] 

875 if not agents: 

876 lines.append(" (no agents loaded)") 

877 else: 

878 for agent, probe in zip(agents, diagnostics["agents"], strict=False): 

879 status = "available" if agent["available"] else "MISSING" 

880 # A CLI agent is identified by its command; an api/local agent has 

881 # none, so name the endpoint it actually talks to rather than 

882 # printing an empty `command=`. 

883 if agent["transport"] == "cli": 

884 address = f"command={agent.get('command', '')}" 

885 else: 

886 address = f"endpoint={agent.get('endpoint') or '(unknown)'}" 

887 # A seat whose configured vendor is not the identity it carries 

888 # says so on its own row, so the row and the panel count below can 

889 # be read together without arithmetic (#701, round 2). 

890 identity = agent.get("vendor_identity") or "" 

891 vendor_field = agent["vendor"] 

892 if identity and identity != normalise_vendor(vendor_field): 

893 vendor_field = f"{vendor_field} -> counts as {identity}" 

894 # Vendor AND adapter on every row (#705), never one standing in for 

895 # both: a reader must be able to tell a Codex seat (openai/openai) 

896 # from a GPT-through-Cursor seat (openai/cli) at a glance, and the 

897 # `--doctor` row is where that difference becomes visible. 

898 lines.append( 

899 f" [{status:>9}] {agent['name']} (vendor={vendor_field}, " 

900 f"adapter={agent.get('adapter') or '(default)'}, {address})" 

901 ) 

902 if agent.get("command"): 

903 resolved = agent.get("resolved") or "(not found on PATH)" 

904 lines.append(f" resolved: {resolved}") 

905 version = agent.get("version") or "unknown" 

906 cap_summary = ", ".join(agent["capabilities"]) or "none" 

907 cap_status = (probe.get("capabilities") or {}).get("status") or "unknown" 

908 summary = ( 

909 f" version={version}, capabilities=[{cap_summary}] " 

910 f"(probe: {cap_status})" 

911 ) 

912 if agent.get("effort"): 

913 summary += f", effort={agent['effort']}" 

914 lines.append(summary) 

915 lines.append("") 

916 

917 lines.append("Config summary") 

918 lines.append("-" * 40) 

919 config = diagnostics["config"] 

920 if config is None: 

921 lines.append(" (config could not be loaded)") 

922 else: 

923 lines.append(f" rounds: {config['rounds']}") 

924 lines.append(f" chair: {config['chair']}") 

925 lines.append(f" context mode: {config['context_mode']}") 

926 enabled = ", ".join(config["enabled_agents"]) or "(none)" 

927 lines.append(f" enabled: {enabled}") 

928 lines.append("") 

929 

930 panel = report["panel"] 

931 lines.append("Cross-vendor readiness") 

932 lines.append("-" * 40) 

933 if diagnostics.get("local_fallback"): 

934 lines.append( 

935 f" panel of a run: the local model '{diagnostics['local_fallback']}' alone " 

936 f"(no jury.toml and no agent CLI: the zero-config fallback)" 

937 ) 

938 lines.append(f" vendors enabled: {panel['vendors_configured']} (by vendor identity)") 

939 lines.append(f" vendors reachable: {panel['vendors_available']} (by vendor identity)") 

940 lines.append(f" min_vendors gate: {panel['min_vendors'] or 'off'}") 

941 ready_text = "yes" if panel["multi_vendor_ready"] else "no" 

942 if diagnostics.get("cross_vendor_ready_reason"): 

943 # One seat is not a cross-vendor panel: "ready" means the gate would not 

944 # fail it, and the reason says why (#863). 

945 ready_text += f" ({diagnostics['cross_vendor_ready_reason']})" 

946 lines.append(f" cross-vendor ready: {ready_text}") 

947 # The number a consumer counts, said in the same breath as readiness (#699). 

948 # "cross-vendor ready: yes" on a bench that cannot supply the reviews a gate 

949 # requires is a true statement that answers the wrong question. 

950 lines.append( 

951 f" reviews for a consumer: at most {panel['reviews_supplied_max']} " 

952 f"({panel['panelists_available']} panel ballot(s); the chairing agent " 

953 f"reviews too, so its ballot is one of them, and the chair's synthesis " 

954 f"record is not a review)" 

955 ) 

956 lines.append(f" min_reviews gate: {panel['min_reviews'] or 'off'}") 

957 lines.append( 

958 " note: this checks availability, not contribution. A reachable CLI " 

959 "can still return no review (#635) — only a run can prove the panel." 

960 ) 

961 lines.append( 

962 " note: counted by vendor identity, the same arithmetic min_vendors " 

963 "uses — a vendor this build does not recognise counts as 'cli', so two " 

964 "such seats are one vendor here and in the run. The Agents rows above " 

965 "show each seat's configured vendor string." 

966 ) 

967 lines.append("") 

968 

969 lines.append("Warnings") 

970 lines.append("-" * 40) 

971 warnings = diagnostics["config_warnings"] 

972 if not warnings: 

973 lines.append(" (none)") 

974 else: 

975 for warning in warnings: 

976 lines.append(f" - {warning}") 

977 lines.append("") 

978 

979 rec = diagnostics.get("recommendations") or {} 

980 lines.append("Next steps") 

981 lines.append("-" * 40) 

982 lines.append(f" ready to run: {'yes' if report['ready'] else 'no'}") 

983 for step in rec.get("steps", []): 

984 lines.append(f" - {step}") 

985 lines.append("") 

986 

987 lines.append( 

988 "Privacy: no telemetry is collected or sent. This report is " 

989 "local-only and redacts secret-like values." 

990 ) 

991 

992 return "\n".join(lines)