Coverage for src/ai_jury/adapters.py: 98%

855 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-30 06:29 +0000

1"""Agent adapters — each wraps one native coding-agent CLI in headless mode. 

2 

3Every adapter turns a prompt into a subprocess invocation and captures stdout as 

4the agent's response. Adapters are intentionally thin: the orchestrator owns the 

5prompt content and the round structure; an adapter only knows how to *invoke its 

6CLI*. 

7 

8Headless invocations (verified against installed CLIs, early 2026). The prompt 

9embeds the redacted diff, so it is delivered on STDIN (never argv) for every 

10real adapter so it is not exposed in the process list (issue #287): 

11 - Claude Code : ``claude -p --output-format text`` (prompt piped via stdin) 

12 - Codex CLI : ``codex exec <args>`` (prompt piped via stdin) 

13 - Antigravity : ``agy --print`` (prompt piped via stdin) 

14""" 

15 

16from __future__ import annotations 

17 

18import contextlib 

19import errno 

20import json 

21import os 

22import re 

23import shutil 

24import signal 

25import subprocess 

26import tempfile 

27import time 

28from dataclasses import dataclass, field 

29 

30from . import config as config_module 

31from . import privilege, redaction 

32from .config import AgentSpec 

33from .findings import emitted_findings_block 

34 

35# Cap on a single local-model HTTP response body (issue #293/F-9). A chat 

36# completion is small; an unbounded read from a malicious/buggy endpoint would 

37# let it OOM the process. 

38_MAX_RESPONSE_BYTES = 16 * 1024 * 1024 

39 

40 

41def _kill_process_group(proc: subprocess.Popen) -> None: 

42 """Best-effort kill of the child's whole process group (issue #293/F-7).""" 

43 if hasattr(os, "killpg"): 43 ↛ 47line 43 didn't jump to line 47 because the condition on line 43 was always true

44 with contextlib.suppress(ProcessLookupError, PermissionError, OSError): 

45 os.killpg(os.getpgid(proc.pid), signal.SIGKILL) 

46 return 

47 with contextlib.suppress(OSError): 

48 proc.kill() 

49 

50 

51def _spawn( 

52 argv: list[str], stdin: str | None, timeout: int, cwd: str | None = None 

53) -> subprocess.CompletedProcess: 

54 """Run a CLI with stdout/stderr captured, killing the whole group on timeout. 

55 

56 ``subprocess.run(timeout=…)`` SIGKILLs only the direct child, so an agent CLI 

57 that wraps node/python can leak orphaned grandchildren (issue #293/F-7). The 

58 child is started in its own session (process-group leader); on timeout the 

59 entire group is killed before re-raising ``TimeoutExpired`` so the caller's 

60 handling is unchanged. ``cwd`` starts the child somewhere other than this 

61 process's working directory (see :func:`_review_workdir`). Returns a 

62 ``CompletedProcess``. 

63 """ 

64 popen_kwargs: dict = { 

65 "stdin": subprocess.PIPE if stdin is not None else None, 

66 "stdout": subprocess.PIPE, 

67 "stderr": subprocess.PIPE, 

68 "text": True, 

69 } 

70 if cwd is not None: 

71 popen_kwargs["cwd"] = cwd 

72 if hasattr(os, "setsid"): 72 ↛ 74line 72 didn't jump to line 74 because the condition on line 72 was always true

73 popen_kwargs["start_new_session"] = True 

74 proc = subprocess.Popen(argv, **popen_kwargs) 

75 try: 

76 out, err = proc.communicate(input=stdin, timeout=timeout) 

77 except subprocess.TimeoutExpired: 

78 _kill_process_group(proc) 

79 with contextlib.suppress(Exception): 

80 proc.communicate() # reap the killed child 

81 raise 

82 return subprocess.CompletedProcess(argv, proc.returncode, out, err) 

83 

84 

85@contextlib.contextmanager 

86def _review_workdir(isolate: bool): 

87 """A fresh, empty directory for one reviewer process, removed afterwards. 

88 

89 Yields ``None`` — "inherit this process's directory" — when ``isolate`` is 

90 false. Otherwise the reviewer starts outside the repository under review, so 

91 whatever that repository carries for the agent CLI to pick up on its own is 

92 out of reach: instruction files (``CLAUDE.md``, ``AGENTS.md``), project 

93 settings with hooks or MCP servers (``.claude/settings.json``, ``.mcp.json``), 

94 and the relative path to a ``.env``. On a PR checkout every one of those is 

95 the author's, and a reviewer needs none of them — its prompt carries the diff. 

96 

97 Cleanup ignores errors, so a file the CLI left behind can never turn a 

98 finished review into a failure; the directory is created per call, so two 

99 seats running in parallel never share one. 

100 """ 

101 if not isolate: 

102 yield None 

103 return 

104 with tempfile.TemporaryDirectory(prefix="ai-jury-review-", ignore_cleanup_errors=True) as path: 

105 yield path 

106 

107 

108def _read_only_extra_args(spec: AgentSpec) -> list[str]: 

109 """The agent's ``extra_args`` with the read-only sandbox added when none is named. 

110 

111 Enforced at the adapter layer (issue #288) so an **empty** ``extra_args`` cannot 

112 produce a write-capable reviewer of an attacker-controlled diff. It injects; it 

113 does not override. A config that names its own sandbox keeps it — ``-s 

114 workspace-write`` is passed through as written — codex's bypass flags are passed 

115 through too, and the ``cli``/``xai`` adapters have no enforcement at all. Those 

116 are the cases :func:`ai_jury.privilege.audit_agent` reports and ``--strict`` 

117 fails on (#750); this function is not the guarantee on its own. 

118 """ 

119 # Keyed on the ADAPTER, not the vendor (issue #705): the sandbox flag is a 

120 # property of the CLI being spawned, not of whose model answers. A seat with 

121 # `vendor = "openai", adapter = "cli", command = "cursor-agent"` must not have 

122 # codex's `-s read-only` spliced into an unrelated binary. 

123 return privilege.enforce_read_only(config_module.spec_adapter(spec), spec.extra_args) 

124 

125 

126def _write_extra_args(spec: AgentSpec) -> list[str]: 

127 """The agent's ``extra_args`` with its vendor write/tool mode enabled (#661). 

128 

129 Reached ONLY from ``jury run-agent --role implement|fix --allow-write``, via 

130 :meth:`Adapter.build_write_argv`. Nothing on the panel path calls it. 

131 """ 

132 return privilege.enable_write(config_module.spec_adapter(spec), spec.extra_args) 

133 

134 

135# Short timeout for capability/version probes. Detection is best-effort and must 

136# never slow down or block a normal run, so probes are deliberately snappy. 

137_VERSION_PROBE_TIMEOUT = 10 

138 

139# Matches a version-looking token, e.g. "1.2", "1.2.3", "v0.45.1". 

140_VERSION_RE = re.compile(r"\d+\.\d+(?:\.\d+)?") 

141 

142# Capability/version probe statuses. 

143CAP_OK = "ok" 

144CAP_UNKNOWN_VERSION = "unknown_version" 

145CAP_UNAVAILABLE = "unavailable" 

146 

147# Stable, typed error taxonomy for failed agent executions. These codes let 

148# reports and CI/policy distinguish retryable from non-retryable failures 

149# instead of pattern-matching free-text error strings. 

150ERR_MISSING_CLI = "missing_cli" 

151ERR_AUTH_REQUIRED = "auth_required" 

152ERR_PERMISSION_PROMPT = "permission_prompt" 

153ERR_TIMEOUT = "timeout" 

154ERR_NONZERO_EXIT = "nonzero_exit" 

155ERR_EMPTY_OUTPUT = "empty_output" 

156ERR_SPAWN_FAILED = "spawn_failed" 

157ERR_RATE_LIMITED = "rate_limited" 

158# Local/HTTP adapter could not reach its server (issue #43): connection refused, 

159# DNS failure, or the local model server is not running. 

160ERR_CONNECTION = "connection_error" 

161# Hosted-API adapter (issue #430): the vendor's API key env var is unset. Distinct 

162# from ERR_AUTH_REQUIRED (a key was sent but the server rejected it) so a report 

163# can tell "never configured" apart from "misconfigured/expired/revoked". 

164ERR_MISSING_API_KEY = "missing_api_key" 

165# Hosted-API adapter (issue #430): the configured key contains a control 

166# character and was rejected BEFORE being sent as a header, rather than 

167# letting http.client raise (and risk echoing a transformed/escaped copy of 

168# the secret in its exception text — see _HostedApiAdapter._invalid_key_reason). 

169ERR_INVALID_API_KEY = "invalid_api_key" 

170# The agent exited 0 and printed something, but that something is not a review 

171# (issue #682): a refusal, or the CLI's own usage/argument-error/version banner 

172# echoed back because the invocation never reached the model. Distinct from 

173# ERR_EMPTY_OUTPUT (nothing at all on stdout) so a report can tell "the CLI is 

174# broken/misinvoked" apart from "the model declined", and distinct from ok=True 

175# so neither can be counted as a vendor that contributed to consensus. #635 is 

176# the shape: `agy` passed every availability probe, printed an argument error, 

177# and the panel silently became single-vendor. 

178ERR_NO_REVIEW = "no_review" 

179ERR_UNKNOWN = "unknown" 

180 

181ERROR_CODES = frozenset( 

182 { 

183 ERR_MISSING_CLI, 

184 ERR_AUTH_REQUIRED, 

185 ERR_PERMISSION_PROMPT, 

186 ERR_TIMEOUT, 

187 ERR_NONZERO_EXIT, 

188 ERR_EMPTY_OUTPUT, 

189 ERR_SPAWN_FAILED, 

190 ERR_RATE_LIMITED, 

191 ERR_CONNECTION, 

192 ERR_MISSING_API_KEY, 

193 ERR_INVALID_API_KEY, 

194 ERR_NO_REVIEW, 

195 ERR_UNKNOWN, 

196 } 

197) 

198 

199# Failures that are worth retrying because they are typically transient (issue 

200# #30): a timeout, a rate-limit, a process that failed to spawn, or a local 

201# server that was briefly unreachable (#43). Auth, missing-CLI, 

202# permission-prompt, empty-output, and generic nonzero-exit are treated as 

203# deterministic — retrying them just burns time and tokens. So is a 

204# no-review output: a misinvoked CLI prints the same usage banner every time. 

205RETRYABLE_ERROR_CODES = frozenset( 

206 { 

207 ERR_TIMEOUT, 

208 ERR_RATE_LIMITED, 

209 ERR_SPAWN_FAILED, 

210 ERR_CONNECTION, 

211 } 

212) 

213 

214 

215# Ordered keyword groups for classify_stderr. Each keyword is matched on word 

216# boundaries (\b...\b) so incidental substrings do NOT trigger a false 

217# classification: bare "auth" matches "auth error" but not "author identity", 

218# and "login" matches "login required" but not "login_attempts" ("_" is a word 

219# char, so there is no boundary inside "login_attempts"). Multi-word phrases 

220# tolerate a space OR "_" between tokens (e.g. "rate limit"/"rate_limit"). 

221def _keyword_pattern(*keywords: str) -> re.Pattern[str]: 

222 parts = [r"[ _]+".join(re.escape(tok) for tok in kw.split()) for kw in keywords] 

223 return re.compile(r"\b(?:" + "|".join(parts) + r")\b") 

224 

225 

226# Order matters: auth and rate-limit signals are checked before the generic 

227# permission and nonzero-exit fallbacks. 

228_AUTH_RE = _keyword_pattern( 

229 "not authenticated", 

230 "unauthenticated", 

231 "authentication", 

232 "unauthorized", 

233 "api key", 

234 "auth", 

235 "log in", 

236 "login", 

237 "credential", 

238 "credentials", 

239) 

240_RATE_LIMIT_RE = _keyword_pattern("rate limit", "429", "quota", "too many requests") 

241_PERMISSION_RE = _keyword_pattern( 

242 "permission", 

243 "permissions", 

244 "approve", 

245 "approval", 

246 "confirm", 

247 "confirmation", 

248) 

249 

250 

251def classify_stderr(returncode: int, stderr: str) -> str: 

252 """Classify a nonzero-exit failure into a typed error code from its stderr. 

253 

254 Token-aware matching against the lowercased stderr: each keyword group is a 

255 word-boundary regex, so incidental substrings (e.g. "author" containing 

256 "auth") never cause a misclassification. Ordering matters (auth and 

257 rate-limit signals are checked before the generic permission and 

258 nonzero-exit fallbacks). Returns one of the ``ERR_*`` codes. 

259 """ 

260 text = (stderr or "").lower() 

261 if _AUTH_RE.search(text): 

262 return ERR_AUTH_REQUIRED 

263 if _RATE_LIMIT_RE.search(text): 

264 return ERR_RATE_LIMITED 

265 if _PERMISSION_RE.search(text): 

266 return ERR_PERMISSION_PROMPT 

267 del returncode 

268 return ERR_NONZERO_EXIT 

269 

270 

271# --- "Exit 0, but nothing reviewable came back" (issue #682) ------------------ 

272# 

273# A CLI that exits 0 and prints its own usage text is indistinguishable, to 

274# everything downstream, from a reviewer that read the diff — the run stays 

275# fail-soft and the panel quietly loses a vendor (#635). These patterns turn 

276# that into a typed failure at the adapter boundary, on SHAPE alone: never on 

277# whether a review is any good, only on whether it is a review at all. 

278 

279#: An output longer than this is treated as a review even if it opens with a 

280#: refusal-shaped sentence. A real review that mentions "I cannot verify X" 

281#: mid-argument must never be discarded; an actual refusal is a short paragraph. 

282_NO_REVIEW_MAX_CHARS = 600 

283 

284#: How much of the output is examined for a usage/argument-error banner. A CLI 

285#: that is going to print one prints it first. 

286_NO_REVIEW_HEAD_CHARS = 400 

287 

288# The #635 class: the launcher rejected the argv, so the model was never 

289# reached. Every one of these is text a CLI writes about ITSELF — and the whole 

290# difficulty is that a review may *talk about* the same words. "Usage of int() 

291# is unsafe" and "Invalid argument passed to calculate_total()" are findings, 

292# not banners, and discarding either costs a panelist and (with the guard 

293# failing closed) can collapse the panel. So each pattern below is matched 

294# against a whole line, and every branch is bounded the way the refusal branch 

295# already is: a banner is short, or it is corroborated by banner structure. 

296 

297#: A launcher's own usage line, as a whole line. Prose that merely opens with 

298#: the word "usage" does not match: the line must go on to look like a synopsis 

299#: (a colon, or a bracketed/flag-shaped operand). 

300_USAGE_LINE_RE = re.compile( 

301 r"usage:\s*\S" # "Usage: agy [options] [prompt]" 

302 r"|usage\s+of\s+\S+:$" # Go's flag package: "Usage of ./agy:" 

303 r"|usage\s+\S+\s+[\[<-]", # "usage agy [options]" 

304 re.IGNORECASE, 

305) 

306 

307#: Text only a launcher writes about itself — it complains about the argv it 

308#: was handed, in the first person of a program. Needs no corroboration beyond 

309#: opening a short output. 

310_CLI_SELF_ERROR_RE = re.compile( 

311 r"(?:error|fatal)\s*:\s*(?:unknown|unrecognized|invalid|unexpected)\s+" 

312 r"(?:flag|option|argument|command|subcommand)\b" 

313 r"|flag needs an argument\b" 

314 r"|.{0,40}\bcommand not found$", 

315 re.IGNORECASE, 

316) 

317 

318#: The same complaint without the ``error:`` prefix. This one IS ambiguous with 

319#: review prose, so it must both name a flag-shaped token ("unknown flag: 

320#: --print") and be corroborated by surrounding banner structure. 

321_BARE_ARG_ERROR_RE = re.compile( 

322 r"(?:unknown|unrecognized|invalid|unexpected)\s+" 

323 r"(?:flag|option|argument|command|subcommand)\b" 

324 r"[\s:=]*[\"'`]?-{1,2}[A-Za-z0-9]", 

325 re.IGNORECASE, 

326) 

327 

328#: Corroborating structure, never a trigger on its own: the options/flags block 

329#: under a synopsis, or the help hint a launcher prints beneath it. The old 

330#: unanchored "for more information, try/see" pattern lives on only here — it 

331#: is a sentence a review can perfectly well contain. 

332_BANNER_STRUCTURE_RE = re.compile( 

333 r"^[ \t]*(?:options|flags|commands|arguments|subcommands)\b[ \t]*:?[ \t]*$" 

334 r"|^[ \t]*-{1,2}[A-Za-z0-9][\w-]*(?:[ \t=,]|$)" 

335 r"|^[ \t]*for more information,? (?:try|see)\b", 

336 re.IGNORECASE | re.MULTILINE, 

337) 

338 

339#: Nothing but a version/probe banner, e.g. "1.1.22" or "agy version 1.1.22". 

340_VERSION_ONLY_RE = re.compile(r"^[\w.@/+-]{0,40}(?:\s+version)?\s*v?\d+\.\d+[\w.+-]*$") 

341 

342#: A refusal, matched on the lowercased text and only inside the length bound 

343#: above. Deliberately narrow: these are first-person declines, not any 

344#: sentence containing the word "cannot". 

345#: 

346#: Matching this is NOT on its own enough to discard an output — see 

347#: :func:`_is_whole_output_refusal`. A reviewer routinely opens with a limit it 

348#: hit ("I do not have access to the migration file, so I reviewed the Python 

349#: only: …") and then reviews; discarding that costs a panelist and, with the 

350#: gate failing closed, turns a passing two-vendor run into exit 3. 

351_REFUSAL_RE = re.compile( 

352 r"i(?:'|\u2019)?m (?:sorry|unable|not able)" 

353 r"|i am (?:sorry|unable|not able)" 

354 r"|i (?:can(?:'|\u2019)?t|cannot|won(?:'|\u2019)?t|will not)" 

355 r" (?:help|assist|do|review|comply|provide|analyz|analys)" 

356 r"|sorry,? (?:but )?i " 

357 r"|as an ai\b" 

358 r"|i (?:do not|don(?:'|\u2019)?t) have (?:access|the ability)" 

359) 

360 

361 

362#: How far into the output a decline may begin and still BE the output. A 

363#: refusal says so first; a review that ran into a limit mid-argument says so 

364#: after paragraphs of review. One short lead-in sentence is allowed for, no more. 

365_REFUSAL_OPENING_CHARS = 80 

366 

367#: Marks of an output that reviewed something, however briefly. Any one of them 

368#: outranks the refusal phrase in front of it: an agent that says it could not 

369#: read one file and then names a defect HAS reviewed, and its finding is the 

370#: thing the panel exists to collect. 

371_REVIEW_SUBSTANCE_RE = re.compile( 

372 r"\blines?\s+\d+" # "line 12", "lines 40-52" 

373 r"|\b[\w./-]+\.[a-z]{1,5}:\d+" # "src/app.py:12" 

374 r"|^@@[ \t]" # a hunk header quoted back 

375 r"|\bi (?:reviewed|read|checked|examined|inspected|looked at|went through)\b" 

376 r"|\b(?:having|after) (?:reviewed|read|checked|examined)\b", 

377 re.IGNORECASE | re.MULTILINE, 

378) 

379 

380#: The adversative pivot that turns a limit into a review: "I'm not able to 

381#: reproduce the race locally, BUT the lock ordering here is wrong." It counts 

382#: only when what follows is prose of its own and is not itself another decline 

383#: — "I'm sorry, but I can't help with this" pivots into the same refusal. 

384_PIVOT_RE = re.compile(r"\b(?:but|however|though|although|that said)\b", re.IGNORECASE) 

385 

386#: How much text must follow a pivot before it can carry a review claim. 

387_PIVOT_MIN_CHARS = 20 

388 

389 

390def _carries_review_substance(body: str) -> bool: 

391 """True when ``body`` says something about the code, not only about itself.""" 

392 if _REVIEW_SUBSTANCE_RE.search(body): 

393 return True 

394 pivot = _PIVOT_RE.search(body) 

395 if pivot is None: 

396 return False 

397 rest = body[pivot.end() :].strip() 

398 return len(rest) >= _PIVOT_MIN_CHARS and _REFUSAL_RE.search(rest.lower()) is None 

399 

400 

401def _is_whole_output_refusal(body: str) -> bool: 

402 """True when the output *is* a decline, not a review that mentions a limit. 

403 

404 Three conditions, all necessary. The output is short (a real decline is a 

405 sentence or two), the decline OPENS it, and nothing in it carries review 

406 substance. Merely containing a refusal phrase somewhere inside the length 

407 bound was the old rule, and it destroyed genuine short reviews — including 

408 ones that named a defect and a line number (#682, round 3). 

409 """ 

410 if len(body) > _NO_REVIEW_MAX_CHARS: 

411 return False 

412 match = _REFUSAL_RE.search(body.lower()) 

413 if match is None or match.start() > _REFUSAL_OPENING_CHARS: 

414 return False 

415 return not _carries_review_substance(body) 

416 

417 

418def _is_cli_banner(body: str) -> bool: 

419 """True when ``body`` *is* a launcher's banner, not prose that mentions one. 

420 

421 Two shapes qualify. A full help dump — a synopsis line plus an options 

422 block — is unmistakable at any length. Anything shorter must both open with 

423 a banner-shaped line and fit inside the same length bound the refusal branch 

424 uses: a real review is not three lines. 

425 """ 

426 head = body[:_NO_REVIEW_HEAD_CHARS] 

427 lines = [line.strip() for line in head.splitlines() if line.strip()] 

428 if not lines: # pragma: no cover - `body` is stripped and non-empty here 

429 return False 

430 synopsis = any(_USAGE_LINE_RE.match(line) for line in lines) 

431 structure = _BANNER_STRUCTURE_RE.search(head) is not None 

432 if synopsis and structure: 

433 return True 

434 if len(body) > _NO_REVIEW_MAX_CHARS: 

435 return False 

436 first = lines[0] 

437 if _USAGE_LINE_RE.match(first) or _CLI_SELF_ERROR_RE.match(first): 

438 return True 

439 return _BARE_ARG_ERROR_RE.match(first) is not None and (synopsis or structure) 

440 

441 

442def no_review_reason(text: str) -> str | None: 

443 """Why ``text`` cannot be a review, or ``None`` when it could be one. 

444 

445 PURE, and a judgement of SHAPE only — it never scores a review's quality. 

446 An adapter calls it on an exit-0 stdout so that a refusal, a usage banner 

447 or a bare version string is recorded as a typed failure (``ERR_NO_REVIEW``) 

448 rather than as a contributing vendor. Fail-soft still applies: the run 

449 continues, but with a dead seat that says so. 

450 

451 Returns a short human-readable reason, safe to embed in an error string 

452 (it names the category, never the agent's text). 

453 """ 

454 body = (text or "").strip() 

455 if not body: 

456 return "the agent produced no output" 

457 # A structured findings block is the one machine-checkable proof that the 

458 # agent answered in the reviewer's contract, and it settles the question 

459 # before any prose heuristic gets a vote: a launcher banner does not emit 

460 # one, and an agent that declines and then reports a finding has reviewed. 

461 if emitted_findings_block(body): 

462 return None 

463 if _is_cli_banner(body): 

464 return "the CLI printed usage or argument-error text instead of a review" 

465 if _VERSION_ONLY_RE.match(body): 

466 return "the CLI printed only a version banner instead of a review" 

467 if _is_whole_output_refusal(body): 

468 return "the agent declined to review" 

469 return None 

470 

471 

472# --- Reasoning effort (issue #662) ------------------------------------------- 

473# 

474# Every vendor expresses "think harder" differently, so the mapping lives HERE, 

475# in one pure function, instead of being spread across the adapters: agy encodes 

476# effort as a model-id suffix, the Anthropic Messages API takes an extended- 

477# thinking token budget, OpenAI-shaped APIs take `reasoning_effort`, Gemini takes 

478# a `thinkingConfig` budget, and the `claude`/`codex` CLIs have no headless knob 

479# at all. `effort_args` is the single place any of that is decided. 

480 

481#: The effort levels accepted by ``[[agent]] effort`` and ``--effort``. 

482EFFORT_LEVELS: tuple[str, ...] = ("low", "medium", "high") 

483 

484# Anthropic extended-thinking budgets (tokens) per level, before clamping. 

485_ANTHROPIC_THINKING_BUDGET = {"low": 2048, "medium": 8192, "high": 32768} 

486# Ceiling on the `max_tokens` this project will ever request from Anthropic. 

487# Thinking tokens are drawn from the same allowance, so `max_tokens` has to 

488# exceed the budget — but per-model caps vary and an unbounded sum would build a 

489# request some models reject outright. 32000 sits inside every current model's 

490# limit, so the `high` budget is clamped to `ceiling - _HOSTED_API_MAX_TOKENS` 

491# (27904) rather than sending its nominal 32768. Documented in 

492# docs/configuration.md. 

493_ANTHROPIC_MAX_TOKENS_CEILING = 32000 

494# Gemini `thinkingConfig.thinkingBudget` (tokens) per level. 

495_GEMINI_THINKING_BUDGET = {"low": 1024, "medium": 8192, "high": 32768} 

496# agy model-id suffixes that already carry an effort level. 

497_AGY_MODEL_SUFFIXES = tuple(f"-{level}" for level in EFFORT_LEVELS) 

498 

499# Vendors that can actually act on an effort level. Everything else (the 

500# `claude`/`codex` CLIs, `local`, `cli`, any unregistered vendor) warns once and 

501# ignores it: a local OpenAI-compatible server is NOT included, because many of 

502# them reject an unknown request field outright, which would turn a hint into a 

503# failed review. 

504_EFFORT_VENDORS = frozenset( 

505 {"google", "anthropic-api", "openai-api", "xai-api", "openai-compatible", "google-api"} 

506) 

507 

508 

509@dataclass(frozen=True) 

510class EffortPlan: 

511 """How one vendor expresses a reasoning-effort level. 

512 

513 ``model`` is a replacement model id (agy, which encodes effort in the id); 

514 ``payload`` is a request-body fragment to merge into the vendor's JSON body; 

515 ``warning`` is the once-per-run operator message when the level is ignored. 

516 """ 

517 

518 supported: bool = True 

519 model: str | None = None 

520 payload: dict = field(default_factory=dict) 

521 warning: str | None = None 

522 

523 

524def effort_supported(vendor: str) -> bool: 

525 """Whether *vendor* can act on an effort level at all (pure).""" 

526 return config_module.normalise_vendor(vendor) in _EFFORT_VENDORS 

527 

528 

529def _anthropic_budget(level: str) -> int: 

530 """Thinking budget for *level*, clamped to the documented ceiling (pure).""" 

531 return min( 

532 _ANTHROPIC_THINKING_BUDGET[level], 

533 _ANTHROPIC_MAX_TOKENS_CEILING - _HOSTED_API_MAX_TOKENS, 

534 ) 

535 

536 

537def _effort_uses_model_listing(vendor: str) -> bool: 

538 """Whether *vendor* expresses effort through the model id (pure). 

539 

540 Those are the vendors whose mapped id can be checked against what the CLI 

541 actually offers, so callers know when discovering a listing is worth a probe. 

542 """ 

543 return config_module.normalise_vendor(vendor) == "google" 

544 

545 

546def effort_args( 

547 vendor: str, 

548 effort: str | None, 

549 model: str | None = None, 

550 known_models: list[str] | None = None, 

551) -> EffortPlan: 

552 """Map ``(vendor, effort, model)`` to the vendor's own effort knob (pure). 

553 

554 An empty/None *effort* is a no-op plan. An unrecognized level raises 

555 ``ValueError`` — ``config.validate_config`` and the ``--effort`` choices 

556 already gate user input, so reaching here with garbage is a programming 

557 error, not something to silently swallow. 

558 

559 ``known_models``, when the caller has discovered one, is the model listing 

560 the vendor actually offers. It is only consulted where effort is expressed 

561 *as* a model id (agy): sending a suffixed id the CLI does not have would 

562 fail the whole review, so an unlisted mapping falls back to the configured 

563 id and warns instead. ``None`` means "no listing available" — the check is 

564 skipped, never guessed. 

565 """ 

566 level = (effort or "").strip().lower() 

567 if not level: 

568 return EffortPlan() 

569 if level not in EFFORT_LEVELS: 

570 raise ValueError(f"unknown effort {effort!r}; expected one of {', '.join(EFFORT_LEVELS)}") 

571 

572 name = config_module.normalise_vendor(vendor) 

573 

574 if name == "google": 

575 # The `agy` CLI selects effort through the model id itself. 

576 current = (model or "").strip() 

577 if not current: 

578 return EffortPlan( 

579 warning=( 

580 f"effort '{level}' needs a configured model for vendor '{vendor}' " 

581 f"(effort is encoded in the model id), ignored" 

582 ) 

583 ) 

584 if current.endswith(_AGY_MODEL_SUFFIXES): 

585 # Already pinned by the operator; an explicit model id wins. 

586 return EffortPlan(model=current) 

587 suffixed = f"{current}-{level}" 

588 if known_models is not None and suffixed not in known_models: 

589 # Better a review at the configured depth than no review at all. 

590 return EffortPlan( 

591 model=current, 

592 warning=( 

593 f"effort '{level}' maps model '{current}' to '{suffixed}', which " 

594 f"vendor '{vendor}' does not offer; using '{current}' unchanged" 

595 ), 

596 ) 

597 return EffortPlan(model=suffixed) 

598 

599 if name == "anthropic-api": 

600 return EffortPlan( 

601 payload={ 

602 "thinking": { 

603 "type": "enabled", 

604 "budget_tokens": _anthropic_budget(level), 

605 } 

606 } 

607 ) 

608 

609 if name in ("openai-api", "xai-api", "openai-compatible"): 

610 return EffortPlan(payload={"reasoning_effort": level}) 

611 

612 if name == "google-api": 

613 return EffortPlan( 

614 payload={ 

615 "generationConfig": { 

616 "thinkingConfig": {"thinkingBudget": _GEMINI_THINKING_BUDGET[level]} 

617 } 

618 } 

619 ) 

620 

621 return EffortPlan(supported=False, warning=f"effort unsupported for {vendor}, ignored") 

622 

623 

624def effort_warnings(agents, adapter_factory=None) -> list[str]: 

625 """Deduped, ordered effort warnings for a panel — one message per run. 

626 

627 Callers (the CLI) print these once before the run rather than once per agent 

628 invocation, so a three-round panel does not repeat the same line nine times. 

629 

630 Pure unless *adapter_factory* is given. With it, an agent whose effort is 

631 expressed as a model id has its mapped id checked against the vendor's real 

632 listing, so "that model does not exist" is reported up front instead of 

633 surfacing as an abstention mid-run. The probe is skipped entirely for agents 

634 with no effort configured and for every other vendor, and any failure 

635 degrades to "no listing" rather than blocking the run. 

636 """ 

637 seen: set[str] = set() 

638 out: list[str] = [] 

639 for spec in agents: 

640 # The adapter, not the vendor (#705): how effort is expressed — a model 

641 # suffix, a request-body field, nothing at all — is a property of the 

642 # protocol the seat is invoked through. 

643 vendor = config_module.spec_adapter(spec) 

644 effort = getattr(spec, "effort", None) 

645 known_models = None 

646 if adapter_factory is not None and effort and _effort_uses_model_listing(vendor): 

647 try: 

648 known_models = adapter_factory(spec).list_models() 

649 except Exception: # noqa: BLE001 - discovery is best-effort 

650 known_models = None 

651 try: 

652 plan = effort_args( 

653 vendor, 

654 effort, 

655 getattr(spec, "model", None), 

656 known_models=known_models, 

657 ) 

658 except ValueError as exc: 

659 # Redact before it reaches a warning/log, like every other str(exc) in this 

660 # module — an effort-args failure can echo attacker/config-influenced text (#828). 

661 message = redaction.redact(str(exc))[0] 

662 else: 

663 message = plan.warning 

664 if message and message not in seen: 

665 seen.add(message) 

666 out.append(message) 

667 return out 

668 

669 

670# Model ids look like `gemini-3.8-flash`, `qwen2.5-coder:7b`, `anthropic/claude-x`. 

671_MODEL_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/-]*$") 

672# Cap a discovered model listing so a chatty/hostile CLI cannot flood diagnostics. 

673_MAX_LISTED_MODELS = 50 

674 

675 

676def parse_model_list(raw: str) -> list[str]: 

677 """Model ids out of a CLI's model listing (pure, best-effort, order-preserving). 

678 

679 Accepts either JSON (a list, or an object with ``models``/``data``) or plain 

680 lines, because a CLI's listing format is not a contract. Anything that does 

681 not look like a model id is dropped rather than guessed at. 

682 """ 

683 text = (raw or "").strip() 

684 if not text: 

685 return [] 

686 

687 ids: list[str] = [] 

688 try: 

689 data = json.loads(text) 

690 except ValueError: 

691 data = None 

692 if isinstance(data, dict): 

693 data = data.get("models") or data.get("data") 

694 if isinstance(data, list): 

695 for item in data: 

696 if isinstance(item, str): 

697 ids.append(item) 

698 elif isinstance(item, dict): 

699 value = item.get("id") or item.get("name") or item.get("model") 

700 if isinstance(value, str): 

701 ids.append(value) 

702 else: 

703 for line in text.splitlines(): 

704 token = line.strip().lstrip("-*\u2022").strip() 

705 token = token.split()[0] if token else "" 

706 if _MODEL_ID_RE.match(token): 

707 ids.append(token) 

708 

709 seen: set[str] = set() 

710 out: list[str] = [] 

711 for value in ids: 

712 value = value.strip() 

713 if value and value not in seen: 

714 seen.add(value) 

715 out.append(value) 

716 return out[:_MAX_LISTED_MODELS] 

717 

718 

719@dataclass 

720class AgentResult: 

721 agent: str 

722 vendor: str 

723 ok: bool 

724 output: str 

725 duration_s: float 

726 error: str | None = None 

727 findings: list = field(default_factory=list) 

728 warnings: list = field(default_factory=list) 

729 error_code: str | None = None 

730 # Number of attempts made for this result (issue #30): 1 means no retry. 

731 # >1 records that a transient failure was retried before this outcome. 

732 attempts: int = 1 

733 # Did this agent emit a structured findings block at all (issue #501)? A 

734 # reviewer that examined the diff and found nothing still emits `[]`; one that 

735 # produced no review emits prose and no block. Without this, both arrive as zero 

736 # findings and the panel reports the same size either way. 

737 structured: bool = False 

738 # The agent process's own exit status (issue #661), or None when there was 

739 # no process to exit: a network adapter, a CLI that was never spawned 

740 # (missing on PATH, spawn failure), or one killed on timeout. `jury 

741 # run-agent` reports it so an orchestrator can act on the real status 

742 # instead of parsing it back out of the error string. 

743 exit_code: int | None = None 

744 # The model id this invocation actually sent (issue #709), stamped by the 

745 # path that ran the adapter from :meth:`Adapter.resolved_model` — the same 

746 # call that put the id in the argv/payload. Empty when nothing was pinned 

747 # (the CLI chose) or when the record was not produced by an invocation. 

748 # A ballot READS this rather than re-deriving it: `vendor` and `adapter` can 

749 # differ since #705, and a second derivation is a second answer. 

750 model: str = "" 

751 

752 

753class Adapter: 

754 """Base adapter. Subclasses build the argv for their CLI.""" 

755 

756 # Declarative capability metadata. Real coding-agent CLIs support a headless 

757 # (non-interactive) invocation and model selection; subclasses override where 

758 # this differs. ``MockAdapter`` reports synthetic capabilities. 

759 SUPPORTS_HEADLESS = True 

760 SUPPORTS_MODEL_SELECTION = True 

761 

762 #: Start every READ-ONLY invocation — the panel's review, debate, verify and 

763 #: synthesis calls, and ``jury run-agent``'s review/gate/chair roles — in a 

764 #: fresh empty directory rather than the one ``jury`` runs in (see 

765 #: :func:`_review_workdir`). A read-only role reads its prompt and nothing 

766 #: else, so it never needs the repository, and a PR checkout's project 

767 #: settings (a ``.claude/settings.json`` hook ran from one, measured) must not 

768 #: reach it. Only a write role (``implement``/``fix`` with ``--allow-write``) 

769 #: stays in ``--cwd``: it has to edit that worktree. On for the three native 

770 #: CLIs, whose behaviour there is known — codex needs 

771 #: ``--skip-git-repo-check`` outside a git repository, and its read-only argv 

772 #: carries it. Off by default, so a bring-your-own ``cli`` seat and a 

773 #: registered custom adapter keep running where they always did: this tool 

774 #: cannot know what an operator's own binary expects of its directory. 

775 ISOLATE_REVIEW_CWD = False 

776 

777 #: Whether this adapter runs its agent as a subprocess — the question the 

778 #: least-privilege audit asks before it looks for a sandbox. ``True`` unless a 

779 #: class says otherwise: the HTTP adapters (``LocalAdapter`` and the 

780 #: hosted-API ones) set ``False``, and so can a custom adapter that calls its 

781 #: backend over the network from ``run()`` (#903). ``register_adapter`` reads 

782 #: it, and only an explicit ``False`` opts out; a registered class that spawns 

783 #: a CLI and sets ``False`` would escape the audit, so the default is the safe one. 

784 SPAWNS_PROCESS = True 

785 

786 # Args passed to the CLI to print its version. Subclasses override if the CLI 

787 # uses a different verb/flag (e.g. ``codex --version``). 

788 _VERSION_ARGS = ("--version",) 

789 

790 def __init__(self, spec: AgentSpec): 

791 self.spec = spec 

792 

793 @property 

794 def name(self) -> str: 

795 return self.spec.name 

796 

797 def available(self) -> bool: 

798 return shutil.which(self.spec.command) is not None 

799 

800 def build_argv(self, prompt: str) -> list[str]: # pragma: no cover - overridden 

801 raise NotImplementedError 

802 

803 def build_write_argv(self, prompt: str) -> list[str]: 

804 """Argv for a WRITE-capable invocation of this CLI (issue #661). 

805 

806 The default is the read-only argv: a vendor with no tool-enabled mode 

807 (and every network adapter, which has no tools at all) simply cannot 

808 widen, and failing closed is the right default. The three native CLI 

809 adapters override it. 

810 """ 

811 return self.build_argv(prompt) 

812 

813 def build_argv_for_role(self, prompt: str, policy=None) -> list[str]: 

814 """Argv for one ``jury run-agent`` role (issue #661). 

815 

816 The single seam between the role policy and an adapter's argv, so the 

817 policy is never re-decided inside :meth:`run`. ``policy=None`` — every 

818 panel invocation — resolves to the unchanged read-only :meth:`build_argv`, 

819 which is why adding this could not alter an existing run. 

820 

821 Both branches end in :meth:`build_argv`, which the base class leaves 

822 ``NotImplementedError``. That is not reachable through a real run: every 

823 adapter that spawns a subprocess implements it, and the ones that do not 

824 (``LocalAdapter`` and the hosted-API adapters) build no argv at all — 

825 they override :meth:`run` and never call this. A new subprocess adapter 

826 that forgets ``build_argv`` therefore fails loudly on its first 

827 invocation rather than running with an empty command line. 

828 """ 

829 if policy is None or not getattr(policy, "write", False): 

830 return self.build_argv(prompt) 

831 return self.build_write_argv(prompt) 

832 

833 def _stdin_for(self, prompt: str) -> str | None: 

834 """Prompt to feed on stdin, or None to pass it in argv (the default).""" 

835 del prompt 

836 return None 

837 

838 def _text_from_stdout(self, raw: str) -> str: 

839 """The reviewer's prose, given the CLI's raw stdout. 

840 

841 Plain text for every adapter but one. `agy` speaks NDJSON events, so it 

842 overrides this rather than teaching `run` about event streams. 

843 """ 

844 return raw 

845 

846 def _version_argv(self) -> list[str]: 

847 """Argv used to probe the CLI's version.""" 

848 return [self.spec.command, *self._VERSION_ARGS] 

849 

850 def effort_plan(self) -> EffortPlan: 

851 """This agent's resolved effort mapping (see :func:`effort_args`). 

852 

853 An invalid level degrades to a no-op plan here: a run must not crash on 

854 a bad config value that ``validate_config`` is responsible for rejecting. 

855 """ 

856 try: 

857 return effort_args( 

858 config_module.spec_adapter(self.spec), 

859 getattr(self.spec, "effort", None), 

860 self.spec.model, 

861 ) 

862 except ValueError: 

863 return EffortPlan() 

864 

865 def resolved_model(self) -> str: 

866 """The model id this adapter sends, byte-for-byte (issue #709). 

867 

868 **The** answer to "which model was asked for", and the only one: every 

869 place a model id leaves this module — the ``--model``/``-m`` argv of the 

870 three CLI adapters, the ``model`` field of every network payload, the 

871 Gemini URL — reads it from here, and :attr:`AgentResult.model` records 

872 what it returned so a ballot can quote the id instead of deriving a 

873 second one. 

874 

875 That second derivation was the defect. :func:`effort_args` is keyed on 

876 the **adapter**, because how effort is expressed is a property of the 

877 protocol a seat is invoked through; ``ai_jury.ballots.requested_model`` 

878 keyed it on the ``vendor``, and since #705 those can differ, so a seat 

879 with ``vendor = google, adapter = cli`` was invoked with ``gemini-3-pro`` 

880 while its ballot reported ``gemini-3-pro-high`` under 

881 ``model_source: requested`` — a field whose whole claim is that it is the 

882 id actually sent. The subclass that consults a live model listing 

883 (:class:`AgyAdapter`) overrides :meth:`effort_plan`, not this, so the 

884 fallback it resolves is recorded here too. 

885 """ 

886 return (self.effort_plan().model or self.spec.model or "").strip() 

887 

888 def list_models(self) -> list[str] | None: 

889 """Model ids this agent could be pointed at, or None when unknown. 

890 

891 Diagnostics only (``jury --doctor --json``). The default is None — most 

892 CLIs have no listing command — and every override is time-boxed and 

893 fail-soft, because doctor must never hang or crash on a probe. 

894 """ 

895 return None 

896 

897 def detect_capabilities(self) -> dict: 

898 """Best-effort probe of this agent's version and capabilities. 

899 

900 Returns a dict shaped like:: 

901 

902 { 

903 "version": "<str|None>", 

904 "supports_headless": bool, 

905 "supports_model_selection": bool, 

906 "raw_version_output": "<short str>", 

907 "status": "ok|unknown_version|unavailable", 

908 "warnings": [...], 

909 } 

910 

911 This is intentionally fast and forgiving: it runs ``<command> --version`` 

912 with a SHORT timeout and swallows ALL errors (missing CLI, timeout, 

913 nonzero exit, garbage output). It NEVER raises, so it is safe to call 

914 from diagnostics without blocking or crashing a run. 

915 """ 

916 caps = { 

917 "version": None, 

918 "supports_headless": self.SUPPORTS_HEADLESS, 

919 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION, 

920 "raw_version_output": "", 

921 "status": CAP_UNAVAILABLE, 

922 "warnings": [], 

923 } 

924 

925 # Not on PATH: report unavailable without spawning a subprocess. 

926 if not self.available(): 

927 return caps 

928 

929 try: 

930 # Via _spawn so the probe also runs in its own process group and the 

931 # whole group is killed on timeout (issue #303/L-1) — matching the 

932 # main run path; a bare subprocess.run would orphan grandchildren. 

933 proc = _spawn(self._version_argv(), None, _VERSION_PROBE_TIMEOUT) 

934 except subprocess.TimeoutExpired: 

935 caps["status"] = CAP_UNKNOWN_VERSION 

936 caps["warnings"].append( 

937 f"version probe for '{self.spec.command}' timed out after {_VERSION_PROBE_TIMEOUT}s" 

938 ) 

939 return caps 

940 except Exception as exc: # noqa: BLE001 - swallow any spawn failure 

941 caps["status"] = CAP_UNKNOWN_VERSION 

942 caps["warnings"].append( 

943 f"version probe for '{self.spec.command}' failed: {redaction.redact(str(exc))[0]}" 

944 ) 

945 return caps 

946 

947 raw = ((proc.stdout or "") + (proc.stderr or "")).strip() 

948 caps["raw_version_output"] = redaction.redact(raw[:200])[0] 

949 match = _VERSION_RE.search(raw) 

950 if proc.returncode == 0 and match: 

951 caps["version"] = match.group(0) 

952 caps["status"] = CAP_OK 

953 else: 

954 caps["status"] = CAP_UNKNOWN_VERSION 

955 caps["warnings"].append( 

956 f"could not determine version of '{self.spec.command}' " 

957 f"(exit {proc.returncode}); capabilities assumed from vendor defaults" 

958 ) 

959 return caps 

960 

961 def run( 

962 self, 

963 prompt: str, 

964 phase: str = "review", 

965 timeout: int | None = None, 

966 role_policy=None, 

967 ) -> AgentResult: 

968 del phase 

969 if not self.available(): 

970 return AgentResult( 

971 self.name, 

972 self.spec.vendor, 

973 False, 

974 "", 

975 0.0, 

976 f"command not found on PATH: {self.spec.command}", 

977 error_code=ERR_MISSING_CLI, 

978 ) 

979 # The effective timeout is the caller's override (the run budget, issue 

980 # #30) when smaller than the agent's own bound, else the agent timeout. 

981 effective_timeout = self.spec.timeout 

982 if timeout is not None: 

983 effective_timeout = max(1, min(self.spec.timeout, int(timeout))) 

984 argv = self.build_argv_for_role(prompt, role_policy) 

985 stdin = self._stdin_for(prompt) 

986 start = time.monotonic() 

987 try: 

988 # A command is a bare name or an absolute path (config validation 

989 # refuses a relative one, #293/F-6), so moving the directory cannot 

990 # change which binary runs. 

991 read_only = not getattr(role_policy, "write", False) 

992 with _review_workdir(self.ISOLATE_REVIEW_CWD and read_only) as cwd: 

993 if cwd is None: 

994 proc = _spawn(argv, stdin, effective_timeout) 

995 else: 

996 proc = _spawn(argv, stdin, effective_timeout, cwd=cwd) 

997 except subprocess.TimeoutExpired: 

998 return AgentResult( 

999 self.name, 

1000 self.spec.vendor, 

1001 False, 

1002 "", 

1003 time.monotonic() - start, 

1004 f"timed out after {effective_timeout}s", 

1005 error_code=ERR_TIMEOUT, 

1006 ) 

1007 except Exception as exc: # noqa: BLE001 - surface any spawn failure 

1008 return AgentResult( 

1009 self.name, 

1010 self.spec.vendor, 

1011 False, 

1012 "", 

1013 time.monotonic() - start, 

1014 f"spawn failed: {redaction.redact(str(exc))[0]}", 

1015 error_code=ERR_SPAWN_FAILED, 

1016 ) 

1017 dur = time.monotonic() - start 

1018 out = self._text_from_stdout((proc.stdout or "").strip()).strip() 

1019 # A nonzero exit is ALWAYS a failure, even with stdout (issue #101): a 

1020 # crashing CLI can still print partial or error output, and counting that 

1021 # as a clean review would silently feed it into consensus, synthesis, and 

1022 # the CI gate. We classify from stderr (falling back to any stdout) and 

1023 # keep a short snippet in the error for debugging — but ok=False, so the 

1024 # orchestrator excludes it. 

1025 if proc.returncode != 0: 

1026 stderr = (proc.stderr or "").strip() 

1027 detail = stderr or out 

1028 # Redact before embedding in the error: a crashing CLI can dump an 

1029 # env var / token into its stderr, and this string is rendered into 

1030 # the report and posted to the PR. Mirrors the LocalAdapter path 

1031 # (#293/F-8); the asymmetry was a secret-leak vector (audit 

1032 # 2026-06-13/N-1). Classify on the raw text (no secrets in codes). 

1033 safe_detail = redaction.redact(detail)[0] 

1034 return AgentResult( 

1035 self.name, 

1036 self.spec.vendor, 

1037 False, 

1038 "", 

1039 dur, 

1040 f"exit {proc.returncode}: {safe_detail[:500]}", 

1041 error_code=classify_stderr(proc.returncode, stderr or out), 

1042 exit_code=proc.returncode, 

1043 ) 

1044 if not out: 

1045 # Exit 0 but nothing on stdout: the agent produced no usable review. 

1046 return AgentResult( 

1047 self.name, 

1048 self.spec.vendor, 

1049 False, 

1050 "", 

1051 dur, 

1052 f"exit {proc.returncode}: empty output", 

1053 error_code=ERR_EMPTY_OUTPUT, 

1054 exit_code=proc.returncode, 

1055 ) 

1056 # Exit 0 with output that is not a review (issue #682): a refusal, or the 

1057 # CLI's own usage/version banner because the argv never reached the model 

1058 # (#635). Counting it as a review is what makes a panel collapse silent. 

1059 no_review = no_review_reason(out) 

1060 if no_review is not None: 

1061 return AgentResult( 

1062 self.name, 

1063 self.spec.vendor, 

1064 False, 

1065 "", 

1066 dur, 

1067 f"no review returned: {no_review}: {redaction.redact(out)[0][:200]}", 

1068 error_code=ERR_NO_REVIEW, 

1069 exit_code=proc.returncode, 

1070 ) 

1071 return AgentResult(self.name, self.spec.vendor, True, out, dur, exit_code=proc.returncode) 

1072 

1073 

1074class ClaudeAdapter(Adapter): 

1075 # The prompt embeds the (redacted) diff and PR/issue context; deliver it on 

1076 # STDIN rather than as a process argument so it is not exposed in `ps` / 

1077 # /proc/<pid>/cmdline to other local users (issue #287). `claude -p` reads 

1078 # the prompt from stdin when no positional prompt is given. 

1079 ISOLATE_REVIEW_CWD = True 

1080 

1081 def _head_argv(self) -> list[str]: 

1082 argv = [self.spec.command, "-p"] 

1083 model = self.resolved_model() 

1084 if model: 

1085 argv += ["--model", model] 

1086 return argv 

1087 

1088 def build_argv(self, prompt: str) -> list[str]: 

1089 del prompt 

1090 return self._head_argv() + _read_only_extra_args(self.spec) 

1091 

1092 def build_write_argv(self, prompt: str) -> list[str]: 

1093 """Implementer invocation: the CLI's own tool set, no deny list (#661).""" 

1094 del prompt 

1095 return self._head_argv() + _write_extra_args(self.spec) 

1096 

1097 def _stdin_for(self, prompt: str) -> str | None: 

1098 return prompt 

1099 

1100 

1101class CodexAdapter(Adapter): 

1102 # Pipe the prompt on stdin (not positionally) so ``codex exec`` never blocks 

1103 # waiting for input in non-interactive runs. Sandbox flags live in extra_args; 

1104 # the shipped default is ``-s read-only`` (secure by default, #100) — the 

1105 # reviewer only reads its prompt, since the jury fetches the diff via ``gh``. 

1106 ISOLATE_REVIEW_CWD = True 

1107 

1108 #: `codex exec` refuses to start outside a git repository ("Not inside a 

1109 #: trusted directory and --skip-git-repo-check was not specified", verified 

1110 #: against codex-cli 0.155.0), and a panel invocation starts in an empty 

1111 #: temporary directory. The check keeps an agent out of a directory nobody 

1112 #: vouched for; a reviewer is kept in check by its read-only sandbox instead, 

1113 #: which ``privilege.audit_agent`` reports on, and the directory is empty. 

1114 _SKIP_GIT_CHECK = "--skip-git-repo-check" 

1115 

1116 def _head_argv(self) -> list[str]: 

1117 argv = [self.spec.command, "exec"] 

1118 model = self.resolved_model() 

1119 if model: 

1120 argv += ["-m", model] 

1121 return argv 

1122 

1123 #: ``codex exec --ephemeral`` ("Run without persisting session files to 

1124 #: disk", codex-cli 0.155.0): a reviewer's session holds the untrusted diff 

1125 #: and is never resumed. 

1126 _EPHEMERAL = "--ephemeral" 

1127 

1128 def build_argv(self, prompt: str) -> list[str]: 

1129 del prompt 

1130 extra = _read_only_extra_args(self.spec) 

1131 # Only an option counts as present: after `--` it is prompt text (#908 review). 

1132 options = privilege._codex_options(extra) 

1133 added = [f for f in (self._SKIP_GIT_CHECK, self._EPHEMERAL) if f not in options] 

1134 return self._head_argv() + added + extra 

1135 

1136 def build_write_argv(self, prompt: str) -> list[str]: 

1137 """Implementer invocation: ``-s workspace-write`` instead of read-only (#661).""" 

1138 del prompt 

1139 return self._head_argv() + _write_extra_args(self.spec) 

1140 

1141 def _stdin_for(self, prompt: str) -> str | None: 

1142 return prompt 

1143 

1144 

1145class AgyAdapter(Adapter): 

1146 # Prompt on STDIN, not argv (issue #287): the redacted diff must not appear in 

1147 # the process list, where any local user can read it. 

1148 # 

1149 # `agy --print` used to read the prompt from stdin (verified against 1.0.6). 

1150 # On 1.1.x `--print` takes a value, so that invocation dies before the model 

1151 # is reached — `flag needs an argument: -print` with an empty `extra_args`, 

1152 # and "took --dangerously-skip-permissions as its prompt" with a non-empty 

1153 # one (#635). The agent passed every availability check and contributed 

1154 # nothing to the panel. 

1155 # 

1156 # The obvious repair — put the prompt in argv — silently undoes #287. So the 

1157 # prompt moves to agy's own stdin channel instead: `--input-format 

1158 # stream-json` reads one NDJSON message per line and requires 

1159 # `--output-format stream-json`. `--print` is not passed at all; the input 

1160 # format implies print mode, and passing it would reintroduce the arity 

1161 # problem. Verified end to end against agy 1.1.22. 

1162 _STREAM_ARGS = ("--input-format", "stream-json", "--output-format", "stream-json") 

1163 

1164 # A panel review starts in an empty temporary directory; agy 1.2.9 answers 

1165 # from one (checked by hand against the installed CLI). 

1166 ISOLATE_REVIEW_CWD = True 

1167 

1168 # `agy models` lists the model ids the CLI can be pointed at. 

1169 _MODELS_ARGS = ("models",) 

1170 

1171 def effort_plan(self) -> EffortPlan: 

1172 """agy's effort IS the model id, so check the mapping against its listing. 

1173 

1174 The listing is probed at most once per adapter instance, and only when an 

1175 effort level is actually configured — a run without one pays nothing, 

1176 which matters because this is on the per-invocation path. 

1177 """ 

1178 if not getattr(self.spec, "effort", None): 

1179 return EffortPlan() 

1180 try: 

1181 return effort_args( 

1182 config_module.spec_adapter(self.spec), 

1183 self.spec.effort, 

1184 self.spec.model, 

1185 known_models=self._cached_model_listing(), 

1186 ) 

1187 except ValueError: 

1188 return EffortPlan() 

1189 

1190 def _cached_model_listing(self) -> list[str] | None: 

1191 """``agy models``, memoized per adapter instance.""" 

1192 if not hasattr(self, "_model_listing"): 

1193 self._model_listing = self.list_models() 

1194 return self._model_listing 

1195 

1196 def _head_argv(self) -> list[str]: 

1197 argv = [self.spec.command, *self._STREAM_ARGS] 

1198 # agy encodes reasoning effort in the model id (`…-flash` -> `…-flash-high`), 

1199 # so effort changes WHICH model is selected rather than adding a flag. 

1200 model = self.resolved_model() 

1201 if model: 

1202 argv += ["--model", model] 

1203 return argv 

1204 

1205 def build_argv(self, prompt: str) -> list[str]: 

1206 del prompt 

1207 return self._head_argv() + _read_only_extra_args(self.spec) 

1208 

1209 def build_write_argv(self, prompt: str) -> list[str]: 

1210 """Implementer invocation: drop the boolean ``--sandbox`` (#661).""" 

1211 del prompt 

1212 return self._head_argv() + _write_extra_args(self.spec) 

1213 

1214 def list_models(self) -> list[str] | None: 

1215 """Model ids from ``agy models`` (time-boxed, fail-soft; issue #662).""" 

1216 if not self.available(): 

1217 return None 

1218 try: 

1219 proc = _spawn([self.spec.command, *self._MODELS_ARGS], None, _VERSION_PROBE_TIMEOUT) 

1220 except Exception: # noqa: BLE001 - discovery is best-effort 

1221 return None 

1222 if proc.returncode != 0: 

1223 return None 

1224 return parse_model_list(proc.stdout or "") or None 

1225 

1226 def _stdin_for(self, prompt: str) -> str | None: 

1227 """One NDJSON frame carrying the prompt. Shape verified against 1.1.22.""" 

1228 return json.dumps({"event": "user", "message": {"role": "user", "content": prompt}}) + "\n" 

1229 

1230 def _text_from_stdout(self, raw: str) -> str: 

1231 """The `result` event's response, out of the NDJSON stream. 

1232 

1233 Falls back to the raw stream when no `result` frame is present rather 

1234 than returning empty: a truncated stream should surface as an 

1235 unparseable review the operator can read, not as a silent abstention — 

1236 an empty review is counted as one, and #625 exists because an 

1237 abstention read as an approval is the expensive failure. 

1238 """ 

1239 response, saw_result = None, False 

1240 for line in raw.splitlines(): 

1241 line = line.strip() 

1242 if not line: 

1243 continue 

1244 try: 

1245 event = json.loads(line) 

1246 except ValueError: 

1247 continue 

1248 if isinstance(event, dict) and event.get("event") == "result": 

1249 saw_result = True 

1250 result = event.get("result") 

1251 if isinstance(result, dict): 1251 ↛ 1240line 1251 didn't jump to line 1240 because the condition on line 1251 was always true

1252 response = result.get("response") 

1253 if saw_result and isinstance(response, str): 

1254 return response 

1255 return raw 

1256 

1257 

1258_DEFAULT_LOCAL_ENDPOINT = "http://localhost:11434/v1" 

1259 

1260 

1261def _http_only_opener(): 

1262 """An opener that handles ONLY http/https (issue #291, SSRF defense). 

1263 

1264 The default ``urllib`` opener honors ``file://`` and ``ftp://``, so an 

1265 attacker-influenced ``endpoint`` could read local files or reach other 

1266 schemes. This OpenerDirector registers no ``FileHandler``/``FTPHandler``, so 

1267 any non-http(s) URL raises ``URLError("unknown url type")`` regardless of 

1268 config validation — defense in depth alongside ``config._endpoint_issues``. 

1269 

1270 It also registers NO ``HTTPRedirectHandler`` (review of #291): otherwise a 

1271 malicious/compromised endpoint could 302-redirect to an internal/metadata 

1272 host (e.g. ``169.254.169.254``) and the opener would follow it, bypassing the 

1273 configured-URL validation. Without the handler a 3xx surfaces as an 

1274 ``HTTPError`` (a failed review) and is never followed. 

1275 """ 

1276 import urllib.request 

1277 

1278 opener = urllib.request.OpenerDirector() 

1279 for handler in ( 

1280 urllib.request.HTTPHandler, 

1281 urllib.request.HTTPSHandler, 

1282 urllib.request.HTTPDefaultErrorHandler, 

1283 urllib.request.HTTPErrorProcessor, 

1284 # UnknownHandler raises URLError("unknown url type: …") for any scheme 

1285 # without a registered handler — so file://, ftp://, etc. fail loudly 

1286 # instead of silently resolving to None. 

1287 urllib.request.UnknownHandler, 

1288 ): 

1289 opener.add_handler(handler()) 

1290 return opener 

1291 

1292 

1293def _open(target, timeout): 

1294 """Open an http/https URL or Request via the restricted opener (issue #291). 

1295 

1296 Single seam for every local-adapter HTTP call so the SSRF-safe opener (no 

1297 file/ftp handlers) is always used. 

1298 """ 

1299 return _http_only_opener().open(target, timeout=timeout) 

1300 

1301 

1302def list_local_models(endpoint: str = _DEFAULT_LOCAL_ENDPOINT) -> list[str]: 

1303 """Model ids a local server lists, or ``[]`` — see :func:`local_model_listing`. 

1304 

1305 ``[]`` means both "the server lists none" and "the listing failed". Callers that 

1306 must tell those apart (the doctor: a server with nothing pulled cannot review) 

1307 use :func:`local_model_listing`, which answers ``None`` for a failed listing. 

1308 """ 

1309 return local_model_listing(endpoint) or [] 

1310 

1311 

1312def local_model_listing(endpoint: str = _DEFAULT_LOCAL_ENDPOINT) -> list[str] | None: 

1313 """List model ids from a local OpenAI-compatible server (issue #109). 

1314 

1315 GETs ``{endpoint}/models`` (the OpenAI-compatible listing that Ollama, 

1316 vLLM, LM Studio, etc. expose) and returns the model ids in their reported 

1317 order. Best-effort and stdlib-only: any failure (server down, bad JSON) 

1318 returns ``None`` — distinct from ``[]``, a server that answered and lists no 

1319 model — so callers can fall back gracefully. 

1320 

1321 The endpoint is validated here at the seam (issue #309) so EVERY caller — 

1322 including the un-gated ``jury init --local-endpoint`` discovery path — gets 

1323 the same SSRF gate that ``config._endpoint_issues`` enforces for config-file 

1324 endpoints: a non-``http(s)`` scheme or a non-loopback host (without the 

1325 ``JURY_ALLOW_REMOTE_ENDPOINT`` opt-in) yields ``None`` without any network call. 

1326 """ 

1327 import json as _json 

1328 

1329 from .config import _endpoint_issues 

1330 

1331 base = (endpoint or _DEFAULT_LOCAL_ENDPOINT).rstrip("/") 

1332 try: 

1333 # SSRF gate INSIDE the try (review of #309): `_endpoint_issues` calls 

1334 # urlsplit, which raises ValueError on a malformed URL (e.g. `http://[::1`); 

1335 # keep the best-effort "any failure -> None" contract rather than crashing. 

1336 if _endpoint_issues(base, "local-endpoint")[0]: # hard-error issues -> refuse 

1337 return None 

1338 url = base if base.endswith("/models") else f"{base}/models" 

1339 with _open(url, _VERSION_PROBE_TIMEOUT) as resp: # noqa: S310 

1340 data = _json.loads(resp.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace")) 

1341 except Exception: # noqa: BLE001 - discovery is best-effort 

1342 return None 

1343 if not isinstance(data, dict) or "data" not in data: 

1344 return None 

1345 models = data["data"] 

1346 if models is None: 

1347 # Ollama with nothing pulled answers `{"object": "list", "data": null}`, not 

1348 # `"data": []` (measured on Ollama 0.34.1) — the very server #849 is about. 

1349 return [] 

1350 if not isinstance(models, list): 

1351 return None 

1352 ids = [m.get("id") for m in models if isinstance(m, dict) and m.get("id")] 

1353 return [str(i) for i in ids] 

1354 

1355 

1356class LocalAdapter(Adapter): 

1357 """Open-weight / local-model reviewer over an OpenAI-compatible API (issue #43). 

1358 

1359 Targets the ``/v1/chat/completions`` endpoint exposed by common local servers 

1360 (Ollama, llama.cpp ``llama-server``, vLLM, LM Studio). It talks plain HTTP via 

1361 the stdlib (``urllib``) — no new dependencies and no subprocess — so one panel 

1362 seat can run free and fully offline, adding model diversity (the load-bearing 

1363 advantage) at zero marginal cost. 

1364 

1365 Configure as a normal ``[[agent]]`` with ``vendor = "local"``, an 

1366 ``endpoint`` (base URL, default ``http://localhost:11434/v1``), and a 

1367 ``model``. ``extra_args`` is unused. An unreachable server fails with the 

1368 typed ``connection_error`` code (issue #29) rather than a crash. 

1369 

1370 ``temperature`` is sent as configured, else the greedy ``0``. Greedy is the 

1371 right default for a reviewer, but some models do not survive it: gpt-oss 

1372 loops in its reasoning at 0 until its output cap and never answers. 

1373 """ 

1374 

1375 SUPPORTS_HEADLESS = True 

1376 SUPPORTS_MODEL_SELECTION = True 

1377 SPAWNS_PROCESS = False # plain HTTP, no subprocess 

1378 

1379 @property 

1380 def endpoint(self) -> str: 

1381 return (self.spec.endpoint or _DEFAULT_LOCAL_ENDPOINT).rstrip("/") 

1382 

1383 def completions_url(self) -> str: 

1384 """Resolve the chat-completions URL from the configured base endpoint. 

1385 

1386 Accepts either a base URL (``…/v1``) or a full completions URL; pure so it 

1387 can be unit-tested without network. 

1388 """ 

1389 base = self.endpoint 

1390 if base.endswith("/chat/completions"): 

1391 return base 

1392 return f"{base}/chat/completions" 

1393 

1394 def build_payload(self, prompt: str) -> dict: 

1395 """Build the OpenAI-compatible chat-completions request body (pure).""" 

1396 return { 

1397 "model": self.resolved_model(), 

1398 "messages": [{"role": "user", "content": prompt}], 

1399 "stream": False, 

1400 # Unset keeps the literal `0` every earlier release sent. 

1401 "temperature": 0 if self.spec.temperature is None else self.spec.temperature, 

1402 } 

1403 

1404 @staticmethod 

1405 def parse_content(data: dict) -> str: 

1406 """Extract the assistant message text from a chat-completions response.""" 

1407 choices = data.get("choices") or [] 

1408 if not choices: 

1409 return "" 

1410 message = choices[0].get("message") or {} 

1411 return (message.get("content") or "").strip() 

1412 

1413 @staticmethod 

1414 def classify_http_status(status: int) -> str: 

1415 """Map an HTTP error status to a typed error code (issue #29).""" 

1416 if status in (401, 403): 

1417 return ERR_AUTH_REQUIRED 

1418 if status == 429: 

1419 return ERR_RATE_LIMITED 

1420 return ERR_NONZERO_EXIT 

1421 

1422 def available(self) -> bool: 

1423 """A local agent is 'available' when its server answers a quick probe. 

1424 

1425 Probes the OpenAI-compatible ``/v1/models`` (or the endpoint root) with a 

1426 short timeout. Network-only; never raises. 

1427 """ 

1428 # Only `urllib.error` here: the request itself goes through `_open`, whose 

1429 # `_http_only_opener` imports `urllib.request` for it. 

1430 import urllib.error 

1431 

1432 url = f"{self.endpoint}/models" 

1433 try: 

1434 with _open(url, _VERSION_PROBE_TIMEOUT) as resp: # noqa: S310 

1435 return 200 <= resp.status < 500 

1436 except urllib.error.HTTPError as exc: 

1437 # A 4xx (e.g. 404 on /models) still means the server is up. 

1438 return exc.code < 500 

1439 except Exception: # noqa: BLE001 - unreachable server -> not available 

1440 return False 

1441 

1442 def list_models(self) -> list[str] | None: 

1443 """Model ids the local server advertises, or None when it has none.""" 

1444 return list_local_models(self.endpoint) or None 

1445 

1446 def detect_capabilities(self) -> dict: 

1447 reachable = self.available() 

1448 return { 

1449 "version": None, 

1450 "supports_headless": self.SUPPORTS_HEADLESS, 

1451 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION, 

1452 "raw_version_output": f"local endpoint {self.endpoint}", 

1453 "status": CAP_OK if reachable else CAP_UNAVAILABLE, 

1454 "warnings": ([] if reachable else [f"local server unreachable at {self.endpoint}"]), 

1455 } 

1456 

1457 def run( 

1458 self, 

1459 prompt: str, 

1460 phase: str = "review", 

1461 timeout: int | None = None, 

1462 role_policy=None, 

1463 ) -> AgentResult: 

1464 import json as _json 

1465 import urllib.error 

1466 import urllib.request 

1467 

1468 del phase 

1469 # A network adapter has no tools, no shell and no filesystem, so a 

1470 # write-enabled role changes nothing about how it is invoked (#661). 

1471 del role_policy 

1472 effective_timeout = self.spec.timeout 

1473 if timeout is not None: 

1474 effective_timeout = max(1, min(self.spec.timeout, int(timeout))) 

1475 body = _json.dumps(self.build_payload(prompt)).encode("utf-8") 

1476 req = urllib.request.Request( 

1477 self.completions_url(), 

1478 data=body, 

1479 headers={"Content-Type": "application/json"}, 

1480 method="POST", 

1481 ) 

1482 start = time.monotonic() 

1483 try: 

1484 with _open(req, effective_timeout) as resp: # noqa: S310 

1485 raw = resp.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace") 

1486 data = _json.loads(raw) 

1487 except urllib.error.HTTPError as exc: 

1488 detail = "" 

1489 try: 

1490 detail = exc.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace")[:300] 

1491 except Exception: # noqa: BLE001 

1492 detail = exc.reason or "" 

1493 # The body is from a possibly-untrusted endpoint and is surfaced in 

1494 # the report; redact recognized secrets before embedding (#293/F-8). 

1495 detail = redaction.redact(detail)[0] 

1496 return AgentResult( 

1497 self.name, 

1498 self.spec.vendor, 

1499 False, 

1500 "", 

1501 time.monotonic() - start, 

1502 f"HTTP {exc.code}: {detail}", 

1503 error_code=self.classify_http_status(exc.code), 

1504 ) 

1505 except TimeoutError: 

1506 return AgentResult( 

1507 self.name, 

1508 self.spec.vendor, 

1509 False, 

1510 "", 

1511 time.monotonic() - start, 

1512 f"timed out after {effective_timeout}s", 

1513 error_code=ERR_TIMEOUT, 

1514 ) 

1515 except urllib.error.URLError as exc: 

1516 return AgentResult( 

1517 self.name, 

1518 self.spec.vendor, 

1519 False, 

1520 "", 

1521 time.monotonic() - start, 

1522 f"could not reach local server at {self.endpoint}: {redaction.redact(str(exc.reason))[0]}", 

1523 error_code=ERR_CONNECTION, 

1524 ) 

1525 except Exception as exc: # noqa: BLE001 - surface any other failure 

1526 return AgentResult( 

1527 self.name, 

1528 self.spec.vendor, 

1529 False, 

1530 "", 

1531 time.monotonic() - start, 

1532 f"local request failed: {redaction.redact(str(exc))[0]}", 

1533 error_code=ERR_UNKNOWN, 

1534 ) 

1535 dur = time.monotonic() - start 

1536 content = self.parse_content(data) 

1537 if not content: 

1538 return AgentResult( 

1539 self.name, 

1540 self.spec.vendor, 

1541 False, 

1542 "", 

1543 dur, 

1544 "local model returned empty content", 

1545 error_code=ERR_EMPTY_OUTPUT, 

1546 ) 

1547 no_review = no_review_reason(content) 

1548 if no_review is not None: 

1549 return AgentResult( 

1550 self.name, 

1551 self.spec.vendor, 

1552 False, 

1553 "", 

1554 dur, 

1555 f"no review returned: {no_review}", 

1556 error_code=ERR_NO_REVIEW, 

1557 ) 

1558 return AgentResult(self.name, self.spec.vendor, True, content, dur) 

1559 

1560 

1561# Hosted vendor API endpoints (issue #430). Fixed, not configurable: unlike 

1562# `local`'s user-supplied `endpoint` (which needs the SSRF validation in 

1563# config._endpoint_issues), a hosted vendor's URL is a known constant, not an 

1564# attacker- or operator-influenceable value, so there is nothing to validate. 

1565_ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages" 

1566_ANTHROPIC_API_VERSION = "2023-06-01" 

1567_OPENAI_API_URL = "https://api.openai.com/v1/chat/completions" 

1568_XAI_API_URL = "https://api.x.ai/v1/chat/completions" 

1569# The Anthropic Messages API requires max_tokens on every request; there is no 

1570# server-side default. Generous enough for a review response, small enough to 

1571# bound cost/latency if a run is ever misconfigured to loop. 

1572_HOSTED_API_MAX_TOKENS = 4096 

1573 

1574 

1575def _hosted_api_status_code(status: int) -> str: 

1576 """Map a hosted-API HTTP status to a typed error code (issue #430). 

1577 

1578 Shared by every hosted-API adapter — identical mapping to 

1579 ``LocalAdapter.classify_http_status`` (401/403 → auth, 429 → rate limit), 

1580 kept as a free function since it has no per-adapter state. 

1581 """ 

1582 if status in (401, 403): 

1583 return ERR_AUTH_REQUIRED 

1584 if status == 429: 

1585 return ERR_RATE_LIMITED 

1586 return ERR_NONZERO_EXIT 

1587 

1588 

1589def _post_json( 

1590 url: str, payload: dict, headers: dict[str, str], timeout: int 

1591) -> tuple[dict | None, str | None, str | None]: 

1592 """POST a JSON body and parse a JSON response (issue #430). 

1593 

1594 Shared HTTP mechanics for the hosted-API adapters: build the request, 

1595 route it through the SSRF-safe opener (``_open``, no file/ftp handlers, no 

1596 redirect following — the same seam ``LocalAdapter`` uses), cap the response 

1597 read at ``_MAX_RESPONSE_BYTES``, and classify any failure into a typed 

1598 error code. Returns ``(response_dict, None, None)`` on success or 

1599 ``(None, error_message, error_code)`` on failure — exactly one shape. 

1600 Response bodies are redacted before being returned in an error message 

1601 since they originate from the network and are surfaced in the report. 

1602 """ 

1603 import json as _json 

1604 import urllib.error 

1605 import urllib.request 

1606 

1607 body = _json.dumps(payload).encode("utf-8") 

1608 req = urllib.request.Request(url, data=body, headers=headers, method="POST") 

1609 try: 

1610 with _open(req, timeout) as resp: # noqa: S310 

1611 raw = resp.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace") 

1612 return _json.loads(raw), None, None 

1613 except urllib.error.HTTPError as exc: 

1614 detail = "" 

1615 try: 

1616 detail = exc.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace")[:300] 

1617 except Exception: # noqa: BLE001 - reading the error body is best-effort 

1618 detail = exc.reason or "" 

1619 detail = redaction.redact(detail)[0] 

1620 return None, f"HTTP {exc.code}: {detail}", _hosted_api_status_code(exc.code) 

1621 except TimeoutError: 

1622 return None, f"timed out after {timeout}s", ERR_TIMEOUT 

1623 except urllib.error.URLError as exc: 

1624 return ( 

1625 None, 

1626 f"could not reach {url}: {redaction.redact(str(exc.reason))[0]}", 

1627 ERR_CONNECTION, 

1628 ) 

1629 except Exception as exc: # noqa: BLE001 - surface any other failure 

1630 return None, f"request failed: {redaction.redact(str(exc))[0]}", ERR_UNKNOWN 

1631 

1632 

1633class _HostedApiAdapter(Adapter): 

1634 """Shared base for hosted-vendor-API reviewers keyed by an env-var API key. 

1635 

1636 No CLI install, no interactive login, no subprocess: just an HTTP call 

1637 over stdlib ``urllib`` to the vendor's real hosted API (issue #430), the 

1638 same no-subprocess/no-new-dependency design as ``LocalAdapter`` but 

1639 pointed at a hosted endpoint instead of a local server. The API key is 

1640 read from the environment ONLY, never from ``jury.toml``, so it cannot 

1641 leak into a checked-in config; the endpoint is a fixed per-vendor 

1642 constant, not a config value, so there is no SSRF surface to guard the 

1643 way `local`'s `endpoint` needs. 

1644 """ 

1645 

1646 SUPPORTS_HEADLESS = True 

1647 SUPPORTS_MODEL_SELECTION = True 

1648 SPAWNS_PROCESS = False # plain HTTP, no subprocess 

1649 

1650 # The environment variable this vendor's credential is read FROM. A name, 

1651 # never a value — deliberately not called `_API_KEY_*`: the constant holds 

1652 # public configuration, and a credential-shaped name on a non-credential is 

1653 # how both a reader and a static analyzer end up misreading this path. 

1654 # Subclasses override. 

1655 _ENV_VAR_NAME: str = "OPENAI_API_KEY" 

1656 

1657 def _env_var_name(self) -> str: 

1658 """Name of the environment variable holding this agent's credential. 

1659 

1660 Rebuilt through :func:`redaction.safe_env_var_name`, because this value 

1661 is *displayed* — it reaches ``jury --doctor``, its JSON export, and 

1662 warning text — while ``[[agent]] api_key_env`` is an arbitrary operator 

1663 string. The sanitizer both bounds it to a real env var name (so a config 

1664 value cannot splice a newline into a JSON document) and severs it from 

1665 the credential-shaped config field it came from. The credential VALUE 

1666 never travels this way; see :meth:`_api_key`. 

1667 """ 

1668 return redaction.safe_env_var_name( 

1669 getattr(self.spec, "api_key_env", None), self._ENV_VAR_NAME 

1670 ) 

1671 

1672 def _api_key(self) -> str: 

1673 """The credential itself. NEVER rendered — only compared and sent. 

1674 

1675 Kept under a deliberately sensitive name so any future flow from here 

1676 into a log or an export is reported rather than blending in. 

1677 """ 

1678 return os.environ.get(self._env_var_name(), "") 

1679 

1680 def _api_url(self) -> str: # pragma: no cover - overridden 

1681 raise NotImplementedError 

1682 

1683 def _invalid_key_reason(self) -> str | None: 

1684 """None if the key is safe to use as an HTTP header value; else why not. 

1685 

1686 A key containing a control character (most plausibly a stray 

1687 trailing ``\\n`` from a file/k8s-secret/`.env` mount) trips CPython's 

1688 ``http.client`` header-injection guard. That guard reports the 

1689 rejected value via ``repr()`` (e.g. an embedded newline becomes the 

1690 two literal characters ``\\`` ``n``), which does **not** byte-for-byte 

1691 match the raw key — so a literal substring scrub of the exception 

1692 text (see :meth:`_scrub_secret`) cannot reliably catch it; the 

1693 transformed text is no longer equal to the original secret. Validate 

1694 and reject *before* the key ever reaches a header instead of trying 

1695 to scrub it back out afterward. 

1696 """ 

1697 if any(ord(ch) < 0x20 or ord(ch) == 0x7F for ch in self._api_key()): 

1698 return ( 

1699 f"{self._env_var_name()} contains a control character (e.g. a stray " 

1700 f"trailing newline from how the secret was loaded) and cannot be " 

1701 f"used as an HTTP header value" 

1702 ) 

1703 return None 

1704 

1705 def _scrub_secret(self, text: str) -> str: 

1706 """Strip the literal API key value from an error message (issue #430). 

1707 

1708 Defense-in-depth alongside ``redaction.redact()`` (which only 

1709 recognizes known vendor-token *shapes* via regex) for any leak path 

1710 NOT already ruled out by :meth:`_invalid_key_reason` — e.g. a 

1711 well-formed key that still ends up quoted in some other library's 

1712 error text. Not a substitute for that check: once a value contains 

1713 control characters, downstream formatting (``repr()``, percent- 

1714 encoding, ...) can transform it before it reaches an error message, 

1715 and a literal match against the *original* key would then silently 

1716 miss it — which is exactly why control characters are rejected 

1717 upfront in :meth:`run` instead of relying on this alone. 

1718 """ 

1719 key = self._api_key() 

1720 if key and key in text: 

1721 return text.replace(key, "[REDACTED]") 

1722 return text 

1723 

1724 def available(self) -> bool: 

1725 """Available when a *usable* API key is set — a fast, network-free check. 

1726 

1727 Unlike ``LocalAdapter.available()`` (which probes the server, since a 

1728 local endpoint's reachability is genuinely uncertain), a hosted 

1729 vendor's API is assumed reachable; the two real unknowns locally are 

1730 whether the operator configured a key at all, and whether it's 

1731 actually usable as a header value (see :meth:`_invalid_key_reason`) — 

1732 a key that will be rejected by :meth:`run` should not report as 

1733 available here either, or a capability check (``jury --doctor``) 

1734 would give a falsely reassuring answer. 

1735 """ 

1736 return bool(self._api_key()) and self._invalid_key_reason() is None 

1737 

1738 def detect_capabilities(self) -> dict: 

1739 key_set = bool(self._api_key()) 

1740 invalid_reason = self._invalid_key_reason() if key_set else None 

1741 has_key = key_set and invalid_reason is None 

1742 if not key_set: 

1743 warnings = [f"{self._env_var_name()} is not set in the environment"] 

1744 elif invalid_reason: 

1745 warnings = [invalid_reason] 

1746 else: 

1747 warnings = [] 

1748 return { 

1749 "version": None, 

1750 "supports_headless": self.SUPPORTS_HEADLESS, 

1751 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION, 

1752 "raw_version_output": f"hosted API {self._api_url()}", 

1753 "status": CAP_OK if has_key else CAP_UNAVAILABLE, 

1754 "warnings": warnings, 

1755 } 

1756 

1757 def build_payload(self, prompt: str) -> dict: # pragma: no cover - overridden 

1758 raise NotImplementedError 

1759 

1760 def _headers(self) -> dict[str, str]: # pragma: no cover - overridden 

1761 raise NotImplementedError 

1762 

1763 @staticmethod 

1764 def parse_content(data: dict) -> str: # pragma: no cover - overridden 

1765 raise NotImplementedError 

1766 

1767 def run( 

1768 self, 

1769 prompt: str, 

1770 phase: str = "review", 

1771 timeout: int | None = None, 

1772 role_policy=None, 

1773 ) -> AgentResult: 

1774 del phase 

1775 # A network adapter has no tools, no shell and no filesystem, so a 

1776 # write-enabled role changes nothing about how it is invoked (#661). 

1777 del role_policy 

1778 # Checked independently of available() (not just "not available()"): 

1779 # available() now also returns False for a key that IS set but 

1780 # invalid, and that case needs its own distinct error_code/message 

1781 # below rather than the misleading "is not set" one. 

1782 if not self._api_key(): 

1783 return AgentResult( 

1784 self.name, 

1785 self.spec.vendor, 

1786 False, 

1787 "", 

1788 0.0, 

1789 f"{self._env_var_name()} is not set in the environment", 

1790 error_code=ERR_MISSING_API_KEY, 

1791 ) 

1792 invalid_reason = self._invalid_key_reason() 

1793 if invalid_reason is not None: 

1794 # Reject before the key ever reaches a header — see 

1795 # _invalid_key_reason for why post-hoc scrubbing can't be trusted 

1796 # here. This message never echoes the key itself. 

1797 return AgentResult( 

1798 self.name, 

1799 self.spec.vendor, 

1800 False, 

1801 "", 

1802 0.0, 

1803 invalid_reason, 

1804 error_code=ERR_INVALID_API_KEY, 

1805 ) 

1806 effective_timeout = self.spec.timeout 

1807 if timeout is not None: 

1808 effective_timeout = max(1, min(self.spec.timeout, int(timeout))) 

1809 start = time.monotonic() 

1810 data, err_msg, err_code = _post_json( 

1811 self._api_url(), self.build_payload(prompt), self._headers(), effective_timeout 

1812 ) 

1813 dur = time.monotonic() - start 

1814 if err_msg is not None: 

1815 return AgentResult( 

1816 self.name, 

1817 self.spec.vendor, 

1818 False, 

1819 "", 

1820 dur, 

1821 self._scrub_secret(err_msg), 

1822 error_code=err_code, 

1823 ) 

1824 content = self.parse_content(data or {}) 

1825 if not content: 

1826 return AgentResult( 

1827 self.name, 

1828 self.spec.vendor, 

1829 False, 

1830 "", 

1831 dur, 

1832 "hosted API returned empty content", 

1833 error_code=ERR_EMPTY_OUTPUT, 

1834 ) 

1835 no_review = no_review_reason(content) 

1836 if no_review is not None: 

1837 return AgentResult( 

1838 self.name, 

1839 self.spec.vendor, 

1840 False, 

1841 "", 

1842 dur, 

1843 f"no review returned: {no_review}", 

1844 error_code=ERR_NO_REVIEW, 

1845 ) 

1846 return AgentResult(self.name, self.spec.vendor, True, content, dur) 

1847 

1848 

1849class AnthropicApiAdapter(_HostedApiAdapter): 

1850 """Hosted Anthropic Messages API reviewer, keyed by ``ANTHROPIC_API_KEY`` (issue #430). 

1851 

1852 Configure as a normal ``[[agent]]`` with ``vendor = "anthropic-api"`` and a 

1853 ``model`` (e.g. a current Claude model id) — no ``command``, no ``claude`` 

1854 CLI install or interactive login needed. 

1855 """ 

1856 

1857 _ENV_VAR_NAME = "ANTHROPIC_API_KEY" 

1858 

1859 def _api_url(self) -> str: 

1860 return _ANTHROPIC_API_URL 

1861 

1862 def build_payload(self, prompt: str) -> dict: 

1863 """Build the Anthropic Messages API request body (pure). 

1864 

1865 With an effort level configured, extended thinking is enabled with the 

1866 mapped token budget. ``max_tokens`` must exceed that budget (thinking 

1867 tokens are drawn from the same allowance), so it is raised to leave the 

1868 original response allowance on top of it — bounded by 

1869 :data:`_ANTHROPIC_MAX_TOKENS_CEILING`, since per-model ``max_tokens`` 

1870 caps vary and an unbounded sum would build a request some models reject. 

1871 """ 

1872 payload = { 

1873 "model": self.resolved_model(), 

1874 "max_tokens": _HOSTED_API_MAX_TOKENS, 

1875 "messages": [{"role": "user", "content": prompt}], 

1876 } 

1877 payload.update(self.effort_plan().payload) 

1878 thinking = payload.get("thinking") 

1879 if isinstance(thinking, dict): 

1880 # Clamped here too, not only in effort_args, so `budget < max_tokens 

1881 # <= ceiling` holds for any plan handed to this builder. 

1882 budget = min( 

1883 int(thinking["budget_tokens"]), 

1884 _ANTHROPIC_MAX_TOKENS_CEILING - _HOSTED_API_MAX_TOKENS, 

1885 ) 

1886 payload["thinking"] = {**thinking, "budget_tokens": budget} 

1887 payload["max_tokens"] = budget + _HOSTED_API_MAX_TOKENS 

1888 return payload 

1889 

1890 def _headers(self) -> dict[str, str]: 

1891 return { 

1892 "Content-Type": "application/json", 

1893 "x-api-key": self._api_key(), 

1894 "anthropic-version": _ANTHROPIC_API_VERSION, 

1895 } 

1896 

1897 @staticmethod 

1898 def parse_content(data: dict) -> str: 

1899 """Extract the assistant text from a Messages API response.""" 

1900 if not isinstance(data, dict): 

1901 return "" 

1902 blocks = data.get("content") or [] 

1903 texts = [ 

1904 block.get("text", "") 

1905 for block in blocks 

1906 if isinstance(block, dict) and block.get("type") == "text" 

1907 ] 

1908 return "".join(texts).strip() 

1909 

1910 

1911class OpenAiApiAdapter(_HostedApiAdapter): 

1912 """Hosted OpenAI Chat Completions API reviewer, keyed by ``OPENAI_API_KEY`` (issue #430). 

1913 

1914 Configure as a normal ``[[agent]]`` with ``vendor = "openai-api"`` and a 

1915 ``model`` (e.g. a current GPT model id) — no ``command``, no ``codex`` CLI 

1916 install or interactive login needed. Same request/response shape as 

1917 ``LocalAdapter`` (both are OpenAI-compatible chat completions), just 

1918 against the real hosted API with an ``Authorization`` header. 

1919 """ 

1920 

1921 _ENV_VAR_NAME = "OPENAI_API_KEY" 

1922 

1923 def _api_url(self) -> str: 

1924 return _OPENAI_API_URL 

1925 

1926 def build_payload(self, prompt: str) -> dict: 

1927 """Build the OpenAI chat-completions request body (pure).""" 

1928 payload = { 

1929 "model": self.resolved_model(), 

1930 "messages": [{"role": "user", "content": prompt}], 

1931 } 

1932 payload.update(self.effort_plan().payload) 

1933 return payload 

1934 

1935 def _headers(self) -> dict[str, str]: 

1936 return { 

1937 "Content-Type": "application/json", 

1938 "Authorization": f"Bearer {self._api_key()}", 

1939 } 

1940 

1941 @staticmethod 

1942 def parse_content(data: dict) -> str: 

1943 """Extract the assistant message text from a chat-completions response.""" 

1944 if not isinstance(data, dict): 

1945 return "" 

1946 choices = data.get("choices") or [] 

1947 if not choices: 

1948 return "" 

1949 message = choices[0].get("message") or {} 

1950 return (message.get("content") or "").strip() 

1951 

1952 

1953class XaiApiAdapter(OpenAiApiAdapter): 

1954 """Hosted xAI (Grok) API reviewer, keyed by ``XAI_API_KEY`` (issue #701). 

1955 

1956 Configure as a normal ``[[agent]]`` with ``vendor = "xai-api"`` and a 

1957 ``model`` (a Grok model id) — no ``command``, no CLI install. xAI serves 

1958 the OpenAI chat-completions shape at its own host, so this is 

1959 :class:`OpenAiApiAdapter` with a different URL and a different credential 

1960 env var; the request body, the response parsing and the effort knob 

1961 (``reasoning_effort``) are inherited rather than re-stated, because a 

1962 second copy of them is a second thing to keep in step. 

1963 

1964 ``vendor = "openai-compatible"`` with ``endpoint = "https://api.x.ai/v1"`` 

1965 still works and is still documented; this spelling exists so a Grok seat 

1966 can carry the vendor identity ``xai-api`` instead of borrowing OpenAI's. 

1967 """ 

1968 

1969 _ENV_VAR_NAME = "XAI_API_KEY" 

1970 

1971 def _api_url(self) -> str: 

1972 return _XAI_API_URL 

1973 

1974 

1975_GEMINI_API_BASE = "https://generativelanguage.googleapis.com/v1beta/models" 

1976 

1977 

1978class GoogleApiAdapter(_HostedApiAdapter): 

1979 """Hosted Google Gemini API reviewer, keyed by ``GEMINI_API_KEY`` (issue #432). 

1980 

1981 Configure as a normal ``[[agent]]`` with ``vendor = "google-api"`` and a 

1982 ``model`` (e.g. a current Gemini model id) — no ``command``, no ``agy`` 

1983 CLI install or interactive login needed. 

1984 

1985 Two differences from the other two hosted adapters: 

1986 

1987 - The Gemini API embeds the model id in the URL **path** 

1988 (``.../models/{model}:generateContent``), not the request body, so 

1989 ``_api_url()`` is built from ``self.spec.model`` on every call rather 

1990 than returning a fixed constant like the other two adapters. 

1991 - The key is sent via the ``x-goog-api-key`` header. Gemini also accepts 

1992 the key as a ``?key=...`` query parameter, but a query-string key is a 

1993 much easier accidental-leak vector (proxy/access logs, anything that 

1994 prints the request URL) than a header — deliberately not supported. 

1995 

1996 A prompt blocked by Gemini's safety filters comes back with an empty 

1997 ``candidates`` list (and a ``promptFeedback.blockReason``); this is not 

1998 distinguished from a genuinely empty response and both currently surface 

1999 as the same generic ``ERR_EMPTY_OUTPUT`` — a possible future refinement, 

2000 not required for parity with the other two adapters. 

2001 """ 

2002 

2003 _ENV_VAR_NAME = "GEMINI_API_KEY" 

2004 

2005 def _api_url(self) -> str: 

2006 # Escape the model id as a single path segment (issue #432 review): an 

2007 # operator-configured model containing reserved URL characters 

2008 # (`/`, `?`, `#`, ...) would otherwise change the request's path/query 

2009 # semantics instead of staying a single `{model}` segment. 

2010 import urllib.parse 

2011 

2012 model = urllib.parse.quote(self.resolved_model(), safe="") 

2013 return f"{_GEMINI_API_BASE}/{model}:generateContent" 

2014 

2015 def build_payload(self, prompt: str) -> dict: 

2016 """Build the Gemini ``generateContent`` request body (pure).""" 

2017 payload: dict = {"contents": [{"parts": [{"text": prompt}]}]} 

2018 payload.update(self.effort_plan().payload) 

2019 return payload 

2020 

2021 def _headers(self) -> dict[str, str]: 

2022 return { 

2023 "Content-Type": "application/json", 

2024 "x-goog-api-key": self._api_key(), 

2025 } 

2026 

2027 @staticmethod 

2028 def parse_content(data: dict) -> str: 

2029 """Extract the assistant text from a ``generateContent`` response.""" 

2030 if not isinstance(data, dict): 

2031 return "" 

2032 candidates = data.get("candidates") or [] 

2033 if not candidates or not isinstance(candidates[0], dict): 

2034 return "" 

2035 content = candidates[0].get("content") 

2036 if not isinstance(content, dict): 

2037 return "" 

2038 parts = content.get("parts") or [] 

2039 texts = [ 

2040 part.get("text", "") 

2041 for part in parts 

2042 if isinstance(part, dict) and isinstance(part.get("text", ""), str) 

2043 ] 

2044 return "".join(texts).strip() 

2045 

2046 

2047class MockAdapter(Adapter): 

2048 """Offline adapter for tests and ``--mock`` runs. 

2049 

2050 Produces deterministic, phase-aware text so the full orchestration pipeline 

2051 can run end-to-end without live CLIs, auth, or token spend. 

2052 """ 

2053 

2054 # Synthetic capabilities: the mock is offline and runs no real CLI. 

2055 SUPPORTS_HEADLESS = True 

2056 SUPPORTS_MODEL_SELECTION = False 

2057 

2058 def available(self) -> bool: 

2059 return True 

2060 

2061 def detect_capabilities(self) -> dict: 

2062 """Deterministic fake capabilities so doctor/tests stay stable offline.""" 

2063 return { 

2064 "version": "mock-1.0", 

2065 "supports_headless": self.SUPPORTS_HEADLESS, 

2066 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION, 

2067 "raw_version_output": "mock-1.0", 

2068 "status": CAP_OK, 

2069 "warnings": [], 

2070 } 

2071 

2072 def run( 

2073 self, 

2074 prompt: str, 

2075 phase: str = "review", 

2076 timeout: int | None = None, 

2077 role_policy=None, 

2078 ) -> AgentResult: 

2079 del prompt, timeout, role_policy 

2080 n = self.name 

2081 if phase == "review": 

2082 body = ( 

2083 # The two lines the review prompt asks every reviewer to open 

2084 # with (#700). The mock speaks the shape a real reviewer is asked 

2085 # for, so `--mock` exercises the scope/testing lift rather than 

2086 # only the inference fallback behind it. 

2087 "Checked: src/example.py\n" 

2088 "Tested: nothing run (offline mock reviewer)\n" 

2089 f"- **[major]** `src/example.py:42` — {n}: unchecked return value " 

2090 f"may swallow an error.\n" 

2091 f"- **[minor]** `src/example.py:7` — {n}: missing docstring.\n\n" 

2092 "```json\n" 

2093 "[\n" 

2094 ' {"severity": "major", "file": "src/example.py", "line": 42, ' 

2095 f'"claim": "{n}: unchecked return value may swallow an error", ' 

2096 '"evidence": "the added code ignores the return value of int(x)", ' 

2097 '"suggested_fix": "check the result and raise on failure", ' 

2098 f'"confidence": "high", "reviewer": "{n}"}},\n' 

2099 ' {"severity": "minor", "file": "src/example.py", "line": 7, ' 

2100 f'"claim": "{n}: missing docstring", ' 

2101 '"evidence": "the new function parse() has no docstring", ' 

2102 '"suggested_fix": "add a one-line docstring", ' 

2103 f'"confidence": "medium", "reviewer": "{n}"}}\n' 

2104 "]\n" 

2105 "```" 

2106 ) 

2107 elif phase == "debate": 

2108 body = ( 

2109 f"## AGREE\n- {n}: confirm the unchecked-return finding at " 

2110 f"`src/example.py:42`.\n" 

2111 f"## DISPUTE\n- {n}: the missing-docstring finding is a nit, not blocking.\n" 

2112 f"## MISSED\n- {n}: no test covers the error branch." 

2113 ) 

2114 elif phase == "verify": 

2115 body = ( 

2116 "Verification: confirming the unchecked-return finding at " 

2117 "`src/example.py:42`; the missing-docstring claim at `:7` is a nit " 

2118 "not supported as blocking.\n\n" 

2119 "```json\n" 

2120 "[\n" 

2121 ' {"file": "src/example.py", "line": 42, ' 

2122 '"claim": "unchecked return value may swallow an error", ' 

2123 '"status": "verified", ' 

2124 '"reasoning": "the added code ignores the return value of int(x)"},\n' 

2125 ' {"file": "src/example.py", "line": 7, ' 

2126 '"claim": "missing docstring", ' 

2127 '"status": "unsupported", ' 

2128 '"reasoning": "a missing docstring is not a defect the diff introduces"}\n' 

2129 "]\n" 

2130 "```" 

2131 ) 

2132 else: # synthesis 

2133 body = ( 

2134 "## Verdict\nREQUEST CHANGES — one confirmed major issue.\n\n" 

2135 "## Consensus findings\n- **[major]** `src/example.py:42` — unchecked " 

2136 "return value (raised by all reviewers).\n\n" 

2137 "## Disputed findings\n- Missing docstring: ruled non-blocking.\n\n" 

2138 "## Notable single-reviewer findings\n- Missing test for the error branch." 

2139 ) 

2140 return AgentResult(n, self.spec.vendor, True, body, 0.0) 

2141 

2142 

2143class GenericOpenAICompatibleAdapter(_HostedApiAdapter): 

2144 """Hosted OpenAI-compatible API reviewer (OpenRouter, DeepSeek, Groq, Mistral API, LiteLLM, etc.). 

2145 

2146 Supports custom ``endpoint``, custom ``api_key_env``, and extra HTTP ``headers``. 

2147 """ 

2148 

2149 _ENV_VAR_NAME = "OPENAI_API_KEY" 

2150 

2151 def _api_url(self) -> str: 

2152 endpoint = (self.spec.endpoint or _OPENAI_API_URL).rstrip("/") 

2153 if endpoint.endswith("/chat/completions"): 2153 ↛ 2154line 2153 didn't jump to line 2154 because the condition on line 2153 was never true

2154 return endpoint 

2155 return f"{endpoint}/chat/completions" 

2156 

2157 def build_payload(self, prompt: str) -> dict: 

2158 payload = { 

2159 "model": self.resolved_model(), 

2160 "messages": [{"role": "user", "content": prompt}], 

2161 } 

2162 payload.update(self.effort_plan().payload) 

2163 return payload 

2164 

2165 def _headers(self) -> dict[str, str]: 

2166 hdrs = { 

2167 "Content-Type": "application/json", 

2168 "Authorization": f"Bearer {self._api_key()}", 

2169 } 

2170 if self.spec.headers: 2170 ↛ 2172line 2170 didn't jump to line 2172 because the condition on line 2170 was always true

2171 hdrs.update(self.spec.headers) 

2172 return hdrs 

2173 

2174 @staticmethod 

2175 def parse_content(data: dict) -> str: 

2176 if not isinstance(data, dict): 2176 ↛ 2177line 2176 didn't jump to line 2177 because the condition on line 2176 was never true

2177 return "" 

2178 choices = data.get("choices") or [] 

2179 if not choices or not isinstance(choices, list): 

2180 return "" 

2181 first = choices[0] 

2182 if not isinstance(first, dict): 2182 ↛ 2183line 2182 didn't jump to line 2183 because the condition on line 2182 was never true

2183 return "" 

2184 msg = first.get("message") or {} 

2185 if not isinstance(msg, dict): 2185 ↛ 2186line 2185 didn't jump to line 2186 because the condition on line 2185 was never true

2186 return "" 

2187 raw_content = msg.get("content") 

2188 if isinstance(raw_content, str): 

2189 return raw_content.strip() 

2190 if isinstance(raw_content, list): 2190 ↛ 2201line 2190 didn't jump to line 2201 because the condition on line 2190 was always true

2191 parts = [] 

2192 for item in raw_content: 

2193 if isinstance(item, dict): 2193 ↛ 2198line 2193 didn't jump to line 2198 because the condition on line 2193 was always true

2194 if (item.get("type") == "text" or "text" in item) and isinstance( 2194 ↛ 2192line 2194 didn't jump to line 2192 because the condition on line 2194 was always true

2195 item.get("text"), str 

2196 ): 

2197 parts.append(item["text"]) 

2198 elif isinstance(item, str): 

2199 parts.append(item) 

2200 return "".join(parts).strip() 

2201 return "" 

2202 

2203 

2204class GenericCLIAdapter(Adapter): 

2205 """Generic CLI adapter for arbitrary coding-agent CLIs (Aider, Goose, OpenHands, Copilot CLI, etc.). 

2206 

2207 Supports configurable prompt delivery modes: 

2208 - ``prompt_mode = "stdin"`` (default): prompt passed via STDIN 

2209 - ``prompt_mode = "arg"``: prompt passed as positional argument on argv 

2210 """ 

2211 

2212 def _prompt_mode(self) -> str: 

2213 return (self.spec.prompt_mode or "stdin").lower() 

2214 

2215 def build_argv(self, prompt: str) -> list[str]: 

2216 """Read-only argv for a configured `cli` profile. 

2217 

2218 Implemented rather than inherited (the base raises) so this adapter goes 

2219 through the same ``build_argv_for_role`` seam as every other one — the 

2220 role policy is then decided in exactly one place for every vendor. 

2221 """ 

2222 argv = [self.spec.command, *_read_only_extra_args(self.spec)] 

2223 return [*argv, prompt] if self._prompt_mode() == "arg" else argv 

2224 

2225 def build_write_argv(self, prompt: str) -> list[str]: 

2226 """Write-capable argv. 

2227 

2228 A `cli` profile has no vendor-specific sandbox flag to add or remove, so 

2229 this resolves to the same configured ``extra_args`` as the read-only 

2230 argv. It is still routed through :func:`_write_extra_args` so the rule 

2231 lives with every other vendor's rather than being special-cased here. 

2232 """ 

2233 argv = [self.spec.command, *_write_extra_args(self.spec)] 

2234 return [*argv, prompt] if self._prompt_mode() == "arg" else argv 

2235 

2236 def _stdin_for(self, prompt: str) -> str | None: 

2237 return None if self._prompt_mode() == "arg" else prompt 

2238 

2239 def _argv_too_long_hint(self, exc: BaseException, prompt: str) -> str: 

2240 """Why the spawn failed, when the prompt on argv made the command line too long (#901). 

2241 

2242 ``prompt_mode = "arg"`` puts the whole prompt in one argument. Linux caps 

2243 one argument at 128 KiB (measured: 131071 bytes spawned, 131072 failed 

2244 with ``E2BIG``), below the 200000-byte ``[jury.diff] max_bytes`` default; 

2245 macOS caps the whole command line at ``ARG_MAX`` (1 MiB, also ``E2BIG``); 

2246 Windows at 32767 characters, which ``CreateProcess`` reports as error 206 

2247 (Microsoft's documentation; not measured here). Said, so the operator is 

2248 not left with a bare "Argument list too long". 

2249 """ 

2250 too_long = isinstance(exc, OSError) and ( 

2251 exc.errno == errno.E2BIG or getattr(exc, "winerror", None) == 206 

2252 ) 

2253 if not too_long or self._prompt_mode() != "arg": 

2254 return "" 

2255 size = len(prompt.encode("utf-8")) 

2256 return ( 

2257 f"; the prompt ({size} bytes) is one command-line argument " 

2258 f'(prompt_mode = "arg"), longer than this system allows. Use ' 

2259 f'prompt_mode = "stdin" if the CLI reads its prompt from stdin, or lower ' 

2260 f"[jury.diff] max_bytes / chunk_max_bytes." 

2261 ) 

2262 

2263 def available(self) -> bool: 

2264 command = self.spec.command or "" 

2265 if not command: 2265 ↛ 2266line 2265 didn't jump to line 2266 because the condition on line 2265 was never true

2266 return False 

2267 return shutil.which(command) is not None 

2268 

2269 def detect_capabilities(self) -> dict: 

2270 if not self.available(): 

2271 return { 

2272 "version": None, 

2273 "supports_headless": None, 

2274 "supports_model_selection": None, 

2275 "raw_version_output": "", 

2276 "status": CAP_UNAVAILABLE, 

2277 "warnings": [f"command '{self.spec.command}' not found on PATH"], 

2278 } 

2279 return { 

2280 "version": "generic-cli", 

2281 "supports_headless": True, 

2282 "supports_model_selection": bool(self.spec.model), 

2283 "raw_version_output": "generic-cli", 

2284 "status": CAP_OK, 

2285 "warnings": [], 

2286 } 

2287 

2288 def run( 

2289 self, 

2290 prompt: str, 

2291 phase: str = "review", 

2292 timeout: int | None = None, 

2293 role_policy=None, 

2294 ) -> AgentResult: 

2295 del phase 

2296 if not self.available(): 

2297 return AgentResult( 

2298 agent=self.spec.name, 

2299 vendor=self.spec.vendor, 

2300 ok=False, 

2301 output="", 

2302 duration_s=0.0, 

2303 error=f"command '{self.spec.command}' not found on PATH.", 

2304 error_code=ERR_MISSING_CLI, 

2305 ) 

2306 effective_timeout = timeout if timeout is not None else self.spec.timeout 

2307 argv = self.build_argv_for_role(prompt, role_policy) 

2308 stdin_content = self._stdin_for(prompt) 

2309 

2310 start = time.monotonic() 

2311 try: 

2312 res = _spawn(argv, stdin_content, timeout=effective_timeout) 

2313 except subprocess.TimeoutExpired: 

2314 return AgentResult( 

2315 self.spec.name, 

2316 self.spec.vendor, 

2317 False, 

2318 "", 

2319 effective_timeout, 

2320 f"execution timed out after {effective_timeout}s.", 

2321 error_code=ERR_TIMEOUT, 

2322 ) 

2323 except Exception as exc: 

2324 duration = time.monotonic() - start 

2325 return AgentResult( 

2326 self.spec.name, 

2327 self.spec.vendor, 

2328 False, 

2329 "", 

2330 duration, 

2331 f"failed to spawn '{self.spec.command}': {redaction.redact(str(exc))[0]}" 

2332 f"{self._argv_too_long_hint(exc, prompt)}", 

2333 error_code=ERR_SPAWN_FAILED, 

2334 ) 

2335 

2336 duration = time.monotonic() - start 

2337 out = (res.stdout or "").strip() 

2338 err = (res.stderr or "").strip() 

2339 

2340 if res.returncode != 0: 

2341 detail = err or out or f"exited with code {res.returncode}" 

2342 safe_detail = redaction.redact(detail)[0] 

2343 err_code = classify_stderr(res.returncode, err or out) 

2344 return AgentResult( 

2345 self.spec.name, 

2346 self.spec.vendor, 

2347 False, 

2348 "", 

2349 duration, 

2350 f"exit {res.returncode}: {safe_detail[:500]}", 

2351 error_code=err_code, 

2352 exit_code=res.returncode, 

2353 ) 

2354 

2355 if not out: 2355 ↛ 2356line 2355 didn't jump to line 2356 because the condition on line 2355 was never true

2356 return AgentResult( 

2357 self.spec.name, 

2358 self.spec.vendor, 

2359 False, 

2360 "", 

2361 duration, 

2362 "agent produced empty output", 

2363 error_code=ERR_EMPTY_OUTPUT, 

2364 exit_code=res.returncode, 

2365 ) 

2366 

2367 no_review = no_review_reason(out) 

2368 if no_review is not None: 

2369 return AgentResult( 

2370 self.spec.name, 

2371 self.spec.vendor, 

2372 False, 

2373 "", 

2374 duration, 

2375 f"no review returned: {no_review}: {redaction.redact(out)[0][:200]}", 

2376 error_code=ERR_NO_REVIEW, 

2377 exit_code=res.returncode, 

2378 ) 

2379 

2380 return AgentResult( 

2381 self.spec.name, self.spec.vendor, True, out, duration, exit_code=res.returncode 

2382 ) 

2383 

2384 

2385#: The adapter registry: protocol name -> the class that builds its argv. Keyed 

2386#: by vendor name because a vendor's shipped adapter IS its default protocol — 

2387#: which is also why the vendor vocabulary doubles as the ``adapter`` vocabulary 

2388#: (``config.recognised_adapters``). A seat naming ``adapter`` picks a row here 

2389#: directly instead of inheriting its vendor's (issue #705). 

2390_VENDOR_ADAPTERS: dict[str, type[Adapter]] = { 

2391 "anthropic": ClaudeAdapter, 

2392 "openai": CodexAdapter, 

2393 "google": AgyAdapter, 

2394 "local": LocalAdapter, 

2395 "anthropic-api": AnthropicApiAdapter, 

2396 "openai-api": OpenAiApiAdapter, 

2397 "google-api": GoogleApiAdapter, 

2398 "xai-api": XaiApiAdapter, 

2399 "openai-compatible": GenericOpenAICompatibleAdapter, 

2400 # `xai` is a bring-your-own-CLI seat (Grok through Cursor's `cursor-agent`, 

2401 # issue #701): the operator supplies `command`/`extra_args`, exactly as for 

2402 # `cli`, but the seat keeps its own vendor identity at the panel gate. 

2403 "xai": GenericCLIAdapter, 

2404 "cli": GenericCLIAdapter, 

2405} 

2406 

2407 

2408def register_adapter(vendor: str, adapter_cls: type[Adapter]) -> None: 

2409 """Register a custom adapter class for a vendor string. 

2410 

2411 The name registered is usable as BOTH a ``vendor`` and an ``adapter`` 

2412 (issue #705): the registry is the adapter vocabulary, so a custom adapter is 

2413 selectable by a seat that keeps some other vendor identity. 

2414 

2415 Registering also teaches ``config`` the name (issue #701): a vendor with an 

2416 adapter behind it is a vendor this run genuinely knows, so it must not warn 

2417 as unknown and must not be folded into the generic ``cli`` identity at the 

2418 cross-vendor gate. The two registries are updated together so they cannot 

2419 disagree about what "known" means. 

2420 """ 

2421 name = config_module.normalise_vendor(vendor) 

2422 if not name: 

2423 # One guard, ahead of both tables. `register_vendor` ignores a name that is 

2424 # empty after normalisation, and this used to write the adapter under the 

2425 # key "" regardless — the two registries disagreeing for exactly the input 

2426 # they were joined to agree on. 

2427 raise ValueError("register_adapter: vendor name is empty after normalisation") 

2428 config_module.register_vendor(name) 

2429 # The class's own answer (#903), so a custom HTTP adapter written as a direct 

2430 # `Adapter` subclass — the documented example — can say it runs no process. 

2431 # Only an explicit `False` opts out: a class that says nothing, or says 

2432 # `None`, spawns, which keeps it under the audit. 

2433 config_module.register_adapter_transport( 

2434 name, spawns=getattr(adapter_cls, "SPAWNS_PROCESS", True) is not False 

2435 ) 

2436 _VENDOR_ADAPTERS[name] = adapter_cls 

2437 

2438 

2439def _registry_state() -> tuple[dict[str, type[Adapter]], set[str], dict[str, bool]]: 

2440 """A copy of every table :func:`register_adapter` writes (#904). 

2441 

2442 Three tables in two modules: the adapter classes here, and ``config``'s 

2443 registered vendor names and transports. Tests that register an adapter must 

2444 put all three back, and restoring them one dict at a time left the ones each 

2445 test forgot — a leaked name failed tests in another module, depending on the 

2446 order the modules ran in. The list of tables lives here, beside the function 

2447 that writes them, so a fourth table is added to the snapshot in the same place. 

2448 """ 

2449 return ( 

2450 dict(_VENDOR_ADAPTERS), 

2451 set(config_module._REGISTERED_VENDORS), 

2452 dict(config_module._REGISTERED_ADAPTER_SPAWNS), 

2453 ) 

2454 

2455 

2456def _restore_registry_state(state: tuple[dict, set, dict]) -> None: 

2457 """Put back a :func:`_registry_state` snapshot, in place. 

2458 

2459 In place, not by rebinding: other modules hold the tables by reference 

2460 (``from ai_jury.adapters import _VENDOR_ADAPTERS``), and a rebound name would 

2461 leave them reading the old object. 

2462 """ 

2463 adapter_classes, vendors, spawns = state 

2464 for table, saved in ( 

2465 (_VENDOR_ADAPTERS, adapter_classes), 

2466 (config_module._REGISTERED_VENDORS, vendors), 

2467 (config_module._REGISTERED_ADAPTER_SPAWNS, spawns), 

2468 ): 

2469 table.clear() 

2470 table.update(saved) 

2471 

2472 

2473def make_adapter(spec: AgentSpec, mock: bool = False) -> Adapter: 

2474 """The adapter that builds *spec*'s command line. 

2475 

2476 Selected by the seat's ADAPTER key, which is its ``vendor`` unless the 

2477 operator named one (issue #705). Keying this on the vendor made the two 

2478 inseparable: a GPT model reached through Cursor's ``cursor-agent`` got 

2479 ``codex exec`` argv it cannot parse, and the only way to a passthrough argv 

2480 was ``vendor = "cli"``, which cost the seat its identity at the panel gate. 

2481 Nothing else about the seat moves — the ballot, the report and 

2482 ``min_vendors`` all still read ``spec.vendor``. 

2483 

2484 Raises ``ConfigError`` when the seat NAMES an adapter this build does not 

2485 have (issue #708). The generic fall-through below still catches an unknown 

2486 *vendor* — that seat named no protocol, so inheriting the generic one is the 

2487 documented behaviour — but it must never catch a named one: falling through 

2488 there is the silent guess #705 exists to remove, and it made every reader 

2489 that builds a seat without validating the config — a caller of 

2490 ``load_config`` with its default ``validate=False``, or of this function 

2491 directly — disagree with the run about the very same file. 

2492 """ 

2493 # ``or None`` mirrors ``AgentSpec.__post_init__``: an adapter that 

2494 # normalises to nothing is no adapter at all, and falls back to the vendor. 

2495 # A raw dict-built or hand-made spec must read the same as a loaded one. 

2496 adapter_error = config_module.unknown_adapter_error( 

2497 getattr(spec, "adapter", None) or None, getattr(spec, "name", "") or "" 

2498 ) 

2499 if adapter_error is not None: 

2500 raise config_module.ConfigError(adapter_error) 

2501 if mock: 

2502 return MockAdapter(spec) 

2503 return adapter_class(spec)(spec) 

2504 

2505 

2506def adapter_class(spec) -> type[Adapter]: 

2507 """The adapter class :func:`make_adapter` builds for *spec* (pure, duck-typed). 

2508 

2509 The ADAPTER key decides first: a registered name always gets its class, 

2510 whatever else the seat carries — so an ``endpoint`` on a ``cli``, ``claude``, 

2511 ``codex`` or ``agy`` seat changes nothing about how it runs. Only a seat with 

2512 no registered adapter falls through to the ``endpoint``/``command`` guess. 

2513 """ 

2514 cls = _VENDOR_ADAPTERS.get(config_module.spec_adapter(spec)) 

2515 if cls is not None: 

2516 return cls 

2517 command = getattr(spec, "command", "") or "" 

2518 if getattr(spec, "endpoint", None) or (getattr(spec, "api_key_env", None) and not command): 

2519 return GenericOpenAICompatibleAdapter 

2520 if command: 

2521 return GenericCLIAdapter 

2522 return AgyAdapter