Coverage for src/ai_jury/adapters.py: 98%
855 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Agent adapters — each wraps one native coding-agent CLI in headless mode.
3Every adapter turns a prompt into a subprocess invocation and captures stdout as
4the agent's response. Adapters are intentionally thin: the orchestrator owns the
5prompt content and the round structure; an adapter only knows how to *invoke its
6CLI*.
8Headless invocations (verified against installed CLIs, early 2026). The prompt
9embeds the redacted diff, so it is delivered on STDIN (never argv) for every
10real adapter so it is not exposed in the process list (issue #287):
11 - Claude Code : ``claude -p --output-format text`` (prompt piped via stdin)
12 - Codex CLI : ``codex exec <args>`` (prompt piped via stdin)
13 - Antigravity : ``agy --print`` (prompt piped via stdin)
14"""
16from __future__ import annotations
18import contextlib
19import errno
20import json
21import os
22import re
23import shutil
24import signal
25import subprocess
26import tempfile
27import time
28from dataclasses import dataclass, field
30from . import config as config_module
31from . import privilege, redaction
32from .config import AgentSpec
33from .findings import emitted_findings_block
35# Cap on a single local-model HTTP response body (issue #293/F-9). A chat
36# completion is small; an unbounded read from a malicious/buggy endpoint would
37# let it OOM the process.
38_MAX_RESPONSE_BYTES = 16 * 1024 * 1024
41def _kill_process_group(proc: subprocess.Popen) -> None:
42 """Best-effort kill of the child's whole process group (issue #293/F-7)."""
43 if hasattr(os, "killpg"): 43 ↛ 47line 43 didn't jump to line 47 because the condition on line 43 was always true
44 with contextlib.suppress(ProcessLookupError, PermissionError, OSError):
45 os.killpg(os.getpgid(proc.pid), signal.SIGKILL)
46 return
47 with contextlib.suppress(OSError):
48 proc.kill()
51def _spawn(
52 argv: list[str], stdin: str | None, timeout: int, cwd: str | None = None
53) -> subprocess.CompletedProcess:
54 """Run a CLI with stdout/stderr captured, killing the whole group on timeout.
56 ``subprocess.run(timeout=…)`` SIGKILLs only the direct child, so an agent CLI
57 that wraps node/python can leak orphaned grandchildren (issue #293/F-7). The
58 child is started in its own session (process-group leader); on timeout the
59 entire group is killed before re-raising ``TimeoutExpired`` so the caller's
60 handling is unchanged. ``cwd`` starts the child somewhere other than this
61 process's working directory (see :func:`_review_workdir`). Returns a
62 ``CompletedProcess``.
63 """
64 popen_kwargs: dict = {
65 "stdin": subprocess.PIPE if stdin is not None else None,
66 "stdout": subprocess.PIPE,
67 "stderr": subprocess.PIPE,
68 "text": True,
69 }
70 if cwd is not None:
71 popen_kwargs["cwd"] = cwd
72 if hasattr(os, "setsid"): 72 ↛ 74line 72 didn't jump to line 74 because the condition on line 72 was always true
73 popen_kwargs["start_new_session"] = True
74 proc = subprocess.Popen(argv, **popen_kwargs)
75 try:
76 out, err = proc.communicate(input=stdin, timeout=timeout)
77 except subprocess.TimeoutExpired:
78 _kill_process_group(proc)
79 with contextlib.suppress(Exception):
80 proc.communicate() # reap the killed child
81 raise
82 return subprocess.CompletedProcess(argv, proc.returncode, out, err)
85@contextlib.contextmanager
86def _review_workdir(isolate: bool):
87 """A fresh, empty directory for one reviewer process, removed afterwards.
89 Yields ``None`` — "inherit this process's directory" — when ``isolate`` is
90 false. Otherwise the reviewer starts outside the repository under review, so
91 whatever that repository carries for the agent CLI to pick up on its own is
92 out of reach: instruction files (``CLAUDE.md``, ``AGENTS.md``), project
93 settings with hooks or MCP servers (``.claude/settings.json``, ``.mcp.json``),
94 and the relative path to a ``.env``. On a PR checkout every one of those is
95 the author's, and a reviewer needs none of them — its prompt carries the diff.
97 Cleanup ignores errors, so a file the CLI left behind can never turn a
98 finished review into a failure; the directory is created per call, so two
99 seats running in parallel never share one.
100 """
101 if not isolate:
102 yield None
103 return
104 with tempfile.TemporaryDirectory(prefix="ai-jury-review-", ignore_cleanup_errors=True) as path:
105 yield path
108def _read_only_extra_args(spec: AgentSpec) -> list[str]:
109 """The agent's ``extra_args`` with the read-only sandbox added when none is named.
111 Enforced at the adapter layer (issue #288) so an **empty** ``extra_args`` cannot
112 produce a write-capable reviewer of an attacker-controlled diff. It injects; it
113 does not override. A config that names its own sandbox keeps it — ``-s
114 workspace-write`` is passed through as written — codex's bypass flags are passed
115 through too, and the ``cli``/``xai`` adapters have no enforcement at all. Those
116 are the cases :func:`ai_jury.privilege.audit_agent` reports and ``--strict``
117 fails on (#750); this function is not the guarantee on its own.
118 """
119 # Keyed on the ADAPTER, not the vendor (issue #705): the sandbox flag is a
120 # property of the CLI being spawned, not of whose model answers. A seat with
121 # `vendor = "openai", adapter = "cli", command = "cursor-agent"` must not have
122 # codex's `-s read-only` spliced into an unrelated binary.
123 return privilege.enforce_read_only(config_module.spec_adapter(spec), spec.extra_args)
126def _write_extra_args(spec: AgentSpec) -> list[str]:
127 """The agent's ``extra_args`` with its vendor write/tool mode enabled (#661).
129 Reached ONLY from ``jury run-agent --role implement|fix --allow-write``, via
130 :meth:`Adapter.build_write_argv`. Nothing on the panel path calls it.
131 """
132 return privilege.enable_write(config_module.spec_adapter(spec), spec.extra_args)
135# Short timeout for capability/version probes. Detection is best-effort and must
136# never slow down or block a normal run, so probes are deliberately snappy.
137_VERSION_PROBE_TIMEOUT = 10
139# Matches a version-looking token, e.g. "1.2", "1.2.3", "v0.45.1".
140_VERSION_RE = re.compile(r"\d+\.\d+(?:\.\d+)?")
142# Capability/version probe statuses.
143CAP_OK = "ok"
144CAP_UNKNOWN_VERSION = "unknown_version"
145CAP_UNAVAILABLE = "unavailable"
147# Stable, typed error taxonomy for failed agent executions. These codes let
148# reports and CI/policy distinguish retryable from non-retryable failures
149# instead of pattern-matching free-text error strings.
150ERR_MISSING_CLI = "missing_cli"
151ERR_AUTH_REQUIRED = "auth_required"
152ERR_PERMISSION_PROMPT = "permission_prompt"
153ERR_TIMEOUT = "timeout"
154ERR_NONZERO_EXIT = "nonzero_exit"
155ERR_EMPTY_OUTPUT = "empty_output"
156ERR_SPAWN_FAILED = "spawn_failed"
157ERR_RATE_LIMITED = "rate_limited"
158# Local/HTTP adapter could not reach its server (issue #43): connection refused,
159# DNS failure, or the local model server is not running.
160ERR_CONNECTION = "connection_error"
161# Hosted-API adapter (issue #430): the vendor's API key env var is unset. Distinct
162# from ERR_AUTH_REQUIRED (a key was sent but the server rejected it) so a report
163# can tell "never configured" apart from "misconfigured/expired/revoked".
164ERR_MISSING_API_KEY = "missing_api_key"
165# Hosted-API adapter (issue #430): the configured key contains a control
166# character and was rejected BEFORE being sent as a header, rather than
167# letting http.client raise (and risk echoing a transformed/escaped copy of
168# the secret in its exception text — see _HostedApiAdapter._invalid_key_reason).
169ERR_INVALID_API_KEY = "invalid_api_key"
170# The agent exited 0 and printed something, but that something is not a review
171# (issue #682): a refusal, or the CLI's own usage/argument-error/version banner
172# echoed back because the invocation never reached the model. Distinct from
173# ERR_EMPTY_OUTPUT (nothing at all on stdout) so a report can tell "the CLI is
174# broken/misinvoked" apart from "the model declined", and distinct from ok=True
175# so neither can be counted as a vendor that contributed to consensus. #635 is
176# the shape: `agy` passed every availability probe, printed an argument error,
177# and the panel silently became single-vendor.
178ERR_NO_REVIEW = "no_review"
179ERR_UNKNOWN = "unknown"
181ERROR_CODES = frozenset(
182 {
183 ERR_MISSING_CLI,
184 ERR_AUTH_REQUIRED,
185 ERR_PERMISSION_PROMPT,
186 ERR_TIMEOUT,
187 ERR_NONZERO_EXIT,
188 ERR_EMPTY_OUTPUT,
189 ERR_SPAWN_FAILED,
190 ERR_RATE_LIMITED,
191 ERR_CONNECTION,
192 ERR_MISSING_API_KEY,
193 ERR_INVALID_API_KEY,
194 ERR_NO_REVIEW,
195 ERR_UNKNOWN,
196 }
197)
199# Failures that are worth retrying because they are typically transient (issue
200# #30): a timeout, a rate-limit, a process that failed to spawn, or a local
201# server that was briefly unreachable (#43). Auth, missing-CLI,
202# permission-prompt, empty-output, and generic nonzero-exit are treated as
203# deterministic — retrying them just burns time and tokens. So is a
204# no-review output: a misinvoked CLI prints the same usage banner every time.
205RETRYABLE_ERROR_CODES = frozenset(
206 {
207 ERR_TIMEOUT,
208 ERR_RATE_LIMITED,
209 ERR_SPAWN_FAILED,
210 ERR_CONNECTION,
211 }
212)
215# Ordered keyword groups for classify_stderr. Each keyword is matched on word
216# boundaries (\b...\b) so incidental substrings do NOT trigger a false
217# classification: bare "auth" matches "auth error" but not "author identity",
218# and "login" matches "login required" but not "login_attempts" ("_" is a word
219# char, so there is no boundary inside "login_attempts"). Multi-word phrases
220# tolerate a space OR "_" between tokens (e.g. "rate limit"/"rate_limit").
221def _keyword_pattern(*keywords: str) -> re.Pattern[str]:
222 parts = [r"[ _]+".join(re.escape(tok) for tok in kw.split()) for kw in keywords]
223 return re.compile(r"\b(?:" + "|".join(parts) + r")\b")
226# Order matters: auth and rate-limit signals are checked before the generic
227# permission and nonzero-exit fallbacks.
228_AUTH_RE = _keyword_pattern(
229 "not authenticated",
230 "unauthenticated",
231 "authentication",
232 "unauthorized",
233 "api key",
234 "auth",
235 "log in",
236 "login",
237 "credential",
238 "credentials",
239)
240_RATE_LIMIT_RE = _keyword_pattern("rate limit", "429", "quota", "too many requests")
241_PERMISSION_RE = _keyword_pattern(
242 "permission",
243 "permissions",
244 "approve",
245 "approval",
246 "confirm",
247 "confirmation",
248)
251def classify_stderr(returncode: int, stderr: str) -> str:
252 """Classify a nonzero-exit failure into a typed error code from its stderr.
254 Token-aware matching against the lowercased stderr: each keyword group is a
255 word-boundary regex, so incidental substrings (e.g. "author" containing
256 "auth") never cause a misclassification. Ordering matters (auth and
257 rate-limit signals are checked before the generic permission and
258 nonzero-exit fallbacks). Returns one of the ``ERR_*`` codes.
259 """
260 text = (stderr or "").lower()
261 if _AUTH_RE.search(text):
262 return ERR_AUTH_REQUIRED
263 if _RATE_LIMIT_RE.search(text):
264 return ERR_RATE_LIMITED
265 if _PERMISSION_RE.search(text):
266 return ERR_PERMISSION_PROMPT
267 del returncode
268 return ERR_NONZERO_EXIT
271# --- "Exit 0, but nothing reviewable came back" (issue #682) ------------------
272#
273# A CLI that exits 0 and prints its own usage text is indistinguishable, to
274# everything downstream, from a reviewer that read the diff — the run stays
275# fail-soft and the panel quietly loses a vendor (#635). These patterns turn
276# that into a typed failure at the adapter boundary, on SHAPE alone: never on
277# whether a review is any good, only on whether it is a review at all.
279#: An output longer than this is treated as a review even if it opens with a
280#: refusal-shaped sentence. A real review that mentions "I cannot verify X"
281#: mid-argument must never be discarded; an actual refusal is a short paragraph.
282_NO_REVIEW_MAX_CHARS = 600
284#: How much of the output is examined for a usage/argument-error banner. A CLI
285#: that is going to print one prints it first.
286_NO_REVIEW_HEAD_CHARS = 400
288# The #635 class: the launcher rejected the argv, so the model was never
289# reached. Every one of these is text a CLI writes about ITSELF — and the whole
290# difficulty is that a review may *talk about* the same words. "Usage of int()
291# is unsafe" and "Invalid argument passed to calculate_total()" are findings,
292# not banners, and discarding either costs a panelist and (with the guard
293# failing closed) can collapse the panel. So each pattern below is matched
294# against a whole line, and every branch is bounded the way the refusal branch
295# already is: a banner is short, or it is corroborated by banner structure.
297#: A launcher's own usage line, as a whole line. Prose that merely opens with
298#: the word "usage" does not match: the line must go on to look like a synopsis
299#: (a colon, or a bracketed/flag-shaped operand).
300_USAGE_LINE_RE = re.compile(
301 r"usage:\s*\S" # "Usage: agy [options] [prompt]"
302 r"|usage\s+of\s+\S+:$" # Go's flag package: "Usage of ./agy:"
303 r"|usage\s+\S+\s+[\[<-]", # "usage agy [options]"
304 re.IGNORECASE,
305)
307#: Text only a launcher writes about itself — it complains about the argv it
308#: was handed, in the first person of a program. Needs no corroboration beyond
309#: opening a short output.
310_CLI_SELF_ERROR_RE = re.compile(
311 r"(?:error|fatal)\s*:\s*(?:unknown|unrecognized|invalid|unexpected)\s+"
312 r"(?:flag|option|argument|command|subcommand)\b"
313 r"|flag needs an argument\b"
314 r"|.{0,40}\bcommand not found$",
315 re.IGNORECASE,
316)
318#: The same complaint without the ``error:`` prefix. This one IS ambiguous with
319#: review prose, so it must both name a flag-shaped token ("unknown flag:
320#: --print") and be corroborated by surrounding banner structure.
321_BARE_ARG_ERROR_RE = re.compile(
322 r"(?:unknown|unrecognized|invalid|unexpected)\s+"
323 r"(?:flag|option|argument|command|subcommand)\b"
324 r"[\s:=]*[\"'`]?-{1,2}[A-Za-z0-9]",
325 re.IGNORECASE,
326)
328#: Corroborating structure, never a trigger on its own: the options/flags block
329#: under a synopsis, or the help hint a launcher prints beneath it. The old
330#: unanchored "for more information, try/see" pattern lives on only here — it
331#: is a sentence a review can perfectly well contain.
332_BANNER_STRUCTURE_RE = re.compile(
333 r"^[ \t]*(?:options|flags|commands|arguments|subcommands)\b[ \t]*:?[ \t]*$"
334 r"|^[ \t]*-{1,2}[A-Za-z0-9][\w-]*(?:[ \t=,]|$)"
335 r"|^[ \t]*for more information,? (?:try|see)\b",
336 re.IGNORECASE | re.MULTILINE,
337)
339#: Nothing but a version/probe banner, e.g. "1.1.22" or "agy version 1.1.22".
340_VERSION_ONLY_RE = re.compile(r"^[\w.@/+-]{0,40}(?:\s+version)?\s*v?\d+\.\d+[\w.+-]*$")
342#: A refusal, matched on the lowercased text and only inside the length bound
343#: above. Deliberately narrow: these are first-person declines, not any
344#: sentence containing the word "cannot".
345#:
346#: Matching this is NOT on its own enough to discard an output — see
347#: :func:`_is_whole_output_refusal`. A reviewer routinely opens with a limit it
348#: hit ("I do not have access to the migration file, so I reviewed the Python
349#: only: …") and then reviews; discarding that costs a panelist and, with the
350#: gate failing closed, turns a passing two-vendor run into exit 3.
351_REFUSAL_RE = re.compile(
352 r"i(?:'|\u2019)?m (?:sorry|unable|not able)"
353 r"|i am (?:sorry|unable|not able)"
354 r"|i (?:can(?:'|\u2019)?t|cannot|won(?:'|\u2019)?t|will not)"
355 r" (?:help|assist|do|review|comply|provide|analyz|analys)"
356 r"|sorry,? (?:but )?i "
357 r"|as an ai\b"
358 r"|i (?:do not|don(?:'|\u2019)?t) have (?:access|the ability)"
359)
362#: How far into the output a decline may begin and still BE the output. A
363#: refusal says so first; a review that ran into a limit mid-argument says so
364#: after paragraphs of review. One short lead-in sentence is allowed for, no more.
365_REFUSAL_OPENING_CHARS = 80
367#: Marks of an output that reviewed something, however briefly. Any one of them
368#: outranks the refusal phrase in front of it: an agent that says it could not
369#: read one file and then names a defect HAS reviewed, and its finding is the
370#: thing the panel exists to collect.
371_REVIEW_SUBSTANCE_RE = re.compile(
372 r"\blines?\s+\d+" # "line 12", "lines 40-52"
373 r"|\b[\w./-]+\.[a-z]{1,5}:\d+" # "src/app.py:12"
374 r"|^@@[ \t]" # a hunk header quoted back
375 r"|\bi (?:reviewed|read|checked|examined|inspected|looked at|went through)\b"
376 r"|\b(?:having|after) (?:reviewed|read|checked|examined)\b",
377 re.IGNORECASE | re.MULTILINE,
378)
380#: The adversative pivot that turns a limit into a review: "I'm not able to
381#: reproduce the race locally, BUT the lock ordering here is wrong." It counts
382#: only when what follows is prose of its own and is not itself another decline
383#: — "I'm sorry, but I can't help with this" pivots into the same refusal.
384_PIVOT_RE = re.compile(r"\b(?:but|however|though|although|that said)\b", re.IGNORECASE)
386#: How much text must follow a pivot before it can carry a review claim.
387_PIVOT_MIN_CHARS = 20
390def _carries_review_substance(body: str) -> bool:
391 """True when ``body`` says something about the code, not only about itself."""
392 if _REVIEW_SUBSTANCE_RE.search(body):
393 return True
394 pivot = _PIVOT_RE.search(body)
395 if pivot is None:
396 return False
397 rest = body[pivot.end() :].strip()
398 return len(rest) >= _PIVOT_MIN_CHARS and _REFUSAL_RE.search(rest.lower()) is None
401def _is_whole_output_refusal(body: str) -> bool:
402 """True when the output *is* a decline, not a review that mentions a limit.
404 Three conditions, all necessary. The output is short (a real decline is a
405 sentence or two), the decline OPENS it, and nothing in it carries review
406 substance. Merely containing a refusal phrase somewhere inside the length
407 bound was the old rule, and it destroyed genuine short reviews — including
408 ones that named a defect and a line number (#682, round 3).
409 """
410 if len(body) > _NO_REVIEW_MAX_CHARS:
411 return False
412 match = _REFUSAL_RE.search(body.lower())
413 if match is None or match.start() > _REFUSAL_OPENING_CHARS:
414 return False
415 return not _carries_review_substance(body)
418def _is_cli_banner(body: str) -> bool:
419 """True when ``body`` *is* a launcher's banner, not prose that mentions one.
421 Two shapes qualify. A full help dump — a synopsis line plus an options
422 block — is unmistakable at any length. Anything shorter must both open with
423 a banner-shaped line and fit inside the same length bound the refusal branch
424 uses: a real review is not three lines.
425 """
426 head = body[:_NO_REVIEW_HEAD_CHARS]
427 lines = [line.strip() for line in head.splitlines() if line.strip()]
428 if not lines: # pragma: no cover - `body` is stripped and non-empty here
429 return False
430 synopsis = any(_USAGE_LINE_RE.match(line) for line in lines)
431 structure = _BANNER_STRUCTURE_RE.search(head) is not None
432 if synopsis and structure:
433 return True
434 if len(body) > _NO_REVIEW_MAX_CHARS:
435 return False
436 first = lines[0]
437 if _USAGE_LINE_RE.match(first) or _CLI_SELF_ERROR_RE.match(first):
438 return True
439 return _BARE_ARG_ERROR_RE.match(first) is not None and (synopsis or structure)
442def no_review_reason(text: str) -> str | None:
443 """Why ``text`` cannot be a review, or ``None`` when it could be one.
445 PURE, and a judgement of SHAPE only — it never scores a review's quality.
446 An adapter calls it on an exit-0 stdout so that a refusal, a usage banner
447 or a bare version string is recorded as a typed failure (``ERR_NO_REVIEW``)
448 rather than as a contributing vendor. Fail-soft still applies: the run
449 continues, but with a dead seat that says so.
451 Returns a short human-readable reason, safe to embed in an error string
452 (it names the category, never the agent's text).
453 """
454 body = (text or "").strip()
455 if not body:
456 return "the agent produced no output"
457 # A structured findings block is the one machine-checkable proof that the
458 # agent answered in the reviewer's contract, and it settles the question
459 # before any prose heuristic gets a vote: a launcher banner does not emit
460 # one, and an agent that declines and then reports a finding has reviewed.
461 if emitted_findings_block(body):
462 return None
463 if _is_cli_banner(body):
464 return "the CLI printed usage or argument-error text instead of a review"
465 if _VERSION_ONLY_RE.match(body):
466 return "the CLI printed only a version banner instead of a review"
467 if _is_whole_output_refusal(body):
468 return "the agent declined to review"
469 return None
472# --- Reasoning effort (issue #662) -------------------------------------------
473#
474# Every vendor expresses "think harder" differently, so the mapping lives HERE,
475# in one pure function, instead of being spread across the adapters: agy encodes
476# effort as a model-id suffix, the Anthropic Messages API takes an extended-
477# thinking token budget, OpenAI-shaped APIs take `reasoning_effort`, Gemini takes
478# a `thinkingConfig` budget, and the `claude`/`codex` CLIs have no headless knob
479# at all. `effort_args` is the single place any of that is decided.
481#: The effort levels accepted by ``[[agent]] effort`` and ``--effort``.
482EFFORT_LEVELS: tuple[str, ...] = ("low", "medium", "high")
484# Anthropic extended-thinking budgets (tokens) per level, before clamping.
485_ANTHROPIC_THINKING_BUDGET = {"low": 2048, "medium": 8192, "high": 32768}
486# Ceiling on the `max_tokens` this project will ever request from Anthropic.
487# Thinking tokens are drawn from the same allowance, so `max_tokens` has to
488# exceed the budget — but per-model caps vary and an unbounded sum would build a
489# request some models reject outright. 32000 sits inside every current model's
490# limit, so the `high` budget is clamped to `ceiling - _HOSTED_API_MAX_TOKENS`
491# (27904) rather than sending its nominal 32768. Documented in
492# docs/configuration.md.
493_ANTHROPIC_MAX_TOKENS_CEILING = 32000
494# Gemini `thinkingConfig.thinkingBudget` (tokens) per level.
495_GEMINI_THINKING_BUDGET = {"low": 1024, "medium": 8192, "high": 32768}
496# agy model-id suffixes that already carry an effort level.
497_AGY_MODEL_SUFFIXES = tuple(f"-{level}" for level in EFFORT_LEVELS)
499# Vendors that can actually act on an effort level. Everything else (the
500# `claude`/`codex` CLIs, `local`, `cli`, any unregistered vendor) warns once and
501# ignores it: a local OpenAI-compatible server is NOT included, because many of
502# them reject an unknown request field outright, which would turn a hint into a
503# failed review.
504_EFFORT_VENDORS = frozenset(
505 {"google", "anthropic-api", "openai-api", "xai-api", "openai-compatible", "google-api"}
506)
509@dataclass(frozen=True)
510class EffortPlan:
511 """How one vendor expresses a reasoning-effort level.
513 ``model`` is a replacement model id (agy, which encodes effort in the id);
514 ``payload`` is a request-body fragment to merge into the vendor's JSON body;
515 ``warning`` is the once-per-run operator message when the level is ignored.
516 """
518 supported: bool = True
519 model: str | None = None
520 payload: dict = field(default_factory=dict)
521 warning: str | None = None
524def effort_supported(vendor: str) -> bool:
525 """Whether *vendor* can act on an effort level at all (pure)."""
526 return config_module.normalise_vendor(vendor) in _EFFORT_VENDORS
529def _anthropic_budget(level: str) -> int:
530 """Thinking budget for *level*, clamped to the documented ceiling (pure)."""
531 return min(
532 _ANTHROPIC_THINKING_BUDGET[level],
533 _ANTHROPIC_MAX_TOKENS_CEILING - _HOSTED_API_MAX_TOKENS,
534 )
537def _effort_uses_model_listing(vendor: str) -> bool:
538 """Whether *vendor* expresses effort through the model id (pure).
540 Those are the vendors whose mapped id can be checked against what the CLI
541 actually offers, so callers know when discovering a listing is worth a probe.
542 """
543 return config_module.normalise_vendor(vendor) == "google"
546def effort_args(
547 vendor: str,
548 effort: str | None,
549 model: str | None = None,
550 known_models: list[str] | None = None,
551) -> EffortPlan:
552 """Map ``(vendor, effort, model)`` to the vendor's own effort knob (pure).
554 An empty/None *effort* is a no-op plan. An unrecognized level raises
555 ``ValueError`` — ``config.validate_config`` and the ``--effort`` choices
556 already gate user input, so reaching here with garbage is a programming
557 error, not something to silently swallow.
559 ``known_models``, when the caller has discovered one, is the model listing
560 the vendor actually offers. It is only consulted where effort is expressed
561 *as* a model id (agy): sending a suffixed id the CLI does not have would
562 fail the whole review, so an unlisted mapping falls back to the configured
563 id and warns instead. ``None`` means "no listing available" — the check is
564 skipped, never guessed.
565 """
566 level = (effort or "").strip().lower()
567 if not level:
568 return EffortPlan()
569 if level not in EFFORT_LEVELS:
570 raise ValueError(f"unknown effort {effort!r}; expected one of {', '.join(EFFORT_LEVELS)}")
572 name = config_module.normalise_vendor(vendor)
574 if name == "google":
575 # The `agy` CLI selects effort through the model id itself.
576 current = (model or "").strip()
577 if not current:
578 return EffortPlan(
579 warning=(
580 f"effort '{level}' needs a configured model for vendor '{vendor}' "
581 f"(effort is encoded in the model id), ignored"
582 )
583 )
584 if current.endswith(_AGY_MODEL_SUFFIXES):
585 # Already pinned by the operator; an explicit model id wins.
586 return EffortPlan(model=current)
587 suffixed = f"{current}-{level}"
588 if known_models is not None and suffixed not in known_models:
589 # Better a review at the configured depth than no review at all.
590 return EffortPlan(
591 model=current,
592 warning=(
593 f"effort '{level}' maps model '{current}' to '{suffixed}', which "
594 f"vendor '{vendor}' does not offer; using '{current}' unchanged"
595 ),
596 )
597 return EffortPlan(model=suffixed)
599 if name == "anthropic-api":
600 return EffortPlan(
601 payload={
602 "thinking": {
603 "type": "enabled",
604 "budget_tokens": _anthropic_budget(level),
605 }
606 }
607 )
609 if name in ("openai-api", "xai-api", "openai-compatible"):
610 return EffortPlan(payload={"reasoning_effort": level})
612 if name == "google-api":
613 return EffortPlan(
614 payload={
615 "generationConfig": {
616 "thinkingConfig": {"thinkingBudget": _GEMINI_THINKING_BUDGET[level]}
617 }
618 }
619 )
621 return EffortPlan(supported=False, warning=f"effort unsupported for {vendor}, ignored")
624def effort_warnings(agents, adapter_factory=None) -> list[str]:
625 """Deduped, ordered effort warnings for a panel — one message per run.
627 Callers (the CLI) print these once before the run rather than once per agent
628 invocation, so a three-round panel does not repeat the same line nine times.
630 Pure unless *adapter_factory* is given. With it, an agent whose effort is
631 expressed as a model id has its mapped id checked against the vendor's real
632 listing, so "that model does not exist" is reported up front instead of
633 surfacing as an abstention mid-run. The probe is skipped entirely for agents
634 with no effort configured and for every other vendor, and any failure
635 degrades to "no listing" rather than blocking the run.
636 """
637 seen: set[str] = set()
638 out: list[str] = []
639 for spec in agents:
640 # The adapter, not the vendor (#705): how effort is expressed — a model
641 # suffix, a request-body field, nothing at all — is a property of the
642 # protocol the seat is invoked through.
643 vendor = config_module.spec_adapter(spec)
644 effort = getattr(spec, "effort", None)
645 known_models = None
646 if adapter_factory is not None and effort and _effort_uses_model_listing(vendor):
647 try:
648 known_models = adapter_factory(spec).list_models()
649 except Exception: # noqa: BLE001 - discovery is best-effort
650 known_models = None
651 try:
652 plan = effort_args(
653 vendor,
654 effort,
655 getattr(spec, "model", None),
656 known_models=known_models,
657 )
658 except ValueError as exc:
659 # Redact before it reaches a warning/log, like every other str(exc) in this
660 # module — an effort-args failure can echo attacker/config-influenced text (#828).
661 message = redaction.redact(str(exc))[0]
662 else:
663 message = plan.warning
664 if message and message not in seen:
665 seen.add(message)
666 out.append(message)
667 return out
670# Model ids look like `gemini-3.8-flash`, `qwen2.5-coder:7b`, `anthropic/claude-x`.
671_MODEL_ID_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:/-]*$")
672# Cap a discovered model listing so a chatty/hostile CLI cannot flood diagnostics.
673_MAX_LISTED_MODELS = 50
676def parse_model_list(raw: str) -> list[str]:
677 """Model ids out of a CLI's model listing (pure, best-effort, order-preserving).
679 Accepts either JSON (a list, or an object with ``models``/``data``) or plain
680 lines, because a CLI's listing format is not a contract. Anything that does
681 not look like a model id is dropped rather than guessed at.
682 """
683 text = (raw or "").strip()
684 if not text:
685 return []
687 ids: list[str] = []
688 try:
689 data = json.loads(text)
690 except ValueError:
691 data = None
692 if isinstance(data, dict):
693 data = data.get("models") or data.get("data")
694 if isinstance(data, list):
695 for item in data:
696 if isinstance(item, str):
697 ids.append(item)
698 elif isinstance(item, dict):
699 value = item.get("id") or item.get("name") or item.get("model")
700 if isinstance(value, str):
701 ids.append(value)
702 else:
703 for line in text.splitlines():
704 token = line.strip().lstrip("-*\u2022").strip()
705 token = token.split()[0] if token else ""
706 if _MODEL_ID_RE.match(token):
707 ids.append(token)
709 seen: set[str] = set()
710 out: list[str] = []
711 for value in ids:
712 value = value.strip()
713 if value and value not in seen:
714 seen.add(value)
715 out.append(value)
716 return out[:_MAX_LISTED_MODELS]
719@dataclass
720class AgentResult:
721 agent: str
722 vendor: str
723 ok: bool
724 output: str
725 duration_s: float
726 error: str | None = None
727 findings: list = field(default_factory=list)
728 warnings: list = field(default_factory=list)
729 error_code: str | None = None
730 # Number of attempts made for this result (issue #30): 1 means no retry.
731 # >1 records that a transient failure was retried before this outcome.
732 attempts: int = 1
733 # Did this agent emit a structured findings block at all (issue #501)? A
734 # reviewer that examined the diff and found nothing still emits `[]`; one that
735 # produced no review emits prose and no block. Without this, both arrive as zero
736 # findings and the panel reports the same size either way.
737 structured: bool = False
738 # The agent process's own exit status (issue #661), or None when there was
739 # no process to exit: a network adapter, a CLI that was never spawned
740 # (missing on PATH, spawn failure), or one killed on timeout. `jury
741 # run-agent` reports it so an orchestrator can act on the real status
742 # instead of parsing it back out of the error string.
743 exit_code: int | None = None
744 # The model id this invocation actually sent (issue #709), stamped by the
745 # path that ran the adapter from :meth:`Adapter.resolved_model` — the same
746 # call that put the id in the argv/payload. Empty when nothing was pinned
747 # (the CLI chose) or when the record was not produced by an invocation.
748 # A ballot READS this rather than re-deriving it: `vendor` and `adapter` can
749 # differ since #705, and a second derivation is a second answer.
750 model: str = ""
753class Adapter:
754 """Base adapter. Subclasses build the argv for their CLI."""
756 # Declarative capability metadata. Real coding-agent CLIs support a headless
757 # (non-interactive) invocation and model selection; subclasses override where
758 # this differs. ``MockAdapter`` reports synthetic capabilities.
759 SUPPORTS_HEADLESS = True
760 SUPPORTS_MODEL_SELECTION = True
762 #: Start every READ-ONLY invocation — the panel's review, debate, verify and
763 #: synthesis calls, and ``jury run-agent``'s review/gate/chair roles — in a
764 #: fresh empty directory rather than the one ``jury`` runs in (see
765 #: :func:`_review_workdir`). A read-only role reads its prompt and nothing
766 #: else, so it never needs the repository, and a PR checkout's project
767 #: settings (a ``.claude/settings.json`` hook ran from one, measured) must not
768 #: reach it. Only a write role (``implement``/``fix`` with ``--allow-write``)
769 #: stays in ``--cwd``: it has to edit that worktree. On for the three native
770 #: CLIs, whose behaviour there is known — codex needs
771 #: ``--skip-git-repo-check`` outside a git repository, and its read-only argv
772 #: carries it. Off by default, so a bring-your-own ``cli`` seat and a
773 #: registered custom adapter keep running where they always did: this tool
774 #: cannot know what an operator's own binary expects of its directory.
775 ISOLATE_REVIEW_CWD = False
777 #: Whether this adapter runs its agent as a subprocess — the question the
778 #: least-privilege audit asks before it looks for a sandbox. ``True`` unless a
779 #: class says otherwise: the HTTP adapters (``LocalAdapter`` and the
780 #: hosted-API ones) set ``False``, and so can a custom adapter that calls its
781 #: backend over the network from ``run()`` (#903). ``register_adapter`` reads
782 #: it, and only an explicit ``False`` opts out; a registered class that spawns
783 #: a CLI and sets ``False`` would escape the audit, so the default is the safe one.
784 SPAWNS_PROCESS = True
786 # Args passed to the CLI to print its version. Subclasses override if the CLI
787 # uses a different verb/flag (e.g. ``codex --version``).
788 _VERSION_ARGS = ("--version",)
790 def __init__(self, spec: AgentSpec):
791 self.spec = spec
793 @property
794 def name(self) -> str:
795 return self.spec.name
797 def available(self) -> bool:
798 return shutil.which(self.spec.command) is not None
800 def build_argv(self, prompt: str) -> list[str]: # pragma: no cover - overridden
801 raise NotImplementedError
803 def build_write_argv(self, prompt: str) -> list[str]:
804 """Argv for a WRITE-capable invocation of this CLI (issue #661).
806 The default is the read-only argv: a vendor with no tool-enabled mode
807 (and every network adapter, which has no tools at all) simply cannot
808 widen, and failing closed is the right default. The three native CLI
809 adapters override it.
810 """
811 return self.build_argv(prompt)
813 def build_argv_for_role(self, prompt: str, policy=None) -> list[str]:
814 """Argv for one ``jury run-agent`` role (issue #661).
816 The single seam between the role policy and an adapter's argv, so the
817 policy is never re-decided inside :meth:`run`. ``policy=None`` — every
818 panel invocation — resolves to the unchanged read-only :meth:`build_argv`,
819 which is why adding this could not alter an existing run.
821 Both branches end in :meth:`build_argv`, which the base class leaves
822 ``NotImplementedError``. That is not reachable through a real run: every
823 adapter that spawns a subprocess implements it, and the ones that do not
824 (``LocalAdapter`` and the hosted-API adapters) build no argv at all —
825 they override :meth:`run` and never call this. A new subprocess adapter
826 that forgets ``build_argv`` therefore fails loudly on its first
827 invocation rather than running with an empty command line.
828 """
829 if policy is None or not getattr(policy, "write", False):
830 return self.build_argv(prompt)
831 return self.build_write_argv(prompt)
833 def _stdin_for(self, prompt: str) -> str | None:
834 """Prompt to feed on stdin, or None to pass it in argv (the default)."""
835 del prompt
836 return None
838 def _text_from_stdout(self, raw: str) -> str:
839 """The reviewer's prose, given the CLI's raw stdout.
841 Plain text for every adapter but one. `agy` speaks NDJSON events, so it
842 overrides this rather than teaching `run` about event streams.
843 """
844 return raw
846 def _version_argv(self) -> list[str]:
847 """Argv used to probe the CLI's version."""
848 return [self.spec.command, *self._VERSION_ARGS]
850 def effort_plan(self) -> EffortPlan:
851 """This agent's resolved effort mapping (see :func:`effort_args`).
853 An invalid level degrades to a no-op plan here: a run must not crash on
854 a bad config value that ``validate_config`` is responsible for rejecting.
855 """
856 try:
857 return effort_args(
858 config_module.spec_adapter(self.spec),
859 getattr(self.spec, "effort", None),
860 self.spec.model,
861 )
862 except ValueError:
863 return EffortPlan()
865 def resolved_model(self) -> str:
866 """The model id this adapter sends, byte-for-byte (issue #709).
868 **The** answer to "which model was asked for", and the only one: every
869 place a model id leaves this module — the ``--model``/``-m`` argv of the
870 three CLI adapters, the ``model`` field of every network payload, the
871 Gemini URL — reads it from here, and :attr:`AgentResult.model` records
872 what it returned so a ballot can quote the id instead of deriving a
873 second one.
875 That second derivation was the defect. :func:`effort_args` is keyed on
876 the **adapter**, because how effort is expressed is a property of the
877 protocol a seat is invoked through; ``ai_jury.ballots.requested_model``
878 keyed it on the ``vendor``, and since #705 those can differ, so a seat
879 with ``vendor = google, adapter = cli`` was invoked with ``gemini-3-pro``
880 while its ballot reported ``gemini-3-pro-high`` under
881 ``model_source: requested`` — a field whose whole claim is that it is the
882 id actually sent. The subclass that consults a live model listing
883 (:class:`AgyAdapter`) overrides :meth:`effort_plan`, not this, so the
884 fallback it resolves is recorded here too.
885 """
886 return (self.effort_plan().model or self.spec.model or "").strip()
888 def list_models(self) -> list[str] | None:
889 """Model ids this agent could be pointed at, or None when unknown.
891 Diagnostics only (``jury --doctor --json``). The default is None — most
892 CLIs have no listing command — and every override is time-boxed and
893 fail-soft, because doctor must never hang or crash on a probe.
894 """
895 return None
897 def detect_capabilities(self) -> dict:
898 """Best-effort probe of this agent's version and capabilities.
900 Returns a dict shaped like::
902 {
903 "version": "<str|None>",
904 "supports_headless": bool,
905 "supports_model_selection": bool,
906 "raw_version_output": "<short str>",
907 "status": "ok|unknown_version|unavailable",
908 "warnings": [...],
909 }
911 This is intentionally fast and forgiving: it runs ``<command> --version``
912 with a SHORT timeout and swallows ALL errors (missing CLI, timeout,
913 nonzero exit, garbage output). It NEVER raises, so it is safe to call
914 from diagnostics without blocking or crashing a run.
915 """
916 caps = {
917 "version": None,
918 "supports_headless": self.SUPPORTS_HEADLESS,
919 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION,
920 "raw_version_output": "",
921 "status": CAP_UNAVAILABLE,
922 "warnings": [],
923 }
925 # Not on PATH: report unavailable without spawning a subprocess.
926 if not self.available():
927 return caps
929 try:
930 # Via _spawn so the probe also runs in its own process group and the
931 # whole group is killed on timeout (issue #303/L-1) — matching the
932 # main run path; a bare subprocess.run would orphan grandchildren.
933 proc = _spawn(self._version_argv(), None, _VERSION_PROBE_TIMEOUT)
934 except subprocess.TimeoutExpired:
935 caps["status"] = CAP_UNKNOWN_VERSION
936 caps["warnings"].append(
937 f"version probe for '{self.spec.command}' timed out after {_VERSION_PROBE_TIMEOUT}s"
938 )
939 return caps
940 except Exception as exc: # noqa: BLE001 - swallow any spawn failure
941 caps["status"] = CAP_UNKNOWN_VERSION
942 caps["warnings"].append(
943 f"version probe for '{self.spec.command}' failed: {redaction.redact(str(exc))[0]}"
944 )
945 return caps
947 raw = ((proc.stdout or "") + (proc.stderr or "")).strip()
948 caps["raw_version_output"] = redaction.redact(raw[:200])[0]
949 match = _VERSION_RE.search(raw)
950 if proc.returncode == 0 and match:
951 caps["version"] = match.group(0)
952 caps["status"] = CAP_OK
953 else:
954 caps["status"] = CAP_UNKNOWN_VERSION
955 caps["warnings"].append(
956 f"could not determine version of '{self.spec.command}' "
957 f"(exit {proc.returncode}); capabilities assumed from vendor defaults"
958 )
959 return caps
961 def run(
962 self,
963 prompt: str,
964 phase: str = "review",
965 timeout: int | None = None,
966 role_policy=None,
967 ) -> AgentResult:
968 del phase
969 if not self.available():
970 return AgentResult(
971 self.name,
972 self.spec.vendor,
973 False,
974 "",
975 0.0,
976 f"command not found on PATH: {self.spec.command}",
977 error_code=ERR_MISSING_CLI,
978 )
979 # The effective timeout is the caller's override (the run budget, issue
980 # #30) when smaller than the agent's own bound, else the agent timeout.
981 effective_timeout = self.spec.timeout
982 if timeout is not None:
983 effective_timeout = max(1, min(self.spec.timeout, int(timeout)))
984 argv = self.build_argv_for_role(prompt, role_policy)
985 stdin = self._stdin_for(prompt)
986 start = time.monotonic()
987 try:
988 # A command is a bare name or an absolute path (config validation
989 # refuses a relative one, #293/F-6), so moving the directory cannot
990 # change which binary runs.
991 read_only = not getattr(role_policy, "write", False)
992 with _review_workdir(self.ISOLATE_REVIEW_CWD and read_only) as cwd:
993 if cwd is None:
994 proc = _spawn(argv, stdin, effective_timeout)
995 else:
996 proc = _spawn(argv, stdin, effective_timeout, cwd=cwd)
997 except subprocess.TimeoutExpired:
998 return AgentResult(
999 self.name,
1000 self.spec.vendor,
1001 False,
1002 "",
1003 time.monotonic() - start,
1004 f"timed out after {effective_timeout}s",
1005 error_code=ERR_TIMEOUT,
1006 )
1007 except Exception as exc: # noqa: BLE001 - surface any spawn failure
1008 return AgentResult(
1009 self.name,
1010 self.spec.vendor,
1011 False,
1012 "",
1013 time.monotonic() - start,
1014 f"spawn failed: {redaction.redact(str(exc))[0]}",
1015 error_code=ERR_SPAWN_FAILED,
1016 )
1017 dur = time.monotonic() - start
1018 out = self._text_from_stdout((proc.stdout or "").strip()).strip()
1019 # A nonzero exit is ALWAYS a failure, even with stdout (issue #101): a
1020 # crashing CLI can still print partial or error output, and counting that
1021 # as a clean review would silently feed it into consensus, synthesis, and
1022 # the CI gate. We classify from stderr (falling back to any stdout) and
1023 # keep a short snippet in the error for debugging — but ok=False, so the
1024 # orchestrator excludes it.
1025 if proc.returncode != 0:
1026 stderr = (proc.stderr or "").strip()
1027 detail = stderr or out
1028 # Redact before embedding in the error: a crashing CLI can dump an
1029 # env var / token into its stderr, and this string is rendered into
1030 # the report and posted to the PR. Mirrors the LocalAdapter path
1031 # (#293/F-8); the asymmetry was a secret-leak vector (audit
1032 # 2026-06-13/N-1). Classify on the raw text (no secrets in codes).
1033 safe_detail = redaction.redact(detail)[0]
1034 return AgentResult(
1035 self.name,
1036 self.spec.vendor,
1037 False,
1038 "",
1039 dur,
1040 f"exit {proc.returncode}: {safe_detail[:500]}",
1041 error_code=classify_stderr(proc.returncode, stderr or out),
1042 exit_code=proc.returncode,
1043 )
1044 if not out:
1045 # Exit 0 but nothing on stdout: the agent produced no usable review.
1046 return AgentResult(
1047 self.name,
1048 self.spec.vendor,
1049 False,
1050 "",
1051 dur,
1052 f"exit {proc.returncode}: empty output",
1053 error_code=ERR_EMPTY_OUTPUT,
1054 exit_code=proc.returncode,
1055 )
1056 # Exit 0 with output that is not a review (issue #682): a refusal, or the
1057 # CLI's own usage/version banner because the argv never reached the model
1058 # (#635). Counting it as a review is what makes a panel collapse silent.
1059 no_review = no_review_reason(out)
1060 if no_review is not None:
1061 return AgentResult(
1062 self.name,
1063 self.spec.vendor,
1064 False,
1065 "",
1066 dur,
1067 f"no review returned: {no_review}: {redaction.redact(out)[0][:200]}",
1068 error_code=ERR_NO_REVIEW,
1069 exit_code=proc.returncode,
1070 )
1071 return AgentResult(self.name, self.spec.vendor, True, out, dur, exit_code=proc.returncode)
1074class ClaudeAdapter(Adapter):
1075 # The prompt embeds the (redacted) diff and PR/issue context; deliver it on
1076 # STDIN rather than as a process argument so it is not exposed in `ps` /
1077 # /proc/<pid>/cmdline to other local users (issue #287). `claude -p` reads
1078 # the prompt from stdin when no positional prompt is given.
1079 ISOLATE_REVIEW_CWD = True
1081 def _head_argv(self) -> list[str]:
1082 argv = [self.spec.command, "-p"]
1083 model = self.resolved_model()
1084 if model:
1085 argv += ["--model", model]
1086 return argv
1088 def build_argv(self, prompt: str) -> list[str]:
1089 del prompt
1090 return self._head_argv() + _read_only_extra_args(self.spec)
1092 def build_write_argv(self, prompt: str) -> list[str]:
1093 """Implementer invocation: the CLI's own tool set, no deny list (#661)."""
1094 del prompt
1095 return self._head_argv() + _write_extra_args(self.spec)
1097 def _stdin_for(self, prompt: str) -> str | None:
1098 return prompt
1101class CodexAdapter(Adapter):
1102 # Pipe the prompt on stdin (not positionally) so ``codex exec`` never blocks
1103 # waiting for input in non-interactive runs. Sandbox flags live in extra_args;
1104 # the shipped default is ``-s read-only`` (secure by default, #100) — the
1105 # reviewer only reads its prompt, since the jury fetches the diff via ``gh``.
1106 ISOLATE_REVIEW_CWD = True
1108 #: `codex exec` refuses to start outside a git repository ("Not inside a
1109 #: trusted directory and --skip-git-repo-check was not specified", verified
1110 #: against codex-cli 0.155.0), and a panel invocation starts in an empty
1111 #: temporary directory. The check keeps an agent out of a directory nobody
1112 #: vouched for; a reviewer is kept in check by its read-only sandbox instead,
1113 #: which ``privilege.audit_agent`` reports on, and the directory is empty.
1114 _SKIP_GIT_CHECK = "--skip-git-repo-check"
1116 def _head_argv(self) -> list[str]:
1117 argv = [self.spec.command, "exec"]
1118 model = self.resolved_model()
1119 if model:
1120 argv += ["-m", model]
1121 return argv
1123 #: ``codex exec --ephemeral`` ("Run without persisting session files to
1124 #: disk", codex-cli 0.155.0): a reviewer's session holds the untrusted diff
1125 #: and is never resumed.
1126 _EPHEMERAL = "--ephemeral"
1128 def build_argv(self, prompt: str) -> list[str]:
1129 del prompt
1130 extra = _read_only_extra_args(self.spec)
1131 # Only an option counts as present: after `--` it is prompt text (#908 review).
1132 options = privilege._codex_options(extra)
1133 added = [f for f in (self._SKIP_GIT_CHECK, self._EPHEMERAL) if f not in options]
1134 return self._head_argv() + added + extra
1136 def build_write_argv(self, prompt: str) -> list[str]:
1137 """Implementer invocation: ``-s workspace-write`` instead of read-only (#661)."""
1138 del prompt
1139 return self._head_argv() + _write_extra_args(self.spec)
1141 def _stdin_for(self, prompt: str) -> str | None:
1142 return prompt
1145class AgyAdapter(Adapter):
1146 # Prompt on STDIN, not argv (issue #287): the redacted diff must not appear in
1147 # the process list, where any local user can read it.
1148 #
1149 # `agy --print` used to read the prompt from stdin (verified against 1.0.6).
1150 # On 1.1.x `--print` takes a value, so that invocation dies before the model
1151 # is reached — `flag needs an argument: -print` with an empty `extra_args`,
1152 # and "took --dangerously-skip-permissions as its prompt" with a non-empty
1153 # one (#635). The agent passed every availability check and contributed
1154 # nothing to the panel.
1155 #
1156 # The obvious repair — put the prompt in argv — silently undoes #287. So the
1157 # prompt moves to agy's own stdin channel instead: `--input-format
1158 # stream-json` reads one NDJSON message per line and requires
1159 # `--output-format stream-json`. `--print` is not passed at all; the input
1160 # format implies print mode, and passing it would reintroduce the arity
1161 # problem. Verified end to end against agy 1.1.22.
1162 _STREAM_ARGS = ("--input-format", "stream-json", "--output-format", "stream-json")
1164 # A panel review starts in an empty temporary directory; agy 1.2.9 answers
1165 # from one (checked by hand against the installed CLI).
1166 ISOLATE_REVIEW_CWD = True
1168 # `agy models` lists the model ids the CLI can be pointed at.
1169 _MODELS_ARGS = ("models",)
1171 def effort_plan(self) -> EffortPlan:
1172 """agy's effort IS the model id, so check the mapping against its listing.
1174 The listing is probed at most once per adapter instance, and only when an
1175 effort level is actually configured — a run without one pays nothing,
1176 which matters because this is on the per-invocation path.
1177 """
1178 if not getattr(self.spec, "effort", None):
1179 return EffortPlan()
1180 try:
1181 return effort_args(
1182 config_module.spec_adapter(self.spec),
1183 self.spec.effort,
1184 self.spec.model,
1185 known_models=self._cached_model_listing(),
1186 )
1187 except ValueError:
1188 return EffortPlan()
1190 def _cached_model_listing(self) -> list[str] | None:
1191 """``agy models``, memoized per adapter instance."""
1192 if not hasattr(self, "_model_listing"):
1193 self._model_listing = self.list_models()
1194 return self._model_listing
1196 def _head_argv(self) -> list[str]:
1197 argv = [self.spec.command, *self._STREAM_ARGS]
1198 # agy encodes reasoning effort in the model id (`…-flash` -> `…-flash-high`),
1199 # so effort changes WHICH model is selected rather than adding a flag.
1200 model = self.resolved_model()
1201 if model:
1202 argv += ["--model", model]
1203 return argv
1205 def build_argv(self, prompt: str) -> list[str]:
1206 del prompt
1207 return self._head_argv() + _read_only_extra_args(self.spec)
1209 def build_write_argv(self, prompt: str) -> list[str]:
1210 """Implementer invocation: drop the boolean ``--sandbox`` (#661)."""
1211 del prompt
1212 return self._head_argv() + _write_extra_args(self.spec)
1214 def list_models(self) -> list[str] | None:
1215 """Model ids from ``agy models`` (time-boxed, fail-soft; issue #662)."""
1216 if not self.available():
1217 return None
1218 try:
1219 proc = _spawn([self.spec.command, *self._MODELS_ARGS], None, _VERSION_PROBE_TIMEOUT)
1220 except Exception: # noqa: BLE001 - discovery is best-effort
1221 return None
1222 if proc.returncode != 0:
1223 return None
1224 return parse_model_list(proc.stdout or "") or None
1226 def _stdin_for(self, prompt: str) -> str | None:
1227 """One NDJSON frame carrying the prompt. Shape verified against 1.1.22."""
1228 return json.dumps({"event": "user", "message": {"role": "user", "content": prompt}}) + "\n"
1230 def _text_from_stdout(self, raw: str) -> str:
1231 """The `result` event's response, out of the NDJSON stream.
1233 Falls back to the raw stream when no `result` frame is present rather
1234 than returning empty: a truncated stream should surface as an
1235 unparseable review the operator can read, not as a silent abstention —
1236 an empty review is counted as one, and #625 exists because an
1237 abstention read as an approval is the expensive failure.
1238 """
1239 response, saw_result = None, False
1240 for line in raw.splitlines():
1241 line = line.strip()
1242 if not line:
1243 continue
1244 try:
1245 event = json.loads(line)
1246 except ValueError:
1247 continue
1248 if isinstance(event, dict) and event.get("event") == "result":
1249 saw_result = True
1250 result = event.get("result")
1251 if isinstance(result, dict): 1251 ↛ 1240line 1251 didn't jump to line 1240 because the condition on line 1251 was always true
1252 response = result.get("response")
1253 if saw_result and isinstance(response, str):
1254 return response
1255 return raw
1258_DEFAULT_LOCAL_ENDPOINT = "http://localhost:11434/v1"
1261def _http_only_opener():
1262 """An opener that handles ONLY http/https (issue #291, SSRF defense).
1264 The default ``urllib`` opener honors ``file://`` and ``ftp://``, so an
1265 attacker-influenced ``endpoint`` could read local files or reach other
1266 schemes. This OpenerDirector registers no ``FileHandler``/``FTPHandler``, so
1267 any non-http(s) URL raises ``URLError("unknown url type")`` regardless of
1268 config validation — defense in depth alongside ``config._endpoint_issues``.
1270 It also registers NO ``HTTPRedirectHandler`` (review of #291): otherwise a
1271 malicious/compromised endpoint could 302-redirect to an internal/metadata
1272 host (e.g. ``169.254.169.254``) and the opener would follow it, bypassing the
1273 configured-URL validation. Without the handler a 3xx surfaces as an
1274 ``HTTPError`` (a failed review) and is never followed.
1275 """
1276 import urllib.request
1278 opener = urllib.request.OpenerDirector()
1279 for handler in (
1280 urllib.request.HTTPHandler,
1281 urllib.request.HTTPSHandler,
1282 urllib.request.HTTPDefaultErrorHandler,
1283 urllib.request.HTTPErrorProcessor,
1284 # UnknownHandler raises URLError("unknown url type: …") for any scheme
1285 # without a registered handler — so file://, ftp://, etc. fail loudly
1286 # instead of silently resolving to None.
1287 urllib.request.UnknownHandler,
1288 ):
1289 opener.add_handler(handler())
1290 return opener
1293def _open(target, timeout):
1294 """Open an http/https URL or Request via the restricted opener (issue #291).
1296 Single seam for every local-adapter HTTP call so the SSRF-safe opener (no
1297 file/ftp handlers) is always used.
1298 """
1299 return _http_only_opener().open(target, timeout=timeout)
1302def list_local_models(endpoint: str = _DEFAULT_LOCAL_ENDPOINT) -> list[str]:
1303 """Model ids a local server lists, or ``[]`` — see :func:`local_model_listing`.
1305 ``[]`` means both "the server lists none" and "the listing failed". Callers that
1306 must tell those apart (the doctor: a server with nothing pulled cannot review)
1307 use :func:`local_model_listing`, which answers ``None`` for a failed listing.
1308 """
1309 return local_model_listing(endpoint) or []
1312def local_model_listing(endpoint: str = _DEFAULT_LOCAL_ENDPOINT) -> list[str] | None:
1313 """List model ids from a local OpenAI-compatible server (issue #109).
1315 GETs ``{endpoint}/models`` (the OpenAI-compatible listing that Ollama,
1316 vLLM, LM Studio, etc. expose) and returns the model ids in their reported
1317 order. Best-effort and stdlib-only: any failure (server down, bad JSON)
1318 returns ``None`` — distinct from ``[]``, a server that answered and lists no
1319 model — so callers can fall back gracefully.
1321 The endpoint is validated here at the seam (issue #309) so EVERY caller —
1322 including the un-gated ``jury init --local-endpoint`` discovery path — gets
1323 the same SSRF gate that ``config._endpoint_issues`` enforces for config-file
1324 endpoints: a non-``http(s)`` scheme or a non-loopback host (without the
1325 ``JURY_ALLOW_REMOTE_ENDPOINT`` opt-in) yields ``None`` without any network call.
1326 """
1327 import json as _json
1329 from .config import _endpoint_issues
1331 base = (endpoint or _DEFAULT_LOCAL_ENDPOINT).rstrip("/")
1332 try:
1333 # SSRF gate INSIDE the try (review of #309): `_endpoint_issues` calls
1334 # urlsplit, which raises ValueError on a malformed URL (e.g. `http://[::1`);
1335 # keep the best-effort "any failure -> None" contract rather than crashing.
1336 if _endpoint_issues(base, "local-endpoint")[0]: # hard-error issues -> refuse
1337 return None
1338 url = base if base.endswith("/models") else f"{base}/models"
1339 with _open(url, _VERSION_PROBE_TIMEOUT) as resp: # noqa: S310
1340 data = _json.loads(resp.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace"))
1341 except Exception: # noqa: BLE001 - discovery is best-effort
1342 return None
1343 if not isinstance(data, dict) or "data" not in data:
1344 return None
1345 models = data["data"]
1346 if models is None:
1347 # Ollama with nothing pulled answers `{"object": "list", "data": null}`, not
1348 # `"data": []` (measured on Ollama 0.34.1) — the very server #849 is about.
1349 return []
1350 if not isinstance(models, list):
1351 return None
1352 ids = [m.get("id") for m in models if isinstance(m, dict) and m.get("id")]
1353 return [str(i) for i in ids]
1356class LocalAdapter(Adapter):
1357 """Open-weight / local-model reviewer over an OpenAI-compatible API (issue #43).
1359 Targets the ``/v1/chat/completions`` endpoint exposed by common local servers
1360 (Ollama, llama.cpp ``llama-server``, vLLM, LM Studio). It talks plain HTTP via
1361 the stdlib (``urllib``) — no new dependencies and no subprocess — so one panel
1362 seat can run free and fully offline, adding model diversity (the load-bearing
1363 advantage) at zero marginal cost.
1365 Configure as a normal ``[[agent]]`` with ``vendor = "local"``, an
1366 ``endpoint`` (base URL, default ``http://localhost:11434/v1``), and a
1367 ``model``. ``extra_args`` is unused. An unreachable server fails with the
1368 typed ``connection_error`` code (issue #29) rather than a crash.
1370 ``temperature`` is sent as configured, else the greedy ``0``. Greedy is the
1371 right default for a reviewer, but some models do not survive it: gpt-oss
1372 loops in its reasoning at 0 until its output cap and never answers.
1373 """
1375 SUPPORTS_HEADLESS = True
1376 SUPPORTS_MODEL_SELECTION = True
1377 SPAWNS_PROCESS = False # plain HTTP, no subprocess
1379 @property
1380 def endpoint(self) -> str:
1381 return (self.spec.endpoint or _DEFAULT_LOCAL_ENDPOINT).rstrip("/")
1383 def completions_url(self) -> str:
1384 """Resolve the chat-completions URL from the configured base endpoint.
1386 Accepts either a base URL (``…/v1``) or a full completions URL; pure so it
1387 can be unit-tested without network.
1388 """
1389 base = self.endpoint
1390 if base.endswith("/chat/completions"):
1391 return base
1392 return f"{base}/chat/completions"
1394 def build_payload(self, prompt: str) -> dict:
1395 """Build the OpenAI-compatible chat-completions request body (pure)."""
1396 return {
1397 "model": self.resolved_model(),
1398 "messages": [{"role": "user", "content": prompt}],
1399 "stream": False,
1400 # Unset keeps the literal `0` every earlier release sent.
1401 "temperature": 0 if self.spec.temperature is None else self.spec.temperature,
1402 }
1404 @staticmethod
1405 def parse_content(data: dict) -> str:
1406 """Extract the assistant message text from a chat-completions response."""
1407 choices = data.get("choices") or []
1408 if not choices:
1409 return ""
1410 message = choices[0].get("message") or {}
1411 return (message.get("content") or "").strip()
1413 @staticmethod
1414 def classify_http_status(status: int) -> str:
1415 """Map an HTTP error status to a typed error code (issue #29)."""
1416 if status in (401, 403):
1417 return ERR_AUTH_REQUIRED
1418 if status == 429:
1419 return ERR_RATE_LIMITED
1420 return ERR_NONZERO_EXIT
1422 def available(self) -> bool:
1423 """A local agent is 'available' when its server answers a quick probe.
1425 Probes the OpenAI-compatible ``/v1/models`` (or the endpoint root) with a
1426 short timeout. Network-only; never raises.
1427 """
1428 # Only `urllib.error` here: the request itself goes through `_open`, whose
1429 # `_http_only_opener` imports `urllib.request` for it.
1430 import urllib.error
1432 url = f"{self.endpoint}/models"
1433 try:
1434 with _open(url, _VERSION_PROBE_TIMEOUT) as resp: # noqa: S310
1435 return 200 <= resp.status < 500
1436 except urllib.error.HTTPError as exc:
1437 # A 4xx (e.g. 404 on /models) still means the server is up.
1438 return exc.code < 500
1439 except Exception: # noqa: BLE001 - unreachable server -> not available
1440 return False
1442 def list_models(self) -> list[str] | None:
1443 """Model ids the local server advertises, or None when it has none."""
1444 return list_local_models(self.endpoint) or None
1446 def detect_capabilities(self) -> dict:
1447 reachable = self.available()
1448 return {
1449 "version": None,
1450 "supports_headless": self.SUPPORTS_HEADLESS,
1451 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION,
1452 "raw_version_output": f"local endpoint {self.endpoint}",
1453 "status": CAP_OK if reachable else CAP_UNAVAILABLE,
1454 "warnings": ([] if reachable else [f"local server unreachable at {self.endpoint}"]),
1455 }
1457 def run(
1458 self,
1459 prompt: str,
1460 phase: str = "review",
1461 timeout: int | None = None,
1462 role_policy=None,
1463 ) -> AgentResult:
1464 import json as _json
1465 import urllib.error
1466 import urllib.request
1468 del phase
1469 # A network adapter has no tools, no shell and no filesystem, so a
1470 # write-enabled role changes nothing about how it is invoked (#661).
1471 del role_policy
1472 effective_timeout = self.spec.timeout
1473 if timeout is not None:
1474 effective_timeout = max(1, min(self.spec.timeout, int(timeout)))
1475 body = _json.dumps(self.build_payload(prompt)).encode("utf-8")
1476 req = urllib.request.Request(
1477 self.completions_url(),
1478 data=body,
1479 headers={"Content-Type": "application/json"},
1480 method="POST",
1481 )
1482 start = time.monotonic()
1483 try:
1484 with _open(req, effective_timeout) as resp: # noqa: S310
1485 raw = resp.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace")
1486 data = _json.loads(raw)
1487 except urllib.error.HTTPError as exc:
1488 detail = ""
1489 try:
1490 detail = exc.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace")[:300]
1491 except Exception: # noqa: BLE001
1492 detail = exc.reason or ""
1493 # The body is from a possibly-untrusted endpoint and is surfaced in
1494 # the report; redact recognized secrets before embedding (#293/F-8).
1495 detail = redaction.redact(detail)[0]
1496 return AgentResult(
1497 self.name,
1498 self.spec.vendor,
1499 False,
1500 "",
1501 time.monotonic() - start,
1502 f"HTTP {exc.code}: {detail}",
1503 error_code=self.classify_http_status(exc.code),
1504 )
1505 except TimeoutError:
1506 return AgentResult(
1507 self.name,
1508 self.spec.vendor,
1509 False,
1510 "",
1511 time.monotonic() - start,
1512 f"timed out after {effective_timeout}s",
1513 error_code=ERR_TIMEOUT,
1514 )
1515 except urllib.error.URLError as exc:
1516 return AgentResult(
1517 self.name,
1518 self.spec.vendor,
1519 False,
1520 "",
1521 time.monotonic() - start,
1522 f"could not reach local server at {self.endpoint}: {redaction.redact(str(exc.reason))[0]}",
1523 error_code=ERR_CONNECTION,
1524 )
1525 except Exception as exc: # noqa: BLE001 - surface any other failure
1526 return AgentResult(
1527 self.name,
1528 self.spec.vendor,
1529 False,
1530 "",
1531 time.monotonic() - start,
1532 f"local request failed: {redaction.redact(str(exc))[0]}",
1533 error_code=ERR_UNKNOWN,
1534 )
1535 dur = time.monotonic() - start
1536 content = self.parse_content(data)
1537 if not content:
1538 return AgentResult(
1539 self.name,
1540 self.spec.vendor,
1541 False,
1542 "",
1543 dur,
1544 "local model returned empty content",
1545 error_code=ERR_EMPTY_OUTPUT,
1546 )
1547 no_review = no_review_reason(content)
1548 if no_review is not None:
1549 return AgentResult(
1550 self.name,
1551 self.spec.vendor,
1552 False,
1553 "",
1554 dur,
1555 f"no review returned: {no_review}",
1556 error_code=ERR_NO_REVIEW,
1557 )
1558 return AgentResult(self.name, self.spec.vendor, True, content, dur)
1561# Hosted vendor API endpoints (issue #430). Fixed, not configurable: unlike
1562# `local`'s user-supplied `endpoint` (which needs the SSRF validation in
1563# config._endpoint_issues), a hosted vendor's URL is a known constant, not an
1564# attacker- or operator-influenceable value, so there is nothing to validate.
1565_ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages"
1566_ANTHROPIC_API_VERSION = "2023-06-01"
1567_OPENAI_API_URL = "https://api.openai.com/v1/chat/completions"
1568_XAI_API_URL = "https://api.x.ai/v1/chat/completions"
1569# The Anthropic Messages API requires max_tokens on every request; there is no
1570# server-side default. Generous enough for a review response, small enough to
1571# bound cost/latency if a run is ever misconfigured to loop.
1572_HOSTED_API_MAX_TOKENS = 4096
1575def _hosted_api_status_code(status: int) -> str:
1576 """Map a hosted-API HTTP status to a typed error code (issue #430).
1578 Shared by every hosted-API adapter — identical mapping to
1579 ``LocalAdapter.classify_http_status`` (401/403 → auth, 429 → rate limit),
1580 kept as a free function since it has no per-adapter state.
1581 """
1582 if status in (401, 403):
1583 return ERR_AUTH_REQUIRED
1584 if status == 429:
1585 return ERR_RATE_LIMITED
1586 return ERR_NONZERO_EXIT
1589def _post_json(
1590 url: str, payload: dict, headers: dict[str, str], timeout: int
1591) -> tuple[dict | None, str | None, str | None]:
1592 """POST a JSON body and parse a JSON response (issue #430).
1594 Shared HTTP mechanics for the hosted-API adapters: build the request,
1595 route it through the SSRF-safe opener (``_open``, no file/ftp handlers, no
1596 redirect following — the same seam ``LocalAdapter`` uses), cap the response
1597 read at ``_MAX_RESPONSE_BYTES``, and classify any failure into a typed
1598 error code. Returns ``(response_dict, None, None)`` on success or
1599 ``(None, error_message, error_code)`` on failure — exactly one shape.
1600 Response bodies are redacted before being returned in an error message
1601 since they originate from the network and are surfaced in the report.
1602 """
1603 import json as _json
1604 import urllib.error
1605 import urllib.request
1607 body = _json.dumps(payload).encode("utf-8")
1608 req = urllib.request.Request(url, data=body, headers=headers, method="POST")
1609 try:
1610 with _open(req, timeout) as resp: # noqa: S310
1611 raw = resp.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace")
1612 return _json.loads(raw), None, None
1613 except urllib.error.HTTPError as exc:
1614 detail = ""
1615 try:
1616 detail = exc.read(_MAX_RESPONSE_BYTES).decode("utf-8", errors="replace")[:300]
1617 except Exception: # noqa: BLE001 - reading the error body is best-effort
1618 detail = exc.reason or ""
1619 detail = redaction.redact(detail)[0]
1620 return None, f"HTTP {exc.code}: {detail}", _hosted_api_status_code(exc.code)
1621 except TimeoutError:
1622 return None, f"timed out after {timeout}s", ERR_TIMEOUT
1623 except urllib.error.URLError as exc:
1624 return (
1625 None,
1626 f"could not reach {url}: {redaction.redact(str(exc.reason))[0]}",
1627 ERR_CONNECTION,
1628 )
1629 except Exception as exc: # noqa: BLE001 - surface any other failure
1630 return None, f"request failed: {redaction.redact(str(exc))[0]}", ERR_UNKNOWN
1633class _HostedApiAdapter(Adapter):
1634 """Shared base for hosted-vendor-API reviewers keyed by an env-var API key.
1636 No CLI install, no interactive login, no subprocess: just an HTTP call
1637 over stdlib ``urllib`` to the vendor's real hosted API (issue #430), the
1638 same no-subprocess/no-new-dependency design as ``LocalAdapter`` but
1639 pointed at a hosted endpoint instead of a local server. The API key is
1640 read from the environment ONLY, never from ``jury.toml``, so it cannot
1641 leak into a checked-in config; the endpoint is a fixed per-vendor
1642 constant, not a config value, so there is no SSRF surface to guard the
1643 way `local`'s `endpoint` needs.
1644 """
1646 SUPPORTS_HEADLESS = True
1647 SUPPORTS_MODEL_SELECTION = True
1648 SPAWNS_PROCESS = False # plain HTTP, no subprocess
1650 # The environment variable this vendor's credential is read FROM. A name,
1651 # never a value — deliberately not called `_API_KEY_*`: the constant holds
1652 # public configuration, and a credential-shaped name on a non-credential is
1653 # how both a reader and a static analyzer end up misreading this path.
1654 # Subclasses override.
1655 _ENV_VAR_NAME: str = "OPENAI_API_KEY"
1657 def _env_var_name(self) -> str:
1658 """Name of the environment variable holding this agent's credential.
1660 Rebuilt through :func:`redaction.safe_env_var_name`, because this value
1661 is *displayed* — it reaches ``jury --doctor``, its JSON export, and
1662 warning text — while ``[[agent]] api_key_env`` is an arbitrary operator
1663 string. The sanitizer both bounds it to a real env var name (so a config
1664 value cannot splice a newline into a JSON document) and severs it from
1665 the credential-shaped config field it came from. The credential VALUE
1666 never travels this way; see :meth:`_api_key`.
1667 """
1668 return redaction.safe_env_var_name(
1669 getattr(self.spec, "api_key_env", None), self._ENV_VAR_NAME
1670 )
1672 def _api_key(self) -> str:
1673 """The credential itself. NEVER rendered — only compared and sent.
1675 Kept under a deliberately sensitive name so any future flow from here
1676 into a log or an export is reported rather than blending in.
1677 """
1678 return os.environ.get(self._env_var_name(), "")
1680 def _api_url(self) -> str: # pragma: no cover - overridden
1681 raise NotImplementedError
1683 def _invalid_key_reason(self) -> str | None:
1684 """None if the key is safe to use as an HTTP header value; else why not.
1686 A key containing a control character (most plausibly a stray
1687 trailing ``\\n`` from a file/k8s-secret/`.env` mount) trips CPython's
1688 ``http.client`` header-injection guard. That guard reports the
1689 rejected value via ``repr()`` (e.g. an embedded newline becomes the
1690 two literal characters ``\\`` ``n``), which does **not** byte-for-byte
1691 match the raw key — so a literal substring scrub of the exception
1692 text (see :meth:`_scrub_secret`) cannot reliably catch it; the
1693 transformed text is no longer equal to the original secret. Validate
1694 and reject *before* the key ever reaches a header instead of trying
1695 to scrub it back out afterward.
1696 """
1697 if any(ord(ch) < 0x20 or ord(ch) == 0x7F for ch in self._api_key()):
1698 return (
1699 f"{self._env_var_name()} contains a control character (e.g. a stray "
1700 f"trailing newline from how the secret was loaded) and cannot be "
1701 f"used as an HTTP header value"
1702 )
1703 return None
1705 def _scrub_secret(self, text: str) -> str:
1706 """Strip the literal API key value from an error message (issue #430).
1708 Defense-in-depth alongside ``redaction.redact()`` (which only
1709 recognizes known vendor-token *shapes* via regex) for any leak path
1710 NOT already ruled out by :meth:`_invalid_key_reason` — e.g. a
1711 well-formed key that still ends up quoted in some other library's
1712 error text. Not a substitute for that check: once a value contains
1713 control characters, downstream formatting (``repr()``, percent-
1714 encoding, ...) can transform it before it reaches an error message,
1715 and a literal match against the *original* key would then silently
1716 miss it — which is exactly why control characters are rejected
1717 upfront in :meth:`run` instead of relying on this alone.
1718 """
1719 key = self._api_key()
1720 if key and key in text:
1721 return text.replace(key, "[REDACTED]")
1722 return text
1724 def available(self) -> bool:
1725 """Available when a *usable* API key is set — a fast, network-free check.
1727 Unlike ``LocalAdapter.available()`` (which probes the server, since a
1728 local endpoint's reachability is genuinely uncertain), a hosted
1729 vendor's API is assumed reachable; the two real unknowns locally are
1730 whether the operator configured a key at all, and whether it's
1731 actually usable as a header value (see :meth:`_invalid_key_reason`) —
1732 a key that will be rejected by :meth:`run` should not report as
1733 available here either, or a capability check (``jury --doctor``)
1734 would give a falsely reassuring answer.
1735 """
1736 return bool(self._api_key()) and self._invalid_key_reason() is None
1738 def detect_capabilities(self) -> dict:
1739 key_set = bool(self._api_key())
1740 invalid_reason = self._invalid_key_reason() if key_set else None
1741 has_key = key_set and invalid_reason is None
1742 if not key_set:
1743 warnings = [f"{self._env_var_name()} is not set in the environment"]
1744 elif invalid_reason:
1745 warnings = [invalid_reason]
1746 else:
1747 warnings = []
1748 return {
1749 "version": None,
1750 "supports_headless": self.SUPPORTS_HEADLESS,
1751 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION,
1752 "raw_version_output": f"hosted API {self._api_url()}",
1753 "status": CAP_OK if has_key else CAP_UNAVAILABLE,
1754 "warnings": warnings,
1755 }
1757 def build_payload(self, prompt: str) -> dict: # pragma: no cover - overridden
1758 raise NotImplementedError
1760 def _headers(self) -> dict[str, str]: # pragma: no cover - overridden
1761 raise NotImplementedError
1763 @staticmethod
1764 def parse_content(data: dict) -> str: # pragma: no cover - overridden
1765 raise NotImplementedError
1767 def run(
1768 self,
1769 prompt: str,
1770 phase: str = "review",
1771 timeout: int | None = None,
1772 role_policy=None,
1773 ) -> AgentResult:
1774 del phase
1775 # A network adapter has no tools, no shell and no filesystem, so a
1776 # write-enabled role changes nothing about how it is invoked (#661).
1777 del role_policy
1778 # Checked independently of available() (not just "not available()"):
1779 # available() now also returns False for a key that IS set but
1780 # invalid, and that case needs its own distinct error_code/message
1781 # below rather than the misleading "is not set" one.
1782 if not self._api_key():
1783 return AgentResult(
1784 self.name,
1785 self.spec.vendor,
1786 False,
1787 "",
1788 0.0,
1789 f"{self._env_var_name()} is not set in the environment",
1790 error_code=ERR_MISSING_API_KEY,
1791 )
1792 invalid_reason = self._invalid_key_reason()
1793 if invalid_reason is not None:
1794 # Reject before the key ever reaches a header — see
1795 # _invalid_key_reason for why post-hoc scrubbing can't be trusted
1796 # here. This message never echoes the key itself.
1797 return AgentResult(
1798 self.name,
1799 self.spec.vendor,
1800 False,
1801 "",
1802 0.0,
1803 invalid_reason,
1804 error_code=ERR_INVALID_API_KEY,
1805 )
1806 effective_timeout = self.spec.timeout
1807 if timeout is not None:
1808 effective_timeout = max(1, min(self.spec.timeout, int(timeout)))
1809 start = time.monotonic()
1810 data, err_msg, err_code = _post_json(
1811 self._api_url(), self.build_payload(prompt), self._headers(), effective_timeout
1812 )
1813 dur = time.monotonic() - start
1814 if err_msg is not None:
1815 return AgentResult(
1816 self.name,
1817 self.spec.vendor,
1818 False,
1819 "",
1820 dur,
1821 self._scrub_secret(err_msg),
1822 error_code=err_code,
1823 )
1824 content = self.parse_content(data or {})
1825 if not content:
1826 return AgentResult(
1827 self.name,
1828 self.spec.vendor,
1829 False,
1830 "",
1831 dur,
1832 "hosted API returned empty content",
1833 error_code=ERR_EMPTY_OUTPUT,
1834 )
1835 no_review = no_review_reason(content)
1836 if no_review is not None:
1837 return AgentResult(
1838 self.name,
1839 self.spec.vendor,
1840 False,
1841 "",
1842 dur,
1843 f"no review returned: {no_review}",
1844 error_code=ERR_NO_REVIEW,
1845 )
1846 return AgentResult(self.name, self.spec.vendor, True, content, dur)
1849class AnthropicApiAdapter(_HostedApiAdapter):
1850 """Hosted Anthropic Messages API reviewer, keyed by ``ANTHROPIC_API_KEY`` (issue #430).
1852 Configure as a normal ``[[agent]]`` with ``vendor = "anthropic-api"`` and a
1853 ``model`` (e.g. a current Claude model id) — no ``command``, no ``claude``
1854 CLI install or interactive login needed.
1855 """
1857 _ENV_VAR_NAME = "ANTHROPIC_API_KEY"
1859 def _api_url(self) -> str:
1860 return _ANTHROPIC_API_URL
1862 def build_payload(self, prompt: str) -> dict:
1863 """Build the Anthropic Messages API request body (pure).
1865 With an effort level configured, extended thinking is enabled with the
1866 mapped token budget. ``max_tokens`` must exceed that budget (thinking
1867 tokens are drawn from the same allowance), so it is raised to leave the
1868 original response allowance on top of it — bounded by
1869 :data:`_ANTHROPIC_MAX_TOKENS_CEILING`, since per-model ``max_tokens``
1870 caps vary and an unbounded sum would build a request some models reject.
1871 """
1872 payload = {
1873 "model": self.resolved_model(),
1874 "max_tokens": _HOSTED_API_MAX_TOKENS,
1875 "messages": [{"role": "user", "content": prompt}],
1876 }
1877 payload.update(self.effort_plan().payload)
1878 thinking = payload.get("thinking")
1879 if isinstance(thinking, dict):
1880 # Clamped here too, not only in effort_args, so `budget < max_tokens
1881 # <= ceiling` holds for any plan handed to this builder.
1882 budget = min(
1883 int(thinking["budget_tokens"]),
1884 _ANTHROPIC_MAX_TOKENS_CEILING - _HOSTED_API_MAX_TOKENS,
1885 )
1886 payload["thinking"] = {**thinking, "budget_tokens": budget}
1887 payload["max_tokens"] = budget + _HOSTED_API_MAX_TOKENS
1888 return payload
1890 def _headers(self) -> dict[str, str]:
1891 return {
1892 "Content-Type": "application/json",
1893 "x-api-key": self._api_key(),
1894 "anthropic-version": _ANTHROPIC_API_VERSION,
1895 }
1897 @staticmethod
1898 def parse_content(data: dict) -> str:
1899 """Extract the assistant text from a Messages API response."""
1900 if not isinstance(data, dict):
1901 return ""
1902 blocks = data.get("content") or []
1903 texts = [
1904 block.get("text", "")
1905 for block in blocks
1906 if isinstance(block, dict) and block.get("type") == "text"
1907 ]
1908 return "".join(texts).strip()
1911class OpenAiApiAdapter(_HostedApiAdapter):
1912 """Hosted OpenAI Chat Completions API reviewer, keyed by ``OPENAI_API_KEY`` (issue #430).
1914 Configure as a normal ``[[agent]]`` with ``vendor = "openai-api"`` and a
1915 ``model`` (e.g. a current GPT model id) — no ``command``, no ``codex`` CLI
1916 install or interactive login needed. Same request/response shape as
1917 ``LocalAdapter`` (both are OpenAI-compatible chat completions), just
1918 against the real hosted API with an ``Authorization`` header.
1919 """
1921 _ENV_VAR_NAME = "OPENAI_API_KEY"
1923 def _api_url(self) -> str:
1924 return _OPENAI_API_URL
1926 def build_payload(self, prompt: str) -> dict:
1927 """Build the OpenAI chat-completions request body (pure)."""
1928 payload = {
1929 "model": self.resolved_model(),
1930 "messages": [{"role": "user", "content": prompt}],
1931 }
1932 payload.update(self.effort_plan().payload)
1933 return payload
1935 def _headers(self) -> dict[str, str]:
1936 return {
1937 "Content-Type": "application/json",
1938 "Authorization": f"Bearer {self._api_key()}",
1939 }
1941 @staticmethod
1942 def parse_content(data: dict) -> str:
1943 """Extract the assistant message text from a chat-completions response."""
1944 if not isinstance(data, dict):
1945 return ""
1946 choices = data.get("choices") or []
1947 if not choices:
1948 return ""
1949 message = choices[0].get("message") or {}
1950 return (message.get("content") or "").strip()
1953class XaiApiAdapter(OpenAiApiAdapter):
1954 """Hosted xAI (Grok) API reviewer, keyed by ``XAI_API_KEY`` (issue #701).
1956 Configure as a normal ``[[agent]]`` with ``vendor = "xai-api"`` and a
1957 ``model`` (a Grok model id) — no ``command``, no CLI install. xAI serves
1958 the OpenAI chat-completions shape at its own host, so this is
1959 :class:`OpenAiApiAdapter` with a different URL and a different credential
1960 env var; the request body, the response parsing and the effort knob
1961 (``reasoning_effort``) are inherited rather than re-stated, because a
1962 second copy of them is a second thing to keep in step.
1964 ``vendor = "openai-compatible"`` with ``endpoint = "https://api.x.ai/v1"``
1965 still works and is still documented; this spelling exists so a Grok seat
1966 can carry the vendor identity ``xai-api`` instead of borrowing OpenAI's.
1967 """
1969 _ENV_VAR_NAME = "XAI_API_KEY"
1971 def _api_url(self) -> str:
1972 return _XAI_API_URL
1975_GEMINI_API_BASE = "https://generativelanguage.googleapis.com/v1beta/models"
1978class GoogleApiAdapter(_HostedApiAdapter):
1979 """Hosted Google Gemini API reviewer, keyed by ``GEMINI_API_KEY`` (issue #432).
1981 Configure as a normal ``[[agent]]`` with ``vendor = "google-api"`` and a
1982 ``model`` (e.g. a current Gemini model id) — no ``command``, no ``agy``
1983 CLI install or interactive login needed.
1985 Two differences from the other two hosted adapters:
1987 - The Gemini API embeds the model id in the URL **path**
1988 (``.../models/{model}:generateContent``), not the request body, so
1989 ``_api_url()`` is built from ``self.spec.model`` on every call rather
1990 than returning a fixed constant like the other two adapters.
1991 - The key is sent via the ``x-goog-api-key`` header. Gemini also accepts
1992 the key as a ``?key=...`` query parameter, but a query-string key is a
1993 much easier accidental-leak vector (proxy/access logs, anything that
1994 prints the request URL) than a header — deliberately not supported.
1996 A prompt blocked by Gemini's safety filters comes back with an empty
1997 ``candidates`` list (and a ``promptFeedback.blockReason``); this is not
1998 distinguished from a genuinely empty response and both currently surface
1999 as the same generic ``ERR_EMPTY_OUTPUT`` — a possible future refinement,
2000 not required for parity with the other two adapters.
2001 """
2003 _ENV_VAR_NAME = "GEMINI_API_KEY"
2005 def _api_url(self) -> str:
2006 # Escape the model id as a single path segment (issue #432 review): an
2007 # operator-configured model containing reserved URL characters
2008 # (`/`, `?`, `#`, ...) would otherwise change the request's path/query
2009 # semantics instead of staying a single `{model}` segment.
2010 import urllib.parse
2012 model = urllib.parse.quote(self.resolved_model(), safe="")
2013 return f"{_GEMINI_API_BASE}/{model}:generateContent"
2015 def build_payload(self, prompt: str) -> dict:
2016 """Build the Gemini ``generateContent`` request body (pure)."""
2017 payload: dict = {"contents": [{"parts": [{"text": prompt}]}]}
2018 payload.update(self.effort_plan().payload)
2019 return payload
2021 def _headers(self) -> dict[str, str]:
2022 return {
2023 "Content-Type": "application/json",
2024 "x-goog-api-key": self._api_key(),
2025 }
2027 @staticmethod
2028 def parse_content(data: dict) -> str:
2029 """Extract the assistant text from a ``generateContent`` response."""
2030 if not isinstance(data, dict):
2031 return ""
2032 candidates = data.get("candidates") or []
2033 if not candidates or not isinstance(candidates[0], dict):
2034 return ""
2035 content = candidates[0].get("content")
2036 if not isinstance(content, dict):
2037 return ""
2038 parts = content.get("parts") or []
2039 texts = [
2040 part.get("text", "")
2041 for part in parts
2042 if isinstance(part, dict) and isinstance(part.get("text", ""), str)
2043 ]
2044 return "".join(texts).strip()
2047class MockAdapter(Adapter):
2048 """Offline adapter for tests and ``--mock`` runs.
2050 Produces deterministic, phase-aware text so the full orchestration pipeline
2051 can run end-to-end without live CLIs, auth, or token spend.
2052 """
2054 # Synthetic capabilities: the mock is offline and runs no real CLI.
2055 SUPPORTS_HEADLESS = True
2056 SUPPORTS_MODEL_SELECTION = False
2058 def available(self) -> bool:
2059 return True
2061 def detect_capabilities(self) -> dict:
2062 """Deterministic fake capabilities so doctor/tests stay stable offline."""
2063 return {
2064 "version": "mock-1.0",
2065 "supports_headless": self.SUPPORTS_HEADLESS,
2066 "supports_model_selection": self.SUPPORTS_MODEL_SELECTION,
2067 "raw_version_output": "mock-1.0",
2068 "status": CAP_OK,
2069 "warnings": [],
2070 }
2072 def run(
2073 self,
2074 prompt: str,
2075 phase: str = "review",
2076 timeout: int | None = None,
2077 role_policy=None,
2078 ) -> AgentResult:
2079 del prompt, timeout, role_policy
2080 n = self.name
2081 if phase == "review":
2082 body = (
2083 # The two lines the review prompt asks every reviewer to open
2084 # with (#700). The mock speaks the shape a real reviewer is asked
2085 # for, so `--mock` exercises the scope/testing lift rather than
2086 # only the inference fallback behind it.
2087 "Checked: src/example.py\n"
2088 "Tested: nothing run (offline mock reviewer)\n"
2089 f"- **[major]** `src/example.py:42` — {n}: unchecked return value "
2090 f"may swallow an error.\n"
2091 f"- **[minor]** `src/example.py:7` — {n}: missing docstring.\n\n"
2092 "```json\n"
2093 "[\n"
2094 ' {"severity": "major", "file": "src/example.py", "line": 42, '
2095 f'"claim": "{n}: unchecked return value may swallow an error", '
2096 '"evidence": "the added code ignores the return value of int(x)", '
2097 '"suggested_fix": "check the result and raise on failure", '
2098 f'"confidence": "high", "reviewer": "{n}"}},\n'
2099 ' {"severity": "minor", "file": "src/example.py", "line": 7, '
2100 f'"claim": "{n}: missing docstring", '
2101 '"evidence": "the new function parse() has no docstring", '
2102 '"suggested_fix": "add a one-line docstring", '
2103 f'"confidence": "medium", "reviewer": "{n}"}}\n'
2104 "]\n"
2105 "```"
2106 )
2107 elif phase == "debate":
2108 body = (
2109 f"## AGREE\n- {n}: confirm the unchecked-return finding at "
2110 f"`src/example.py:42`.\n"
2111 f"## DISPUTE\n- {n}: the missing-docstring finding is a nit, not blocking.\n"
2112 f"## MISSED\n- {n}: no test covers the error branch."
2113 )
2114 elif phase == "verify":
2115 body = (
2116 "Verification: confirming the unchecked-return finding at "
2117 "`src/example.py:42`; the missing-docstring claim at `:7` is a nit "
2118 "not supported as blocking.\n\n"
2119 "```json\n"
2120 "[\n"
2121 ' {"file": "src/example.py", "line": 42, '
2122 '"claim": "unchecked return value may swallow an error", '
2123 '"status": "verified", '
2124 '"reasoning": "the added code ignores the return value of int(x)"},\n'
2125 ' {"file": "src/example.py", "line": 7, '
2126 '"claim": "missing docstring", '
2127 '"status": "unsupported", '
2128 '"reasoning": "a missing docstring is not a defect the diff introduces"}\n'
2129 "]\n"
2130 "```"
2131 )
2132 else: # synthesis
2133 body = (
2134 "## Verdict\nREQUEST CHANGES — one confirmed major issue.\n\n"
2135 "## Consensus findings\n- **[major]** `src/example.py:42` — unchecked "
2136 "return value (raised by all reviewers).\n\n"
2137 "## Disputed findings\n- Missing docstring: ruled non-blocking.\n\n"
2138 "## Notable single-reviewer findings\n- Missing test for the error branch."
2139 )
2140 return AgentResult(n, self.spec.vendor, True, body, 0.0)
2143class GenericOpenAICompatibleAdapter(_HostedApiAdapter):
2144 """Hosted OpenAI-compatible API reviewer (OpenRouter, DeepSeek, Groq, Mistral API, LiteLLM, etc.).
2146 Supports custom ``endpoint``, custom ``api_key_env``, and extra HTTP ``headers``.
2147 """
2149 _ENV_VAR_NAME = "OPENAI_API_KEY"
2151 def _api_url(self) -> str:
2152 endpoint = (self.spec.endpoint or _OPENAI_API_URL).rstrip("/")
2153 if endpoint.endswith("/chat/completions"): 2153 ↛ 2154line 2153 didn't jump to line 2154 because the condition on line 2153 was never true
2154 return endpoint
2155 return f"{endpoint}/chat/completions"
2157 def build_payload(self, prompt: str) -> dict:
2158 payload = {
2159 "model": self.resolved_model(),
2160 "messages": [{"role": "user", "content": prompt}],
2161 }
2162 payload.update(self.effort_plan().payload)
2163 return payload
2165 def _headers(self) -> dict[str, str]:
2166 hdrs = {
2167 "Content-Type": "application/json",
2168 "Authorization": f"Bearer {self._api_key()}",
2169 }
2170 if self.spec.headers: 2170 ↛ 2172line 2170 didn't jump to line 2172 because the condition on line 2170 was always true
2171 hdrs.update(self.spec.headers)
2172 return hdrs
2174 @staticmethod
2175 def parse_content(data: dict) -> str:
2176 if not isinstance(data, dict): 2176 ↛ 2177line 2176 didn't jump to line 2177 because the condition on line 2176 was never true
2177 return ""
2178 choices = data.get("choices") or []
2179 if not choices or not isinstance(choices, list):
2180 return ""
2181 first = choices[0]
2182 if not isinstance(first, dict): 2182 ↛ 2183line 2182 didn't jump to line 2183 because the condition on line 2182 was never true
2183 return ""
2184 msg = first.get("message") or {}
2185 if not isinstance(msg, dict): 2185 ↛ 2186line 2185 didn't jump to line 2186 because the condition on line 2185 was never true
2186 return ""
2187 raw_content = msg.get("content")
2188 if isinstance(raw_content, str):
2189 return raw_content.strip()
2190 if isinstance(raw_content, list): 2190 ↛ 2201line 2190 didn't jump to line 2201 because the condition on line 2190 was always true
2191 parts = []
2192 for item in raw_content:
2193 if isinstance(item, dict): 2193 ↛ 2198line 2193 didn't jump to line 2198 because the condition on line 2193 was always true
2194 if (item.get("type") == "text" or "text" in item) and isinstance( 2194 ↛ 2192line 2194 didn't jump to line 2192 because the condition on line 2194 was always true
2195 item.get("text"), str
2196 ):
2197 parts.append(item["text"])
2198 elif isinstance(item, str):
2199 parts.append(item)
2200 return "".join(parts).strip()
2201 return ""
2204class GenericCLIAdapter(Adapter):
2205 """Generic CLI adapter for arbitrary coding-agent CLIs (Aider, Goose, OpenHands, Copilot CLI, etc.).
2207 Supports configurable prompt delivery modes:
2208 - ``prompt_mode = "stdin"`` (default): prompt passed via STDIN
2209 - ``prompt_mode = "arg"``: prompt passed as positional argument on argv
2210 """
2212 def _prompt_mode(self) -> str:
2213 return (self.spec.prompt_mode or "stdin").lower()
2215 def build_argv(self, prompt: str) -> list[str]:
2216 """Read-only argv for a configured `cli` profile.
2218 Implemented rather than inherited (the base raises) so this adapter goes
2219 through the same ``build_argv_for_role`` seam as every other one — the
2220 role policy is then decided in exactly one place for every vendor.
2221 """
2222 argv = [self.spec.command, *_read_only_extra_args(self.spec)]
2223 return [*argv, prompt] if self._prompt_mode() == "arg" else argv
2225 def build_write_argv(self, prompt: str) -> list[str]:
2226 """Write-capable argv.
2228 A `cli` profile has no vendor-specific sandbox flag to add or remove, so
2229 this resolves to the same configured ``extra_args`` as the read-only
2230 argv. It is still routed through :func:`_write_extra_args` so the rule
2231 lives with every other vendor's rather than being special-cased here.
2232 """
2233 argv = [self.spec.command, *_write_extra_args(self.spec)]
2234 return [*argv, prompt] if self._prompt_mode() == "arg" else argv
2236 def _stdin_for(self, prompt: str) -> str | None:
2237 return None if self._prompt_mode() == "arg" else prompt
2239 def _argv_too_long_hint(self, exc: BaseException, prompt: str) -> str:
2240 """Why the spawn failed, when the prompt on argv made the command line too long (#901).
2242 ``prompt_mode = "arg"`` puts the whole prompt in one argument. Linux caps
2243 one argument at 128 KiB (measured: 131071 bytes spawned, 131072 failed
2244 with ``E2BIG``), below the 200000-byte ``[jury.diff] max_bytes`` default;
2245 macOS caps the whole command line at ``ARG_MAX`` (1 MiB, also ``E2BIG``);
2246 Windows at 32767 characters, which ``CreateProcess`` reports as error 206
2247 (Microsoft's documentation; not measured here). Said, so the operator is
2248 not left with a bare "Argument list too long".
2249 """
2250 too_long = isinstance(exc, OSError) and (
2251 exc.errno == errno.E2BIG or getattr(exc, "winerror", None) == 206
2252 )
2253 if not too_long or self._prompt_mode() != "arg":
2254 return ""
2255 size = len(prompt.encode("utf-8"))
2256 return (
2257 f"; the prompt ({size} bytes) is one command-line argument "
2258 f'(prompt_mode = "arg"), longer than this system allows. Use '
2259 f'prompt_mode = "stdin" if the CLI reads its prompt from stdin, or lower '
2260 f"[jury.diff] max_bytes / chunk_max_bytes."
2261 )
2263 def available(self) -> bool:
2264 command = self.spec.command or ""
2265 if not command: 2265 ↛ 2266line 2265 didn't jump to line 2266 because the condition on line 2265 was never true
2266 return False
2267 return shutil.which(command) is not None
2269 def detect_capabilities(self) -> dict:
2270 if not self.available():
2271 return {
2272 "version": None,
2273 "supports_headless": None,
2274 "supports_model_selection": None,
2275 "raw_version_output": "",
2276 "status": CAP_UNAVAILABLE,
2277 "warnings": [f"command '{self.spec.command}' not found on PATH"],
2278 }
2279 return {
2280 "version": "generic-cli",
2281 "supports_headless": True,
2282 "supports_model_selection": bool(self.spec.model),
2283 "raw_version_output": "generic-cli",
2284 "status": CAP_OK,
2285 "warnings": [],
2286 }
2288 def run(
2289 self,
2290 prompt: str,
2291 phase: str = "review",
2292 timeout: int | None = None,
2293 role_policy=None,
2294 ) -> AgentResult:
2295 del phase
2296 if not self.available():
2297 return AgentResult(
2298 agent=self.spec.name,
2299 vendor=self.spec.vendor,
2300 ok=False,
2301 output="",
2302 duration_s=0.0,
2303 error=f"command '{self.spec.command}' not found on PATH.",
2304 error_code=ERR_MISSING_CLI,
2305 )
2306 effective_timeout = timeout if timeout is not None else self.spec.timeout
2307 argv = self.build_argv_for_role(prompt, role_policy)
2308 stdin_content = self._stdin_for(prompt)
2310 start = time.monotonic()
2311 try:
2312 res = _spawn(argv, stdin_content, timeout=effective_timeout)
2313 except subprocess.TimeoutExpired:
2314 return AgentResult(
2315 self.spec.name,
2316 self.spec.vendor,
2317 False,
2318 "",
2319 effective_timeout,
2320 f"execution timed out after {effective_timeout}s.",
2321 error_code=ERR_TIMEOUT,
2322 )
2323 except Exception as exc:
2324 duration = time.monotonic() - start
2325 return AgentResult(
2326 self.spec.name,
2327 self.spec.vendor,
2328 False,
2329 "",
2330 duration,
2331 f"failed to spawn '{self.spec.command}': {redaction.redact(str(exc))[0]}"
2332 f"{self._argv_too_long_hint(exc, prompt)}",
2333 error_code=ERR_SPAWN_FAILED,
2334 )
2336 duration = time.monotonic() - start
2337 out = (res.stdout or "").strip()
2338 err = (res.stderr or "").strip()
2340 if res.returncode != 0:
2341 detail = err or out or f"exited with code {res.returncode}"
2342 safe_detail = redaction.redact(detail)[0]
2343 err_code = classify_stderr(res.returncode, err or out)
2344 return AgentResult(
2345 self.spec.name,
2346 self.spec.vendor,
2347 False,
2348 "",
2349 duration,
2350 f"exit {res.returncode}: {safe_detail[:500]}",
2351 error_code=err_code,
2352 exit_code=res.returncode,
2353 )
2355 if not out: 2355 ↛ 2356line 2355 didn't jump to line 2356 because the condition on line 2355 was never true
2356 return AgentResult(
2357 self.spec.name,
2358 self.spec.vendor,
2359 False,
2360 "",
2361 duration,
2362 "agent produced empty output",
2363 error_code=ERR_EMPTY_OUTPUT,
2364 exit_code=res.returncode,
2365 )
2367 no_review = no_review_reason(out)
2368 if no_review is not None:
2369 return AgentResult(
2370 self.spec.name,
2371 self.spec.vendor,
2372 False,
2373 "",
2374 duration,
2375 f"no review returned: {no_review}: {redaction.redact(out)[0][:200]}",
2376 error_code=ERR_NO_REVIEW,
2377 exit_code=res.returncode,
2378 )
2380 return AgentResult(
2381 self.spec.name, self.spec.vendor, True, out, duration, exit_code=res.returncode
2382 )
2385#: The adapter registry: protocol name -> the class that builds its argv. Keyed
2386#: by vendor name because a vendor's shipped adapter IS its default protocol —
2387#: which is also why the vendor vocabulary doubles as the ``adapter`` vocabulary
2388#: (``config.recognised_adapters``). A seat naming ``adapter`` picks a row here
2389#: directly instead of inheriting its vendor's (issue #705).
2390_VENDOR_ADAPTERS: dict[str, type[Adapter]] = {
2391 "anthropic": ClaudeAdapter,
2392 "openai": CodexAdapter,
2393 "google": AgyAdapter,
2394 "local": LocalAdapter,
2395 "anthropic-api": AnthropicApiAdapter,
2396 "openai-api": OpenAiApiAdapter,
2397 "google-api": GoogleApiAdapter,
2398 "xai-api": XaiApiAdapter,
2399 "openai-compatible": GenericOpenAICompatibleAdapter,
2400 # `xai` is a bring-your-own-CLI seat (Grok through Cursor's `cursor-agent`,
2401 # issue #701): the operator supplies `command`/`extra_args`, exactly as for
2402 # `cli`, but the seat keeps its own vendor identity at the panel gate.
2403 "xai": GenericCLIAdapter,
2404 "cli": GenericCLIAdapter,
2405}
2408def register_adapter(vendor: str, adapter_cls: type[Adapter]) -> None:
2409 """Register a custom adapter class for a vendor string.
2411 The name registered is usable as BOTH a ``vendor`` and an ``adapter``
2412 (issue #705): the registry is the adapter vocabulary, so a custom adapter is
2413 selectable by a seat that keeps some other vendor identity.
2415 Registering also teaches ``config`` the name (issue #701): a vendor with an
2416 adapter behind it is a vendor this run genuinely knows, so it must not warn
2417 as unknown and must not be folded into the generic ``cli`` identity at the
2418 cross-vendor gate. The two registries are updated together so they cannot
2419 disagree about what "known" means.
2420 """
2421 name = config_module.normalise_vendor(vendor)
2422 if not name:
2423 # One guard, ahead of both tables. `register_vendor` ignores a name that is
2424 # empty after normalisation, and this used to write the adapter under the
2425 # key "" regardless — the two registries disagreeing for exactly the input
2426 # they were joined to agree on.
2427 raise ValueError("register_adapter: vendor name is empty after normalisation")
2428 config_module.register_vendor(name)
2429 # The class's own answer (#903), so a custom HTTP adapter written as a direct
2430 # `Adapter` subclass — the documented example — can say it runs no process.
2431 # Only an explicit `False` opts out: a class that says nothing, or says
2432 # `None`, spawns, which keeps it under the audit.
2433 config_module.register_adapter_transport(
2434 name, spawns=getattr(adapter_cls, "SPAWNS_PROCESS", True) is not False
2435 )
2436 _VENDOR_ADAPTERS[name] = adapter_cls
2439def _registry_state() -> tuple[dict[str, type[Adapter]], set[str], dict[str, bool]]:
2440 """A copy of every table :func:`register_adapter` writes (#904).
2442 Three tables in two modules: the adapter classes here, and ``config``'s
2443 registered vendor names and transports. Tests that register an adapter must
2444 put all three back, and restoring them one dict at a time left the ones each
2445 test forgot — a leaked name failed tests in another module, depending on the
2446 order the modules ran in. The list of tables lives here, beside the function
2447 that writes them, so a fourth table is added to the snapshot in the same place.
2448 """
2449 return (
2450 dict(_VENDOR_ADAPTERS),
2451 set(config_module._REGISTERED_VENDORS),
2452 dict(config_module._REGISTERED_ADAPTER_SPAWNS),
2453 )
2456def _restore_registry_state(state: tuple[dict, set, dict]) -> None:
2457 """Put back a :func:`_registry_state` snapshot, in place.
2459 In place, not by rebinding: other modules hold the tables by reference
2460 (``from ai_jury.adapters import _VENDOR_ADAPTERS``), and a rebound name would
2461 leave them reading the old object.
2462 """
2463 adapter_classes, vendors, spawns = state
2464 for table, saved in (
2465 (_VENDOR_ADAPTERS, adapter_classes),
2466 (config_module._REGISTERED_VENDORS, vendors),
2467 (config_module._REGISTERED_ADAPTER_SPAWNS, spawns),
2468 ):
2469 table.clear()
2470 table.update(saved)
2473def make_adapter(spec: AgentSpec, mock: bool = False) -> Adapter:
2474 """The adapter that builds *spec*'s command line.
2476 Selected by the seat's ADAPTER key, which is its ``vendor`` unless the
2477 operator named one (issue #705). Keying this on the vendor made the two
2478 inseparable: a GPT model reached through Cursor's ``cursor-agent`` got
2479 ``codex exec`` argv it cannot parse, and the only way to a passthrough argv
2480 was ``vendor = "cli"``, which cost the seat its identity at the panel gate.
2481 Nothing else about the seat moves — the ballot, the report and
2482 ``min_vendors`` all still read ``spec.vendor``.
2484 Raises ``ConfigError`` when the seat NAMES an adapter this build does not
2485 have (issue #708). The generic fall-through below still catches an unknown
2486 *vendor* — that seat named no protocol, so inheriting the generic one is the
2487 documented behaviour — but it must never catch a named one: falling through
2488 there is the silent guess #705 exists to remove, and it made every reader
2489 that builds a seat without validating the config — a caller of
2490 ``load_config`` with its default ``validate=False``, or of this function
2491 directly — disagree with the run about the very same file.
2492 """
2493 # ``or None`` mirrors ``AgentSpec.__post_init__``: an adapter that
2494 # normalises to nothing is no adapter at all, and falls back to the vendor.
2495 # A raw dict-built or hand-made spec must read the same as a loaded one.
2496 adapter_error = config_module.unknown_adapter_error(
2497 getattr(spec, "adapter", None) or None, getattr(spec, "name", "") or ""
2498 )
2499 if adapter_error is not None:
2500 raise config_module.ConfigError(adapter_error)
2501 if mock:
2502 return MockAdapter(spec)
2503 return adapter_class(spec)(spec)
2506def adapter_class(spec) -> type[Adapter]:
2507 """The adapter class :func:`make_adapter` builds for *spec* (pure, duck-typed).
2509 The ADAPTER key decides first: a registered name always gets its class,
2510 whatever else the seat carries — so an ``endpoint`` on a ``cli``, ``claude``,
2511 ``codex`` or ``agy`` seat changes nothing about how it runs. Only a seat with
2512 no registered adapter falls through to the ``endpoint``/``command`` guess.
2513 """
2514 cls = _VENDOR_ADAPTERS.get(config_module.spec_adapter(spec))
2515 if cls is not None:
2516 return cls
2517 command = getattr(spec, "command", "") or ""
2518 if getattr(spec, "endpoint", None) or (getattr(spec, "api_key_env", None) and not command):
2519 return GenericOpenAICompatibleAdapter
2520 if command:
2521 return GenericCLIAdapter
2522 return AgyAdapter