Coverage for src/ai_jury/config.py: 99%
528 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Configuration loading for the jury.
3Config is TOML (see ``jury.toml``). The loader is tolerant: a missing config
4file falls back to a sensible built-in default so the tool runs out of the box.
5"""
7from __future__ import annotations
9import os
10import tomllib
11from dataclasses import dataclass, field
12from dataclasses import fields as dataclass_fields
13from pathlib import Path
14from urllib.parse import urlsplit
16from .ci import fail_on_error
17from .redaction import ENV_VAR_NAME_RULE, redact, safe_env_var_name
19# Hosts that are safe to reach over plaintext http and never an SSRF target.
20_LOOPBACK_HOSTS = ("localhost", "127.0.0.1", "::1", "[::1]")
22# Upper bound on a config/policy TOML file (issue #316/L-5). A real config is a
23# few KB; refuse a multi-MB / pathological file so `tomllib` can't be driven to
24# exhaust memory (the file may be attacker-supplied when jury runs from a PR
25# checkout). Mirrors the cache's _MAX_CACHE_BYTES.
26_MAX_CONFIG_BYTES = 4 * 1024 * 1024
29def _read_toml_bounded(path: Path) -> dict:
30 """Parse a TOML file with a size cap (issue #316/L-5)."""
31 with path.open("rb") as fh:
32 raw = fh.read(_MAX_CONFIG_BYTES + 1)
33 if len(raw) > _MAX_CONFIG_BYTES:
34 raise ConfigError(f"config file '{path}' exceeds the {_MAX_CONFIG_BYTES}-byte limit.")
35 try:
36 text = raw.decode("utf-8")
37 except UnicodeDecodeError:
38 # TOML is UTF-8 by spec; surface a clean error instead of a raw
39 # UnicodeDecodeError (review of #316 — the prior tomllib.load crashed the
40 # same way on bad bytes; now it's a ConfigError).
41 raise ConfigError(f"config file '{path}' is not valid UTF-8.") from None
42 try:
43 return tomllib.loads(text)
44 except tomllib.TOMLDecodeError as exc:
45 raise ConfigError(f"invalid TOML in config file '{path}': {redact(str(exc))[0]}") from None
48def _is_relative_path_command(command: str) -> bool:
49 """True for a relative command that contains a path separator (#293/F-6).
51 A bare name (``codex``) is fine — it is resolved on PATH. An absolute path
52 (``/usr/bin/codex``) is fine — it is explicit. A relative path with a
53 separator (``./tools/codex``, ``bin/agy``) is rejected because it resolves a
54 binary from an attacker-influenceable working-directory-relative location.
55 """
56 cmd_str = str(command or "")
57 has_sep = "/" in cmd_str or "\\" in cmd_str or (os.altsep is not None and os.altsep in cmd_str)
58 return has_sep and not Path(cmd_str).is_absolute()
61# Env opt-in for a non-loopback local endpoint. It lives in the environment, NOT
62# in jury.toml, on purpose (review of #291): the threat model is an
63# attacker-controlled config, so the opt-in must sit OUTSIDE the surface the
64# attacker controls. Without it, a non-loopback host (incl. cloud-metadata
65# 169.254.169.254) is a hard error so an attacker config cannot drive an
66# SSRF POST to an internal address — matching the default-secure F-1 posture.
67_ALLOW_REMOTE_ENDPOINT_ENV = "JURY_ALLOW_REMOTE_ENDPOINT"
69# Opt-in strict mode (issue #296): when set, every agent ``command`` must be an
70# absolute path — rejecting even a bare name, whose PATH resolution an attacker
71# who controls the CI runner's PATH could hijack with a shim. Off by default so
72# the convenient bare-name (``claude``) keeps working for local use.
73_REQUIRE_ABSOLUTE_COMMAND_ENV = "JURY_REQUIRE_ABSOLUTE_COMMAND"
76def _endpoint_issues(endpoint: str, label: str) -> tuple[list[str], list[str]]:
77 """Validate a local-agent ``endpoint`` URL (issue #291, SSRF defense).
79 Returns ``(errors, warnings)``. A non-``http``/``https`` scheme is a hard
80 error (blocks ``file://``/``ftp://`` and other SSRF primitives). A non-loopback
81 host is also a hard error UNLESS the operator opts in via the
82 ``JURY_ALLOW_REMOTE_ENDPOINT`` environment variable (a remote model server is
83 a legitimate but riskier choice the attacker-controlled config must not be
84 able to select on its own); when opted in it degrades to a warning, plus a
85 cleartext warning for plaintext ``http``.
86 """
87 errors: list[str] = []
88 warnings: list[str] = []
89 if not isinstance(endpoint, str): 89 ↛ 90line 89 didn't jump to line 90 because the condition on line 89 was never true
90 errors.append(f"agent '{label}' endpoint must be a string (got {endpoint!r}).")
91 return errors, warnings
92 # `urlsplit` raises ValueError on a malformed URL (e.g. `http://[::1`,
93 # "Invalid IPv6 URL"). Convert that to a hard config error (issue #315) so
94 # `validate_config` reports it cleanly instead of crashing with a stack trace
95 # — the malformed string is, by definition, not a usable endpoint.
96 try:
97 parsed = urlsplit(endpoint)
98 parsed.hostname # noqa: B018 - also raises ValueError on a bad IPv6 host
99 except (ValueError, TypeError, AttributeError):
100 errors.append(f"agent '{label}' endpoint '{endpoint}' is not a valid URL.")
101 return errors, warnings
102 scheme = (parsed.scheme or "").lower()
103 if scheme not in ("http", "https"):
104 errors.append(
105 f"agent '{label}' endpoint scheme '{parsed.scheme or '(none)'}' is "
106 f"not allowed; use http or https."
107 )
108 return errors, warnings
109 host = (parsed.hostname or "").lower()
110 if host in _LOOPBACK_HOSTS:
111 return errors, warnings
112 if not os.environ.get(_ALLOW_REMOTE_ENDPOINT_ENV):
113 errors.append(
114 f"agent '{label}' endpoint host '{host or '(none)'}' is not loopback; "
115 f"a non-loopback model server (incl. internal/metadata addresses) is "
116 f"refused by default. Set {_ALLOW_REMOTE_ENDPOINT_ENV}=1 in the "
117 f"environment to allow a trusted remote endpoint."
118 )
119 return errors, warnings
120 warnings.append(
121 f"agent '{label}' endpoint host '{host or '(none)'}' is not loopback; "
122 f"the (redacted) diff is sent to a remote server — ensure it is trusted "
123 f"and not an internal/metadata address."
124 )
125 if scheme == "http":
126 warnings.append(
127 f"agent '{label}' endpoint uses plaintext http to a non-loopback "
128 f"host; prefer https so the prompt is not sent in cleartext."
129 )
130 return errors, warnings
133#: Distinct vendors a run must have heard from before it may call itself a
134#: cross-vendor consensus (issue #682). The shipped default of 2 FAILS CLOSED:
135#: before it, a three-vendor panel that collapsed to one exited 0 with a verdict
136#: and nothing said so (#635). Lower it, or set 0, to opt out — see
137#: ``[jury.ci] min_vendors`` and ``--no-min-vendors``.
138DEFAULT_MIN_VENDORS = 2
141def _non_negative_int(value, default: int) -> int:
142 """A non-negative int from raw config data, falling back to ``default``.
144 ``validate_config`` refuses a malformed or negative value (docs audit
145 2026-09-29), so a run started from ``jury`` never reaches the fallback. It is
146 for a Python caller that materialises without validating: there, a
147 malformed value falls back to the default rather than raising, because the
148 default is the *safe* direction for a fail-closed gate. A negative value
149 falls back too — clamping it to ``0`` used to turn ``min_vendors = -1``
150 into "guard off", the one direction a fail-closed gate must never drift.
151 """
152 try:
153 number = int(value)
154 except (TypeError, ValueError):
155 return default
156 return number if number >= 0 else default
159DEFAULT_CONFIG: dict = {
160 "jury": {
161 "rounds": 2,
162 "chair": "claude",
163 "timeout": 600,
164 "parallel": True,
165 "verify": True,
166 "ci": {
167 "fail_on": ["critical", "major"],
168 "ignore_unverified": True,
169 "min_vendors": DEFAULT_MIN_VENDORS,
170 },
171 "context": {"mode": "diff-only", "redact_secrets": True},
172 },
173 # Execution controls (issue #30) are optional and conservative by default:
174 # no overall/per-phase budget and zero retries, so out-of-the-box behaviour
175 # is unchanged. They live under [jury] and are documented in
176 # docs/configuration.md.
177 "agent": [
178 {
179 "name": "claude",
180 "vendor": "anthropic",
181 "command": "claude",
182 # The reviewer only needs its prompt, which already carries the diff,
183 # so it gets no tools at all: `--tools ""` leaves no built-in tool
184 # available, the deny list names every write, shell, read, network and
185 # subagent tool as a second layer, `--strict-mcp-config` with no
186 # `--mcp-config` keeps the user's own MCP servers out, `--safe-mode`
187 # drops CLAUDE.md, hooks, skills and plugins (a PR checkout's project
188 # settings included), `--no-session-persistence` writes no transcript
189 # of the untrusted diff to ~/.claude/projects, and `dontAsk` denies
190 # a tool call instead of prompting (so `-p` cannot hang) or approving
191 # it. `privilege.enforce_read_only` injects the same lockdown into a
192 # seat configured without it; `privilege._CLAUDE_DENIED_TOOLS` is the
193 # list below, and a test keeps the two equal.
194 "extra_args": [
195 "--output-format",
196 "text",
197 "--tools",
198 "",
199 "--disallowed-tools",
200 "Edit,Write,NotebookEdit,Bash,Read,Grep,Glob,WebFetch,WebSearch,Task,Agent",
201 "--strict-mcp-config",
202 "--safe-mode",
203 "--no-session-persistence",
204 "--permission-mode",
205 "dontAsk",
206 ],
207 },
208 {
209 "name": "codex",
210 "vendor": "openai",
211 "command": "codex",
212 # `codex exec` reads the prompt from stdin (see CodexAdapter) and only
213 # needs to READ it and print a review — the diff is fetched by the
214 # jury process (`gh`), not the agent. So the secure default is a
215 # read-only sandbox (issue #100); widen it (e.g. `-s workspace-write`
216 # or `danger-full-access`) only if your workflow truly needs it.
217 "extra_args": ["-s", "read-only"],
218 },
219 # No `agy` here: see AGY_AGENT below. A run with no jury.toml never
220 # spawns it; claude + codex are still two vendors, so the shipped
221 # DEFAULT_MIN_VENDORS guard is met by the default panel alone.
222 ],
223}
226#: The Antigravity seat, for a config that asks for it by name. It is NOT in
227#: :data:`DEFAULT_CONFIG`'s panel, because agy cannot be confined for a reviewer
228#: of untrusted diffs: `--dangerously-skip-permissions` auto-approves its tools
229#: (without it, headless agy denies the first command a review tries and returns
230#: no review), and `--sandbox` (issue #100) restricts its terminal but, measured
231#: on agy 1.2.9, did not stop it reading or writing files outside its working
232#: directory or reaching the network. agy offers no flag that removes its tools.
233#: `jury init --agents agy` and `jury run-agent --agent agy` still use this entry;
234#: a panel that seats it draws a least-privilege warning (`--strict` fails).
235AGY_AGENT: dict = {
236 "name": "agy",
237 "vendor": "google",
238 "command": "agy",
239 "extra_args": ["--dangerously-skip-permissions", "--sandbox"],
240}
242#: Why agy is opt-in, in the words every "no reviewer" message and `jury init`
243#: use, so they cannot drift apart.
244AGY_OPT_IN_NOTE = (
245 "agy is not in the default panel because it cannot be confined for untrusted "
246 "diffs (it reads, writes and reaches the network even with --sandbox); add it "
247 "explicitly in jury.toml if you accept that (`jury init --agents agy` writes "
248 "the seat)"
249)
252def agy_opt_in_hint(adapter_keys, which) -> str | None:
253 """:data:`AGY_OPT_IN_NOTE` when agy is installed but no seat uses it, else None.
255 Pure but for *which* (``shutil.which`` in production), so a caller that has
256 just found no usable reviewer can tell an agy-only machine why its one CLI
257 was not used, instead of telling it to install a CLI it already has.
258 *adapter_keys* are the adapter keys of the configured seats; ``google`` is
259 agy's.
260 """
261 if "google" in set(adapter_keys):
262 return None
263 if not which(AGY_AGENT["command"]):
264 return None
265 return AGY_OPT_IN_NOTE
268# Vendors that talk HTTP directly (no CLI subprocess), so they need no
269# `command`: `local` (a user-supplied OpenAI-compatible server, issue #43) and
270# the hosted-API adapters (a real vendor API keyed by an env-var API key,
271# issue #430/#432, joined by `xai-api` in #701).
272_NO_COMMAND_VENDORS = (
273 "local",
274 "anthropic-api",
275 "openai-api",
276 "google-api",
277 "xai-api",
278 "openai-compatible",
279)
281KNOWN_VENDORS = (
282 "anthropic",
283 "openai",
284 "google",
285 "xai",
286 "local",
287 "anthropic-api",
288 "openai-api",
289 "google-api",
290 "xai-api",
291 "openai-compatible",
292 "cli",
293)
295#: The vendor identity every unrecognised vendor collapses into (issue #701).
296#: ``cli`` is a real, documented vendor — "some CLI I brought myself" — and the
297#: generic fallback lands a seat in exactly that bucket, so that is the identity
298#: it carries for the cross-vendor gate.
299GENERIC_VENDOR = "cli"
301#: Vendors serviced by the generic bring-your-own-CLI adapter. The operator
302#: supplies ``command``/``extra_args``; the tool knows no vendor-specific
303#: sandbox flag to add or remove for them (issue #701).
304GENERIC_CLI_VENDORS = ("cli", "xai")
306#: The built-in adapters that spawn ``command`` rather than make an HTTP call:
307#: every known vendor that is not a commandless one (claude, codex, agy and the
308#: bring-your-own CLI). None of them reads ``endpoint`` — ``make_adapter`` picks
309#: them by adapter key before it looks at ``endpoint`` — so an ``endpoint`` on
310#: one of these seats is refused by :func:`validate_config` (#901 review).
311CLI_ADAPTERS: tuple[str, ...] = tuple(v for v in KNOWN_VENDORS if v not in _NO_COMMAND_VENDORS)
313#: Vendors registered at runtime through ``adapters.register_adapter`` — the
314#: documented extension point for a custom adapter. Registering an adapter is
315#: what makes a vendor name *known*: without it the name is a typo as far as
316#: this tool can tell, and :func:`vendor_identity` folds it into
317#: ``GENERIC_VENDOR``. Module-level mutable state, deliberately: it mirrors
318#: ``adapters._VENDOR_ADAPTERS``, which is mutable for the same reason.
319_REGISTERED_VENDORS: set[str] = set()
322def register_vendor(vendor: str) -> None:
323 """Record *vendor* as a recognised name (called by ``register_adapter``)."""
324 name = normalise_vendor(vendor)
325 if name:
326 _REGISTERED_VENDORS.add(name)
329#: Adapter keys registered at runtime, and whether the registered class spawns a
330#: process (#901 review). ``adapters.register_adapter`` records it here, beside
331#: the class it stores, so :func:`spawns_process` can answer for a custom adapter
332#: without importing ``adapters``. A registration replaces the built-in answer
333#: for the same key, exactly as it replaces the built-in class.
334_REGISTERED_ADAPTER_SPAWNS: dict[str, bool] = {}
337def register_adapter_transport(adapter: str, *, spawns: bool) -> None:
338 """Record whether the adapter registered under *adapter* spawns a process."""
339 name = normalise_vendor(adapter)
340 if name:
341 _REGISTERED_ADAPTER_SPAWNS[name] = spawns
344def spawns_process(spec) -> bool:
345 """Whether the adapter ``make_adapter`` builds for *spec* spawns a CLI (pure).
347 The least-privilege audit must ask this of the same selection the spawner
348 makes (#901 review): it used to skip any seat with an ``endpoint``, while
349 ``make_adapter`` picks the adapter by key first and ignores ``endpoint`` on
350 every CLI adapter — so a ``cli`` aider seat with a stray ``endpoint`` ran
351 aider and was audited as an HTTP seat, clean.
353 It lives here, not in ``adapters``, so ``privilege`` can ask it without an
354 import cycle (``adapters`` imports ``privilege`` for ``enforce_read_only``;
355 CodeQL ``py/cyclic-import``). It follows ``adapters.adapter_class`` step for
356 step, and ``tests/test_privilege.py`` holds the two to the same answer for
357 every built-in key, a registered HTTP and a registered CLI adapter, and each
358 fallback shape:
360 1. a registered key: what the registered class declares — its
361 ``SPAWNS_PROCESS`` attribute, ``True`` unless it says otherwise (#903);
362 2. a built-in key: HTTP for the commandless ones (``local``, the hosted APIs,
363 ``openai-compatible``), a process for :data:`CLI_ADAPTERS`;
364 3. no adapter for the key: HTTP when the seat names an ``endpoint``, or an
365 ``api_key_env`` with no ``command`` (the OpenAI-compatible adapter);
366 otherwise a process (the generic CLI adapter, or agy's with no command).
367 """
368 key = spec_adapter(spec)
369 if key in _REGISTERED_ADAPTER_SPAWNS:
370 return _REGISTERED_ADAPTER_SPAWNS[key]
371 if key in _NO_COMMAND_VENDORS:
372 return False
373 if key in CLI_ADAPTERS:
374 return True
375 command = getattr(spec, "command", "") or ""
376 return not (
377 getattr(spec, "endpoint", None) or (getattr(spec, "api_key_env", None) and not command)
378 )
381def recognised_vendors() -> tuple[str, ...]:
382 """Every vendor name this run understands: shipped plus runtime-registered."""
383 return (*KNOWN_VENDORS, *sorted(_REGISTERED_VENDORS - set(KNOWN_VENDORS)))
386def normalise_vendor(vendor) -> str:
387 """The single spelling of a configured vendor every rule reads (pure).
389 ``vendor = "XAI-API"`` and ``vendor = " xai-api "`` name the same vendor as
390 ``vendor = "xai-api"``, so they must normalise to it *before* any rule looks
391 at them. This is the one place that decides what a configured vendor string
392 means: validation, the adapter lookup and :func:`vendor_identity` all go
393 through it, so a seat cannot pass validation under one spelling and reach
394 the cross-vendor gate under another (issue #701, review round 2).
396 Anything that is not a string is not a vendor name, so it normalises to
397 ``""`` rather than raising: ``vendor = 3`` is a config mistake to warn
398 about, not a crash.
399 """
400 return vendor.strip().lower() if isinstance(vendor, str) else ""
403def is_recognised_vendor(vendor) -> bool:
404 """Whether *vendor* names a vendor the tool actually knows (pure-ish)."""
405 key = normalise_vendor(vendor)
406 return bool(key) and key in set(recognised_vendors())
409def is_commandless_vendor(vendor) -> bool:
410 """Whether *vendor* talks HTTP directly and so needs no ``command`` (pure).
412 The one reader of :data:`_NO_COMMAND_VENDORS`. Validation asked this
413 question of the raw string while doctor asked it of a copy of the same
414 tuple, which is how ``vendor = "XAI-API"`` could be a recognised vendor and
415 still be told it was missing a ``command``.
416 """
417 key = normalise_vendor(vendor)
418 return bool(key) and (key in _NO_COMMAND_VENDORS or key.endswith("-api"))
421def vendor_identity(vendor: str) -> str:
422 """The identity a seat carries for the cross-vendor gate (issue #701).
424 A recognised vendor keeps its own name. Everything else — a typo, a vendor
425 this build predates, anything routed to the generic fallback — answers to
426 ``GENERIC_VENDOR``, because that is what it is: two seats the tool could not
427 identify are not two perspectives, and counting them as two is how a bench
428 satisfies ``min_vendors`` without being diverse (#682, reached through
429 configuration). The raw string is still what the ballots and the report
430 carry; only the *gate* collapses, so provenance is never rewritten.
432 Returns ``""`` for an empty vendor, which counts as no vendor at all.
433 """
434 name = normalise_vendor(vendor)
435 if not name:
436 return ""
437 return name if name in set(recognised_vendors()) else GENERIC_VENDOR
440def recognised_adapters() -> tuple[str, ...]:
441 """Every name a seat may give ``[[agent]] adapter`` (issue #705).
443 The adapter vocabulary IS the vendor vocabulary, because ``adapters``' own
444 registry is keyed by vendor name: ``anthropic`` selects the claude protocol,
445 ``cli`` the bring-your-own-CLI passthrough, ``openai-api`` the hosted API
446 call. Derived rather than re-typed so the two cannot drift, and so
447 ``register_adapter`` — which teaches this build a name — makes that name
448 usable as an ``adapter`` in the same breath.
449 """
450 return recognised_vendors()
453def is_recognised_adapter(adapter) -> bool:
454 """Whether *adapter* names an adapter this build actually has (pure-ish)."""
455 key = normalise_vendor(adapter)
456 return bool(key) and key in set(recognised_adapters())
459def unknown_adapter_error(adapter, label: str = "") -> str | None:
460 """The ONE diagnosis for an ``adapter`` name this build does not have (pure).
462 Returns the message, or ``None`` when the seat names no adapter (it inherits
463 its vendor's) or names one that exists.
465 Single because the name is read in two places that must not disagree
466 (issue #708). :func:`validate_config` reads it before a run and turns this
467 into a hard error; ``adapters.make_adapter`` reads it again at the moment the
468 command line is built, and refuses instead of falling through to the generic
469 adapter. Both are needed. With only the first, every caller that loads a
470 config *without* validation — ``--doctor``, ``jury run-agent`` — silently got
471 a ``GenericCLIAdapter``, so ``--doctor`` printed three ``[available]`` seats
472 and ``ready to run: yes`` for a file a real run refused outright. With only
473 the second, the same typo would surface mid-run rather than before the panel
474 starts.
476 The fall-through was the silent guess #705 exists to remove: a name this
477 build does not have is a typo, or a plugin that was not loaded, and invoking
478 the seat through *some other* protocol answers neither.
479 """
480 if adapter is None or is_recognised_adapter(adapter):
481 return None
482 who = f"agent '{label}'" if label else "agent"
483 return (
484 f"{who} has unknown adapter {adapter!r} (expected one "
485 f"of {', '.join(recognised_adapters())}). 'adapter' names the protocol "
486 f"used to build the command line; 'vendor' names the identity the "
487 f"cross-vendor gate counts."
488 )
491def adapter_key(vendor, adapter=None) -> str:
492 """The protocol a seat is invoked through, from its two config keys (pure).
494 ``vendor`` answers "what is this seat?" — the identity the cross-vendor gate
495 counts (:func:`vendor_identity`). ``adapter`` answers "how is its command
496 line built?". Before issue #705 one key answered both, so a GPT model
497 reached through Cursor's ``cursor-agent`` had to choose between running
498 (``vendor = "cli"``, and three such seats are then one vendor) and being
499 counted (``vendor = "openai"``, and ``cursor-agent exec`` is not a command).
501 An unset ``adapter`` falls back to the vendor, which is what makes every
502 configuration written before this key existed byte-identical in argv.
503 """
504 return normalise_vendor(adapter) or normalise_vendor(vendor)
507def spec_adapter(spec) -> str:
508 """:func:`adapter_key` for any spec-like object (duck-typed, pure).
510 Readers outside this module — ``privilege``, ``doctor``, ``adapters`` — are
511 handed specs by tests as well as by the loader, so they ask here instead of
512 reaching for two attributes each and disagreeing about the fallback.
513 """
514 return adapter_key(getattr(spec, "vendor", ""), getattr(spec, "adapter", None))
517KNOWN_TOP_LEVEL_KEYS = ("jury", "agent")
518KNOWN_JURY_KEYS = (
519 "rounds",
520 "chair",
521 "timeout",
522 "parallel",
523 "verify",
524 # Nested tables. The keys inside them are checked too, against
525 # `KNOWN_NESTED_JURY_KEYS` — defined below, beside the dataclasses it
526 # derives them from (issue #719).
527 "ci",
528 "context",
529 "seed",
530 "anonymize_debate",
531 "prefer_non_reviewer_chair",
532 # Demote a local-only finding to non-blocking severity (issue #442).
533 "demote_local_only",
534 # Execution controls (issue #30).
535 "total_timeout",
536 "phase_timeout",
537 "retries",
538 # Adaptive rounds (issue #40).
539 "max_rounds",
540 "early_stop",
541 # Risk-aware auto-depth (issue #120).
542 "auto_depth",
543 # Full-transcript / verbose rendering (rendering-only; not in config_hash).
544 "transcript",
545 # Final-verdict mode: "chair" synthesis or panel "vote" (rendering-only).
546 "decision",
547 # Animated theater view defaults (rendering-only; issue #364).
548 "theater",
549 "theater_style",
550 # Large-diff handling (issue #31); a nested table, like `ci`/`context`.
551 "diff",
552 # The report footer (issue #911); a nested table too. Rendering-only:
553 # not in `config_hash`.
554 "output",
555 # Risk-aware tiered model routing (issue #524) and the static-analysis
556 # pre-pass (issue #523). Both are read by `_from_dict` and documented in
557 # docs/configuration.md, so `--strict-config` must not reject them (#715).
558 "routing",
559 "hints",
560)
561KNOWN_AGENT_KEYS = (
562 "name",
563 "vendor",
564 # The protocol used to build this seat's command line (issue #705).
565 # Optional: it defaults to the vendor's shipped adapter.
566 "adapter",
567 "command",
568 "model",
569 "timeout",
570 "enabled",
571 "extra_args",
572 # OpenAI-compatible local/open-weight endpoint (issue #43).
573 "endpoint",
574 # Universal agent extensions
575 "api_key_env",
576 "prompt_mode",
577 "headers",
578 # Reasoning effort (issue #662): mapped per vendor in adapters.effort_args.
579 "effort",
580 # Cost tier (issue #714): what tiered routing reads to decide which seats
581 # sit on a routine diff and which one anchors it.
582 "tier",
583 # Sampling temperature for a local seat's chat-completions request. Some
584 # open-weight models (gpt-oss) loop under the greedy default of 0.
585 "temperature",
586)
588#: Inclusive bounds for ``[[agent]] temperature`` — the OpenAI chat-completions
589#: range, which every OpenAI-compatible local server accepts.
590TEMPERATURE_MIN = 0.0
591TEMPERATURE_MAX = 2.0
594def is_valid_temperature(value) -> bool:
595 """Whether *value* is a usable ``[[agent]] temperature`` (pure).
597 A TOML integer or float in ``[TEMPERATURE_MIN, TEMPERATURE_MAX]``. ``bool``
598 is refused although it subclasses ``int`` (``temperature = true`` is a typo,
599 not 1.0), and ``nan`` fails the range test like any other out-of-range value.
600 """
601 if isinstance(value, bool) or not isinstance(value, (int, float)):
602 return False
603 return TEMPERATURE_MIN <= value <= TEMPERATURE_MAX
606#: Accepted values for ``[[agent]] effort`` / ``--effort`` (issue #662).
607#: Duplicated as a literal rather than imported from ``adapters`` so config
608#: validation stays free of any adapter import (``adapters`` imports ``config``).
609#: ``tests/test_adapters.py`` pins the two lists together.
610KNOWN_EFFORTS = ("low", "medium", "high")
612#: Accepted values for ``[[agent]] tier`` (issue #714). ``frontier`` is the
613#: default and the anchor-capable kind; ``economical`` is a seat tiered routing
614#: may put on a routine diff in place of the frontier seats it benches. The
615#: operator says which is which — no model-name heuristics.
616KNOWN_TIERS = ("frontier", "economical")
617DEFAULT_TIER = "frontier"
619#: Accepted values for ``[jury] routing`` (issue #524). ``standard`` is the
620#: default uniform panel; ``tiered`` is the risk-aware panel that reads each
621#: seat's :data:`KNOWN_TIERS` value. The closed vocabulary is the companion of
622#: ``tier``'s (#747): the two keys are one feature, and a misspelling of either
623#: silently produces the panel the operator did not ask for.
624KNOWN_ROUTINGS = ("standard", "tiered")
625DEFAULT_ROUTING = "standard"
628class ConfigError(Exception):
629 """Raised when a jury configuration is invalid."""
632# The `[[agent]] headers` messages live here, not inline, because TWO paths
633# reject the same shapes and must say the same thing (issue #716, review r1):
634# `validate_config` collects them into its error list, and `_from_dict` raises
635# one directly for a config that reached materialisation unvalidated —
636# `load_config` defaults to `validate=False`, and a Python caller that loads a
637# config without asking for validation reaches `_from_dict` directly.
638#
639# Every message names the AGENT and, where it helps, the header NAME, and none
640# of them quotes the offending VALUE back: a header is exactly where an
641# `Authorization: Bearer …` credential lives, and these are printed to terminals
642# and pasted into issues. The type says what is wrong without reproducing what
643# is secret — the precedent `api_key_env` set for a rejected value.
644def _headers_not_a_table_message(label: str, value) -> str:
645 """`headers` is not a table at all — it cannot become headers (hard)."""
646 return f"agent '{label}' headers must be a table of string keys (got {type(value).__name__})."
649def _headers_bad_key_message(label: str, key) -> str:
650 """A key that is not a string cannot be a header name (hard).
652 tomllib cannot produce one — every TOML key, bare or quoted, parses to
653 `str` — so this guards a config dict built in Python (a test, an embedder
654 calling `validate_config`/`_from_dict` directly) rather than a written file.
655 """
656 return (
657 f"agent '{label}' headers has a non-string header name "
658 f"({type(key).__name__}); header names must be strings."
659 )
662def _headers_coerced_value_warning(label: str, header: str, value) -> str:
663 """A non-string value still becomes a header — so this is soft.
665 `X-Retries = 3` is a working fallback: it is sent as `X-Retries: 3`. Under
666 this module's split (unusable ⇒ hard error, working fallback ⇒ warning, as
667 for a malformed `api_key_env` name) that is a warning, and `--strict-config`
668 is where an operator asks for it to be fatal.
669 """
670 return (
671 f"agent '{label}' headers value for '{header}' is not a string "
672 f"({type(value).__name__}); it was coerced to a string before being sent."
673 )
676# The nested `[jury.*]` shape message lives here for the same reason the
677# `headers` ones do (issue #729): TWO paths reject the same shape and must say
678# the same thing. `validate_config` collects it into its error list, and the
679# `_*_from_dict` readers raise it directly for a config that reached
680# materialisation unvalidated — `load_config` defaults to `validate=False`, and
681# `jury run-agent` keeps it that way on purpose.
682#
683# `[jury.ci]` and `[jury.diff]` were checked with this wording written out
684# inline; `[jury.context]` was not checked at all, so `context = "diff-only"`
685# (written where `[jury.context] mode = "diff-only"` was meant) passed
686# `--config-validate --strict-config` and then died in `_context_from_dict` with
687# `'str' object has no attribute 'get'`.
688#: Top-level shape messages, shared by ``validate_config`` and ``_from_dict`` so the
689#: validating and the non-validating path (``jury run-agent``, ``load_config``'s
690#: default) say the same thing (issue #732).
691_JURY_NOT_A_TABLE = "[jury] must be a table."
692_AGENTS_NOT_AN_ARRAY = "[[agent]] must be an array of tables."
695def _agent_not_a_table_message(idx: int) -> str:
696 return f"agent[{idx}] must be a table."
699def _nested_table_message(table: str) -> str:
700 """`[jury.<table>]` is not a table, so its reader cannot read it (hard)."""
701 return f"[jury.{table}] must be a table."
704#: The numeric bounds on the `[jury]` scalars, keyed by the setting's dotted
705#: path under `[jury]`. Each entry is `(minimum, phrase, optional)`: *phrase* is
706#: the rule as the message has always stated it — `timeout` says "a positive
707#: integer" and `rounds` "an integer >= 1", which mean the same thing and are
708#: both kept because they are what operators have read for releases — and
709#: *optional* marks a setting whose absence is legal, which the message says as
710#: "when set".
711#:
712#: The table exists because these bounds have TWO surfaces (issue #748). Every
713#: one of them is also a CLI flag, and the flags were assigned onto the config
714#: *after* `validate_config` had already run, so nothing range-checked them:
715#: `--rounds 0` was accepted, ran a full Round 1 and exited 0 while `rounds = 0`
716#: in `jury.toml` was a hard error — one value, one field, two answers, on a
717#: flag `docs/parameters.md` documents as `≥ 1`. `bound_error` is the single
718#: reader, so a bound cannot be stated twice and drift apart.
719_NUMERIC_BOUNDS: dict[str, tuple[int, str, bool]] = {
720 "rounds": (1, "an integer >= 1", False),
721 "timeout": (1, "a positive integer", False),
722 "retries": (0, "an integer >= 0", False),
723 # Execution controls (issue #30).
724 "total_timeout": (1, "a positive integer", True),
725 "phase_timeout": (1, "a positive integer", True),
726 # Adaptive rounds (issue #40).
727 "max_rounds": (1, "an integer >= 1", True),
728 # Large-diff handling (issue #31).
729 "diff.max_bytes": (1, "a positive integer", True),
730 "diff.chunk_max_bytes": (1, "a positive integer", True),
731 # The fail-closed CI guards (docs audit 2026-09-29). Both used to be clamped
732 # rather than checked: `min_vendors = -1` or `--min-vendors -5` became 0,
733 # which is the documented opt-out, so a typo turned the cross-vendor guard
734 # off and nothing said so.
735 "ci.min_vendors": (0, "an integer >= 0", True),
736 "ci.min_reviews": (0, "an integer >= 0", True),
737}
740#: The boolean `[jury]` settings `validate_config` type-checks, as dotted paths
741#: under `[jury]` (docs audit 2026-09-29). `theater` and `output.attribution`
742#: are checked on their own, above this table's reader, with the same message.
743_BOOL_SETTINGS: tuple[str, ...] = (
744 "parallel",
745 "verify",
746 "anonymize_debate",
747 "prefer_non_reviewer_chair",
748 "demote_local_only",
749 "early_stop",
750 "auto_depth",
751 "transcript",
752 "hints",
753 "ci.ignore_unverified",
754 "context.redact_secrets",
755 "diff.chunk",
756 "diff.exclude_generated",
757)
759#: `[jury.context] mode` values; `_context_from_dict` reads anything else as the first.
760KNOWN_CONTEXT_MODES: tuple[str, ...] = ("diff-only", "expanded")
762#: `[[agent]] prompt_mode` values, as `GenericCLIAdapter._prompt_mode` reads them.
763KNOWN_PROMPT_MODES: tuple[str, ...] = ("stdin", "arg")
766def bound_error(setting: str, value, where: str | None = None) -> str | None:
767 """The message for *value* breaking the bound on ``setting``, else ``None``.
769 *setting* is a key of :data:`_NUMERIC_BOUNDS`. *where* names the surface the
770 value was written on and defaults to the ``jury.<setting>`` config path; the
771 CLI passes the flag (``--rounds``), so an operator who has no ``jury.toml``
772 at all is not pointed at a key they never wrote. The rule, the bound and the
773 quoted value are identical either way — that is the point of one reader.
775 ``None`` means "not set" and passes for an optional setting. A bool is not
776 an integer here: ``rounds = true`` is a mistake, not ``rounds = 1``.
777 """
778 minimum, phrase, optional = _NUMERIC_BOUNDS[setting]
779 if value is None and optional:
780 return None
781 if isinstance(value, int) and not isinstance(value, bool) and value >= minimum:
782 return None
783 suffix = " when set" if optional else ""
784 return f"{where or f'jury.{setting}'} must be {phrase}{suffix} (got {value!r})."
787def validate_config(data: dict, strict: bool = False) -> list:
788 """Validate a raw config dict.
790 Raises ``ConfigError`` with an actionable message on hard-invalid input
791 (rounds < 1, timeout <= 0, duplicate agent names, empty/missing command,
792 no agents at all). Returns a list of warning strings for soft issues
793 (unknown vendor, chair not an enabled agent, unknown keys).
795 When ``strict`` is True, soft issues raise ``ConfigError`` instead of
796 being returned as warnings.
797 """
798 warnings: list = []
799 errors: list = []
801 if not isinstance(data, dict):
802 raise ConfigError("config root must be a table/dict.")
804 # Unknown top-level keys (soft).
805 for key in data:
806 if key not in KNOWN_TOP_LEVEL_KEYS:
807 warnings.append(
808 f"unknown top-level key '{key}' (expected one of "
809 f"{', '.join(KNOWN_TOP_LEVEL_KEYS)})."
810 )
812 jury = data.get("jury", {})
813 if not isinstance(jury, dict):
814 raise ConfigError(_JURY_NOT_A_TABLE)
816 for key in jury:
817 if key not in KNOWN_JURY_KEYS:
818 warnings.append(
819 f"unknown key 'jury.{key}' (expected one of {', '.join(KNOWN_JURY_KEYS)})."
820 )
822 # Unknown keys INSIDE the nested `[jury.*]` tables (soft, issue #719).
823 #
824 # The loop above stops at the table names: it sees `jury.ci` and is happy.
825 # `_ci_from_dict`/`_context_from_dict`/`_diff_from_dict` then take the keys
826 # they know and drop the rest, so `[jury.ci] min_vendor = 3` (the
827 # cross-vendor gate, one `s` short) passed `--config-validate
828 # --strict-config` clean and the panel ran on the default of 2. Same soft
829 # severity as the top level — the config still loads, it just does not mean
830 # what it says — with the dotted path in the message so the operator is
831 # told which table to look in.
832 #
833 # A non-table is reported as a shape error and NOT iterated for unknown
834 # keys: iterating a string would produce one warning per character, and the
835 # value cannot carry keys anyway. All three tables are checked in this one
836 # loop (issue #729) — `ci` and `diff` had the check written out inline below
837 # and `context` had none, which is exactly how the missing one went unnoticed.
838 for table, known_nested in KNOWN_NESTED_JURY_KEYS.items():
839 nested = jury.get(table)
840 if nested is None:
841 continue
842 if not isinstance(nested, dict):
843 errors.append(_nested_table_message(table))
844 continue
845 for key in nested:
846 if key not in known_nested:
847 warnings.append(
848 f"unknown key 'jury.{table}.{key}' (expected one of {', '.join(known_nested)})."
849 )
851 # The bounded `[jury]` scalars (hard): rounds >= 1, timeout > 0, the
852 # optional execution budgets (issue #30) > 0 and retries >= 0. Every rule
853 # and every message comes from `_NUMERIC_BOUNDS` through `bound_error`,
854 # which is also what the CLI flags writing these same settings are checked
855 # against, so the two surfaces cannot answer differently (issue #748). A
856 # required key that is absent is validated at its default; an absent
857 # optional one is None and passes.
858 for setting, value in (
859 ("rounds", jury.get("rounds", 1)),
860 ("timeout", jury.get("timeout", 600)),
861 ("total_timeout", jury.get("total_timeout")),
862 ("phase_timeout", jury.get("phase_timeout")),
863 ("retries", jury.get("retries", 0)),
864 ):
865 message = bound_error(setting, value)
866 if message:
867 errors.append(message)
869 # Final-verdict mode (issue #220): "chair" or "vote".
870 decision = jury.get("decision")
871 if decision is not None and str(decision).strip().lower() not in ("chair", "vote"):
872 errors.append(f"jury.decision must be 'chair' or 'vote' (got {decision!r}).")
874 # Animated theater defaults (issue #364): theater is a bool, style is enum.
875 theater = jury.get("theater")
876 if theater is not None and not isinstance(theater, bool):
877 errors.append(f"jury.theater must be true or false (got {theater!r}).")
878 style = jury.get("theater_style")
879 if style is not None and str(style).strip().lower() not in ("flat", "pixel"):
880 errors.append(f"jury.theater_style must be 'flat' or 'pixel' (got {style!r}).")
882 # Panel routing (hard when present and not a known kind, issue #747), for the
883 # reason `[[agent]] tier` is hard: nothing equals "tiered" but "tiered", so
884 # `routing = "teired"` would otherwise be read as the default and the run
885 # would quietly buy the uniform panel the operator asked it not to.
886 routing = jury.get("routing")
887 if routing is not None and (
888 not isinstance(routing, str) or routing.strip().lower() not in KNOWN_ROUTINGS
889 ):
890 errors.append(f"jury.routing must be one of {', '.join(KNOWN_ROUTINGS)} (got {routing!r}).")
892 # The attribution footer (issue #911) is a bool, like `theater`: a string
893 # "false" is truthy, so accepting it would keep the footer the operator turned off.
894 output_cfg = jury.get("output")
895 if isinstance(output_cfg, dict):
896 attribution = output_cfg.get("attribution")
897 if attribution is not None and not isinstance(attribution, bool):
898 errors.append(f"jury.output.attribution must be true or false (got {attribution!r}).")
900 # Every other boolean switch (hard, docs audit 2026-09-29), for the reason
901 # #911 gave `attribution`: they were read with `bool(...)`, and the string
902 # "false" is truthy, so `verify = "false"` ran verification, passing
903 # `--strict-config`. TOML has real booleans; a quoted one is a mistake.
904 for dotted in _BOOL_SETTINGS:
905 table, _, key = dotted.rpartition(".")
906 holder = jury.get(table) if table else jury
907 value = holder.get(key) if isinstance(holder, dict) else None
908 if value is not None and not isinstance(value, bool):
909 errors.append(f"jury.{dotted} must be true or false (got {value!r}).")
910 # `[jury.context] mode` (hard, same audit): anything else was quietly read as
911 # "diff-only", so `mode = "expand"` reviewed without the context asked for.
912 context_mode = (
913 jury.get("context", {}).get("mode") if isinstance(jury.get("context"), dict) else None
914 )
915 if context_mode is not None and not (
916 isinstance(context_mode, str) and context_mode.strip().lower() in KNOWN_CONTEXT_MODES
917 ):
918 errors.append(
919 f"jury.context.mode must be one of {', '.join(KNOWN_CONTEXT_MODES)} "
920 f"(got {context_mode!r})."
921 )
923 # Adaptive rounds (issue #40): max_rounds >= 1 (hard); early_stop is a bool.
924 max_rounds_error = bound_error("max_rounds", jury.get("max_rounds"))
925 if max_rounds_error:
926 errors.append(max_rounds_error)
928 # Large-diff handling (issue #31): [jury.diff] sizes are positive ints.
929 diff_cfg = jury.get("diff", {})
930 if isinstance(diff_cfg, dict):
931 for key in ("max_bytes", "chunk_max_bytes"):
932 message = bound_error(f"diff.{key}", diff_cfg.get(key))
933 if message:
934 errors.append(message)
936 # CI gate severities (issue #718): hard, like `effort`, and for the same
937 # reason. `fail_on = ["majr"]` matches no group, so the one setting that
938 # decides whether CI fails would report a green PASS quoting the typo, on
939 # every run, forever — a silently disabled gate is worse than no gate.
940 ci_cfg = jury.get("ci", {})
941 if isinstance(ci_cfg, dict):
942 fail_on = ci_cfg.get("fail_on")
943 if fail_on is not None:
944 # A scalar is accepted here because `_ci_from_dict` wraps one in a
945 # list; validating the raw value would refuse a config that loads.
946 message = fail_on_error(
947 fail_on if isinstance(fail_on, list) else [fail_on], "jury.ci.fail_on"
948 )
949 if message:
950 errors.append(message)
951 # The fail-closed guards (hard, docs audit 2026-09-29): a negative or
952 # non-integer value used to be clamped or defaulted at load, so
953 # `min_vendors = -1` silently disabled the cross-vendor guard and
954 # `min_reviews = "lots"` silently read as "off", both passing
955 # `--strict-config`. Same table and message as the `--min-vendors` /
956 # `--min-reviews` flags.
957 for key in ("min_vendors", "min_reviews"):
958 message = bound_error(f"ci.{key}", ci_cfg.get(key))
959 if message:
960 errors.append(message)
962 agents_data = data.get("agent", [])
963 if not isinstance(agents_data, list):
964 raise ConfigError(_AGENTS_NOT_AN_ARRAY)
966 # At least one agent (hard).
967 if not agents_data:
968 errors.append("no agents configured; define at least one [[agent]] entry.")
970 seen_names: set = set()
971 enabled_names: set = set()
972 for idx, agent in enumerate(agents_data):
973 if not isinstance(agent, dict):
974 errors.append(_agent_not_a_table_message(idx))
975 continue
977 for key in agent:
978 if key not in KNOWN_AGENT_KEYS:
979 warnings.append(
980 f"unknown key 'agent[{idx}].{key}' (expected one of "
981 f"{', '.join(KNOWN_AGENT_KEYS)})."
982 )
984 name = agent.get("name", "")
985 label = name or f"agent[{idx}]"
987 # Normalise the vendor ONCE, here, and let every later rule read the
988 # normalised value (issue #701, review round 2). A vendor that is
989 # recognised must be recognised by every rule: deciding "commandless
990 # API vendor" on the raw string made `vendor = "XAI-API"` a known
991 # vendor that was nonetheless failed for having no `command`.
992 # `vendor_value` is kept for the messages, which quote what the
993 # operator actually wrote.
994 vendor_value = agent.get("vendor", "")
995 vendor = normalise_vendor(vendor_value)
997 # The adapter is the PROTOCOL, the vendor is the IDENTITY (issue #705).
998 # Every question below of the form "how is this seat invoked?" — does it
999 # need a `command`, which sandbox flag does it accept — is asked of the
1000 # adapter, which is the vendor itself unless the operator said otherwise.
1001 adapter_value = agent.get("adapter")
1002 adapter = adapter_key(vendor, adapter_value)
1004 # Unknown adapter (HARD, unlike an unknown vendor). An unknown vendor
1005 # still names a seat that can run; it just answers to `cli` at the gate.
1006 # An unknown adapter names a protocol this build does not have, and the
1007 # only fallbacks are to guess — which is precisely the failure #705 is
1008 # about: `openai` guessed `codex exec` onto `cursor-agent` and the seat
1009 # died in half a second, mid-run, having already been paid for. A typo
1010 # here is named before the panel starts, like `effort`. The message
1011 # itself lives in `unknown_adapter_error`, because `make_adapter` asks
1012 # the same question at the other end and must give the same answer
1013 # (issue #708).
1014 adapter_error = unknown_adapter_error(adapter_value, label)
1015 if adapter_error is not None:
1016 errors.append(adapter_error)
1018 # Unique, non-empty name (hard for duplicates).
1019 if not name:
1020 errors.append(f"agent[{idx}] is missing a non-empty 'name'.")
1021 elif name in seen_names:
1022 errors.append(f"duplicate agent name '{name}'.")
1023 else:
1024 seen_names.add(name)
1026 # A local OpenAI-compatible agent (issue #43) talks to an HTTP
1027 # ``endpoint`` (default ``http://localhost:11434/v1``) instead of a CLI,
1028 # so it does not require a ``command``; it does need a ``model``. A
1029 # hosted-API agent (issue #430) likewise talks HTTP instead of a CLI —
1030 # to the vendor's fixed, non-configurable endpoint — so it also needs
1031 # no ``command``, but does need a ``model``; it has no ``endpoint`` to
1032 # validate since the URL isn't a config value. Every other vendor
1033 # requires a non-empty ``command``.
1034 command = agent.get("command", "")
1035 has_endpoint = bool(agent.get("endpoint"))
1036 # A CLI adapter does not read `endpoint` (#901 review): `make_adapter`
1037 # selects it by key and spawns `command` whatever `endpoint` says. The key
1038 # was worse than unused — it made this check treat the seat as HTTP, skip
1039 # every `command` rule below, and made the least-privilege audit skip the
1040 # seat — so a seat that runs a CLI and names an endpoint is refused, and
1041 # its `command` is still checked.
1042 if has_endpoint and adapter in CLI_ADAPTERS:
1043 errors.append(
1044 f"agent '{label}' names an 'endpoint', but its adapter '{adapter}' "
1045 f"runs 'command' and never reads one; remove 'endpoint', or use "
1046 f"vendor/adapter 'local' or 'openai-compatible' for an HTTP seat."
1047 )
1048 has_endpoint = False
1049 # Asked of the ADAPTER, not the vendor (#705): whether a seat needs a
1050 # `command` is a fact about how it is invoked. `vendor = "openai",
1051 # adapter = "cli"` runs a CLI and needs one; `vendor = "openai",
1052 # adapter = "openai-api"` makes an HTTP call and does not.
1053 is_local_or_http = is_commandless_vendor(adapter) or has_endpoint
1054 if is_local_or_http:
1055 if not agent.get("model"):
1056 who = (
1057 f"vendor '{vendor_value}'"
1058 if adapter == vendor
1059 else f"adapter '{adapter_value}'"
1060 )
1061 warnings.append(
1062 f"agent '{label}' ({who}) has no 'model'; the "
1063 f"server or API call will likely reject the request."
1064 )
1065 endpoint = agent.get("endpoint")
1066 if endpoint:
1067 e_errors, e_warnings = _endpoint_issues(endpoint, label)
1068 errors.extend(e_errors)
1069 warnings.extend(e_warnings)
1070 elif not command:
1071 # A custom adapter registered with `SPAWNS_PROCESS = False` is its own
1072 # invocation — its `run()` calls the backend — so it has no `command`
1073 # to be missing (docs audit 2026-09-29). docs/configuration.md has
1074 # always shown such a seat without one, and this refused it. A
1075 # registered adapter that spawns still needs one, like a built-in CLI.
1076 if _REGISTERED_ADAPTER_SPAWNS.get(adapter) is not False:
1077 errors.append(f"agent '{label}' is missing a non-empty 'command'.")
1078 elif _is_relative_path_command(command):
1079 # A relative path with separators (e.g. ./tools/codex, bin/agy) could
1080 # resolve a binary from an attacker-influenced location (#293/F-6).
1081 # Require a bare name (resolved on PATH) or an absolute path.
1082 errors.append(
1083 f"agent '{label}' command '{command}' is a relative path; use a "
1084 f"bare name (resolved on PATH) or an absolute path."
1085 )
1086 elif os.environ.get(_REQUIRE_ABSOLUTE_COMMAND_ENV) and not Path(command).is_absolute():
1087 # Strict opt-in (issue #296): in a hardened/CI context, refuse even a
1088 # bare name so a poisoned PATH can't resolve a shim — require an
1089 # absolute path for every agent command.
1090 errors.append(
1091 f"agent '{label}' command '{command}' is not an absolute path; "
1092 f"{_REQUIRE_ABSOLUTE_COMMAND_ENV} requires every agent command to "
1093 f"be an absolute path."
1094 )
1096 # `enabled` and `prompt_mode` (hard, docs audit 2026-09-29): `enabled =
1097 # "false"` is truthy and seated the agent it meant to bench, and any
1098 # `prompt_mode` but "arg" was read as "stdin", so `prompt_mode = "args"`
1099 # piped the prompt to a CLI that expected it on argv.
1100 enabled_value = agent.get("enabled")
1101 if enabled_value is not None and not isinstance(enabled_value, bool):
1102 errors.append(f"agent '{label}' enabled must be true or false (got {enabled_value!r}).")
1103 prompt_mode = agent.get("prompt_mode")
1104 if prompt_mode is not None and not (
1105 isinstance(prompt_mode, str) and prompt_mode.lower() in KNOWN_PROMPT_MODES
1106 ):
1107 errors.append(
1108 f"agent '{label}' prompt_mode must be one of "
1109 f"{', '.join(KNOWN_PROMPT_MODES)} (got {prompt_mode!r})."
1110 )
1112 # Per-agent timeout (hard if present and invalid).
1113 a_timeout = agent.get("timeout", 600)
1114 if not isinstance(a_timeout, int) or isinstance(a_timeout, bool) or a_timeout <= 0:
1115 errors.append(
1116 f"agent '{label}' timeout must be a positive integer (got {a_timeout!r})."
1117 )
1119 # `api_key_env` names an environment variable and is echoed into
1120 # diagnostics, so it is bounded to a real env var name (see
1121 # redaction.safe_env_var_name). Warn rather than fail: the vendor
1122 # default still works, but the operator should not discover the
1123 # silent fallback by wondering why their variable is ignored.
1124 #
1125 # The rejected value is deliberately NOT quoted back. Reaching this
1126 # branch means it contains characters outside the safe set — which is
1127 # exactly the class (control characters, ANSI escapes, quotes) that must
1128 # not be written to a terminal or spliced into a report. Naming the
1129 # agent and stating the rule locates the problem without reproducing it.
1130 env_var = agent.get("api_key_env")
1131 if env_var is not None and safe_env_var_name(env_var, "") != env_var:
1132 warnings.append(
1133 f"agent '{label}' api_key_env is not a valid environment variable "
1134 f"name (expected {ENV_VAR_NAME_RULE}); the vendor default will be used."
1135 )
1137 # `headers` carries the extra HTTP headers a hosted adapter sends
1138 # (issue #716). It was never checked at all: `_from_dict` turned every
1139 # non-table into `{}`, so `headers = "Authorization = Bearer y"` passed
1140 # `--config-validate` AND `--strict-config` and the seat ran with no
1141 # extra headers — failing at the remote API, or, for a routing header
1142 # the provider honours a default for, reviewing quietly against a
1143 # backend nobody chose.
1144 #
1145 # The severity follows this module's split, not the fact that the key is
1146 # newly checked: a shape that CANNOT become headers is hard (like an
1147 # unknown `effort`), a shape that becomes working headers by a documented
1148 # coercion is a warning (like a malformed `api_key_env` name that falls
1149 # back to the vendor default). Hard: not a table, and a non-string key —
1150 # neither can be a header name or map. Soft: a non-string VALUE, which
1151 # `_from_dict` sends as `str(value)`.
1152 raw_headers = agent.get("headers")
1153 if raw_headers is not None and not isinstance(raw_headers, dict):
1154 errors.append(_headers_not_a_table_message(label, raw_headers))
1155 elif isinstance(raw_headers, dict):
1156 for header_name, header_value in raw_headers.items():
1157 if not isinstance(header_name, str):
1158 errors.append(_headers_bad_key_message(label, header_name))
1159 elif not isinstance(header_value, str):
1160 warnings.append(
1161 _headers_coerced_value_warning(label, header_name, header_value)
1162 )
1164 # Reasoning effort (hard when present and not a known level, issue
1165 # #662): a typo like `effort = "max"` would otherwise be silently
1166 # dropped, and the operator would pay for a run they think is deeper
1167 # than it is.
1168 effort = agent.get("effort")
1169 if effort is not None and (
1170 not isinstance(effort, str) or effort.strip().lower() not in KNOWN_EFFORTS
1171 ):
1172 errors.append(
1173 f"agent '{label}' effort must be one of "
1174 f"{', '.join(KNOWN_EFFORTS)} (got {effort!r})."
1175 )
1177 # Cost tier (hard when present and not a known kind, issue #714), for
1178 # the reason `effort` is hard: `tier = "cheap"` would otherwise be read
1179 # as the default and the seat silently treated as frontier — the exact
1180 # opposite of what the operator wrote.
1181 tier = agent.get("tier")
1182 if tier is not None and (
1183 not isinstance(tier, str) or tier.strip().lower() not in KNOWN_TIERS
1184 ):
1185 errors.append(
1186 f"agent '{label}' tier must be one of {', '.join(KNOWN_TIERS)} (got {tier!r})."
1187 )
1189 # Sampling temperature. Hard when present and not a number in range,
1190 # for the reason `effort` is: `temperature = "1"` or `= 5` would
1191 # otherwise fall back to the greedy default, and a model that loops at 0
1192 # (gpt-oss) returns an empty review the operator believes they fixed.
1193 # Only the local adapter sends it; on any other seat it is a warning,
1194 # because the seat still runs — without the setting the operator wrote.
1195 temperature = agent.get("temperature")
1196 if temperature is not None:
1197 if not is_valid_temperature(temperature):
1198 errors.append(
1199 f"agent '{label}' temperature must be a number from "
1200 f"{TEMPERATURE_MIN:g} to {TEMPERATURE_MAX:g} (got {temperature!r})."
1201 )
1202 elif adapter != "local":
1203 warnings.append(
1204 f"agent '{label}' temperature applies only to local seats "
1205 f"(adapter '{adapter}' does not send it); ignored."
1206 )
1208 # Known vendor (soft). The warning names the CONSEQUENCE, not just the
1209 # fact: the fallback seat still runs, but it answers to `cli` at the
1210 # cross-vendor gate, so two of them are one vendor (issue #701).
1211 if not is_recognised_vendor(vendor):
1212 warnings.append(
1213 f"agent '{label}' has unknown vendor '{vendor_value}' (expected one "
1214 f"of {', '.join(recognised_vendors())}); using the generic "
1215 f"'{GENERIC_VENDOR}' fallback, which counts as vendor "
1216 f"'{GENERIC_VENDOR}' for min_vendors — two such seats are one "
1217 f"vendor, not two."
1218 )
1220 if name and agent.get("enabled", True):
1221 enabled_names.add(name)
1223 # Chair must reference an enabled agent (soft). The literal "rotate" is a
1224 # valid special value (deterministic per-run rotation) and never warns.
1225 # Only a chair the operator WROTE is checked (docs audit 2026-09-29): an
1226 # absent one is `default_chair`, the first enabled agent, which is enabled
1227 # by construction. This used to assume `claude`, so a config with no
1228 # `claude` seat and no `chair` failed `--strict-config` over a value the run
1229 # never used.
1230 chair = jury.get("chair")
1231 if chair is not None and enabled_names and chair != "rotate" and chair not in enabled_names:
1232 warnings.append(
1233 f"jury.chair '{chair}' is not an enabled agent (enabled: "
1234 f"{', '.join(sorted(enabled_names)) or 'none'}); the first "
1235 "enabled agent will be used as fallback."
1236 )
1238 if errors:
1239 raise ConfigError("invalid configuration:\n - " + "\n - ".join(errors))
1241 if strict and warnings:
1242 raise ConfigError(
1243 "configuration warnings treated as errors (strict mode):\n - "
1244 + "\n - ".join(warnings)
1245 )
1247 return warnings
1250@dataclass
1251class AgentSpec:
1252 """One configured seat on the panel.
1254 ``vendor`` is normalised on construction (issue #701, review round 3) and is
1255 therefore the ONLY spelling any reader ever sees. Round 2 normalised at each
1256 rule instead, which left every new reader free to forget: ``_detect_warnings``
1257 compared the raw string while ``_unavailable_reason`` compared the normalised
1258 one, so a single ``vendor = "XAI-API"`` seat got two different diagnoses out
1259 of one ``--doctor`` run. Doing it here — in the one place a spec comes into
1260 existence, whatever built it — is what makes "every rule reads one value"
1261 true by construction rather than by review.
1263 Normalising is only ``strip().lower()``, so provenance survives it: the
1264 operator's vendor *name* is preserved, only its whitespace and case are not.
1265 Messages that must quote the file verbatim (``validate_config``) read the raw
1266 TOML dict, not this field. A non-string vendor normalises to ``""`` — a
1267 config mistake to warn about, not a crash in a later ``.lower()``.
1268 """
1270 name: str
1271 vendor: str
1272 command: str = ""
1273 model: str | None = None
1274 timeout: int = 600
1275 enabled: bool = True
1276 extra_args: list[str] = field(default_factory=list)
1277 # OpenAI-compatible base URL for a local/open-weight or hosted API agent (issue #43).
1278 endpoint: str | None = None
1279 # Universal agent extensions
1280 api_key_env: str | None = None
1281 prompt_mode: str | None = None
1282 headers: dict[str, str] = field(default_factory=dict)
1283 # Reasoning effort: "low" | "medium" | "high" (issue #662). None leaves the
1284 # vendor default alone. Mapped per vendor by ``adapters.effort_args``.
1285 effort: str | None = None
1286 # The protocol this seat is invoked through (issue #705): which adapter
1287 # builds its command line. ``None`` means "the vendor's shipped adapter",
1288 # which is what every configuration written before this key existed means —
1289 # so their argv is unchanged, byte for byte. Declared last, after every
1290 # other field, so no positional construction site is silently re-bound.
1291 # Read it through :attr:`adapter_key`, never directly: the fallback to the
1292 # vendor belongs in one place.
1293 adapter: str | None = None
1294 # Cost tier (issue #714): "frontier" (default; may anchor a routed panel) or
1295 # "economical" (seated on routine diffs in place of benched frontier seats).
1296 # Normalised on construction like `vendor`; an unknown value is refused by
1297 # `validate_config` and falls back to the default here so the reader never
1298 # carries a spelling no rule knows.
1299 tier: str = DEFAULT_TIER
1300 # Sampling temperature a local seat sends. ``None`` keeps the local
1301 # adapter's greedy default of 0, byte for byte. Declared last for the reason
1302 # `adapter` was: no positional construction site is re-bound.
1303 temperature: float | None = None
1305 def __post_init__(self) -> None:
1306 # The single normalisation point. Every construction site — `_from_dict`,
1307 # `runagent`'s built-in templates, `cli`'s ad-hoc local seat, a test —
1308 # goes through it, so no reader downstream can be handed the raw form.
1309 self.vendor = normalise_vendor(self.vendor)
1310 tier = str(self.tier).strip().lower() if isinstance(self.tier, str) else ""
1311 self.tier = tier if tier in KNOWN_TIERS else DEFAULT_TIER
1312 # Same treatment for the adapter, and for the same reason (#701 r3):
1313 # `adapter = "CLI"` and `adapter = "cli"` name one protocol, so they must
1314 # be one string before any lookup sees them. A key that is *present* but
1315 # normalises to nothing — `7`, `" "` — is refused here rather than read
1316 # as "unset": `validate_config` already calls it an unknown adapter, and a
1317 # caller of `load_config` (default `validate=False`) or of `make_adapter`
1318 # builds seats without validating, so treating it as a fallback to the
1319 # vendor let that path run what validation refused.
1320 if self.adapter is not None:
1321 key = normalise_vendor(self.adapter)
1322 if not key:
1323 raise ConfigError(
1324 unknown_adapter_error(self.adapter, self.name)
1325 or f"agent {self.name!r}: adapter {self.adapter!r} names no protocol"
1326 )
1327 self.adapter = key
1329 @property
1330 def adapter_key(self) -> str:
1331 """The adapter name this seat resolves to — its ``adapter`` or its vendor."""
1332 return adapter_key(self.vendor, self.adapter)
1335@dataclass
1336class CiConfig:
1337 fail_on: list[str] = field(default_factory=lambda: ["critical", "major"])
1338 ignore_unverified: bool = True
1339 # Distinct vendors that must have CONTRIBUTED a review before the run is
1340 # allowed to stand as cross-vendor consensus (issue #682). Fails closed at
1341 # 2: the product claim is a cross-vendor jury, and a panel that collapsed to
1342 # one vendor is a different thing wearing the same output. Only applies when
1343 # the run claimed consensus in the first place — a panel with fewer distinct
1344 # vendors enabled than this is never failed by the default (see
1345 # ``metadata.collapse_reason``). ``0`` disables the guard entirely.
1346 min_vendors: int = DEFAULT_MIN_VENDORS
1347 # REVIEWS the run must be able to hand a downstream consumer before it is
1348 # worth running at all (issue #699) — one per ballot that *reviewed*, i.e.
1349 # ``panel.is_review``: a substantive scope and a voting verdict, so an
1350 # abstention is not one (#700). That is the
1351 # number a gate like `keel review --from-jury` counts: it splits the
1352 # ``reviewers`` array on ``role`` and reads the ``chair`` entry as the
1353 # panel's consensus, not as a review. ``0`` (the default)
1354 # disables it: most consumers have no minimum, and a gate that fails closed
1355 # here would break every single-agent install. Set it to what your consumer
1356 # requires and the shortfall is named before the panel runs, instead of after
1357 # the review has already been paid for.
1358 #
1359 # Deliberately NOT in ``config_hash`` (unlike ``min_vendors``): it changes
1360 # neither the orchestration nor what the panel finds, only whether the
1361 # resulting bundle is accepted, and it is re-evaluated on every run including
1362 # a cache hit — so a cached outcome cannot smuggle a shortfall past it, and
1363 # adding it would invalidate every existing cache entry for nothing.
1364 min_reviews: int = 0
1367@dataclass
1368class ContextConfig:
1369 mode: str = "diff-only" # "diff-only" or "expanded"
1370 redact_secrets: bool = True
1373@dataclass
1374class DiffConfig:
1375 """Large-diff handling policy (issue #31).
1377 ``max_bytes`` is the size (UTF-8 bytes, measured after filtering) above which
1378 a diff is either chunked or rejected. ``chunk`` enables per-file chunking;
1379 ``chunk_max_bytes`` bounds each chunk (defaults to ``max_bytes``).
1380 ``exclude_generated`` drops binary and common generated/vendored files;
1381 ``exclude``/``include`` are extra path-glob deny/allow lists.
1382 """
1384 max_bytes: int = 200_000
1385 chunk: bool = False
1386 chunk_max_bytes: int | None = None
1387 exclude_generated: bool = True
1388 exclude: list[str] = field(default_factory=list)
1389 include: list[str] = field(default_factory=list)
1392@dataclass
1393class OutputConfig:
1394 """The markdown report's footer (issue #911).
1396 ``attribution`` keeps the ai-jury footer that ends the markdown report and
1397 every comment ``jury`` posts, now naming the seats that returned a review.
1398 ``False`` removes it. Rendering-only, so it is not in ``config_hash``.
1399 """
1401 attribution: bool = True
1404#: Known keys inside each nested ``[jury.*]`` table (issue #719).
1405#:
1406#: Every one is DERIVED from the dataclass the matching ``*_from_dict`` reader
1407#: builds, rather than written out a second time, so a new field cannot be added
1408#: to ``CiConfig``/``ContextConfig``/``DiffConfig`` and leave this list behind —
1409#: which is how the top-level ``KNOWN_JURY_KEYS`` list has drifted before (#715).
1410#: The readers happen to accept exactly their dataclass's field names and no
1411#: aliases; ``tests/test_config_validation.py`` pins each tuple to the keys its
1412#: reader actually reads, so an alias added later must be added here too.
1413#:
1414#: These live here, below the dataclasses, rather than beside ``KNOWN_JURY_KEYS``
1415#: because deriving them needs the classes to exist. ``validate_config`` reads
1416#: them at call time, so the ordering in the module is not a problem.
1417KNOWN_CI_KEYS = tuple(f.name for f in dataclass_fields(CiConfig))
1418KNOWN_CONTEXT_KEYS = tuple(f.name for f in dataclass_fields(ContextConfig))
1419KNOWN_DIFF_KEYS = tuple(f.name for f in dataclass_fields(DiffConfig))
1420KNOWN_OUTPUT_KEYS = tuple(f.name for f in dataclass_fields(OutputConfig))
1422#: The nested ``[jury.*]`` tables ``_from_dict`` reads, and the keys each one
1423#: knows. ``ci``/``context``/``diff``/``output`` are the complete set: every other member of
1424#: ``KNOWN_JURY_KEYS`` is a scalar (``theater`` is a bool, ``routing`` a string),
1425#: so there is no other sub-table for a typo to disappear into.
1426KNOWN_NESTED_JURY_KEYS: dict[str, tuple[str, ...]] = {
1427 "ci": KNOWN_CI_KEYS,
1428 "context": KNOWN_CONTEXT_KEYS,
1429 "diff": KNOWN_DIFF_KEYS,
1430 "output": KNOWN_OUTPUT_KEYS,
1431}
1434@dataclass
1435class JuryConfig:
1436 rounds: int = 2
1437 chair: str = "claude"
1438 timeout: int = 600
1439 parallel: bool = True
1440 verify: bool = True
1441 agents: list[AgentSpec] = field(default_factory=list)
1442 ci: CiConfig = field(default_factory=CiConfig)
1443 context: ContextConfig = field(default_factory=ContextConfig)
1444 diff: DiffConfig = field(default_factory=DiffConfig)
1445 # The markdown report's footer (issue #911).
1446 output: OutputConfig = field(default_factory=OutputConfig)
1447 # Optional run seed. Controls the shared run RNG used by randomized
1448 # orchestration features (see orchestrator.run_jury). LLM output itself
1449 # is never made deterministic by this; only the orchestration around it.
1450 seed: int | None = None
1451 # Anonymize peer reviews shown in the round-2 debate (Chatham House rule,
1452 # issue #37): strip vendor/agent identity, relabel as "Reviewer A/B/...",
1453 # and randomize per-debater presentation order via the shared run RNG so
1454 # neither identity nor position is a stable signal. The rendered report
1455 # still attributes findings by real name. Set False for the old
1456 # identity-labeled debate path.
1457 anonymize_debate: bool = True
1458 # Prefer a chair that was NOT a round-1 reviewer when a usable non-reviewer
1459 # is available (issue #38), mitigating chair self-preference bias. Has no
1460 # effect when chair == "rotate" (rotation already picks among usable agents)
1461 # or when an explicit usable chair name is configured.
1462 prefer_non_reviewer_chair: bool = False
1463 # Demote a finding to non-blocking severity when every reviewer who raised it
1464 # is vendor "local" and no cloud reviewer corroborates it (issue #442).
1465 # Rejected alternative: a numeric per-reviewer trust weight — this categorical
1466 # rule is auditable in one line where a coefficient invites silent drift.
1467 # Off by default so the out-of-the-box CI gate is unchanged.
1468 demote_local_only: bool = False
1469 # Execution controls (issue #30). All optional and off by default so the
1470 # out-of-the-box run is unchanged. ``total_timeout``/``phase_timeout`` cap the
1471 # whole run / a single phase (None = uncapped); the effective per-agent-call
1472 # timeout is the minimum of the agent timeout, the phase budget, and the
1473 # remaining total budget. ``retries`` is the number of EXTRA attempts for
1474 # transient (retryable) failures — 0 means try once.
1475 total_timeout: int | None = None
1476 phase_timeout: int | None = None
1477 retries: int = 0
1478 # Adaptive rounds (issue #40). When ``early_stop`` is True the orchestrator
1479 # decides whether to run the debate round(s) from the round-1 convergence
1480 # signal instead of always honouring a fixed ``rounds``: a unanimous panel
1481 # stops after round 1, and disagreement runs debate up to ``max_rounds``.
1482 # A CLI ``--rounds`` (or any explicit fixed-N intent) disables early stop so
1483 # benchmarking stays reproducible. ``max_rounds`` defaults to ``rounds``.
1484 max_rounds: int | None = None
1485 early_stop: bool = False
1486 # Risk-aware auto-depth (issue #120): when True, the CLI sets rounds/verify/
1487 # early_stop from a cheap pre-review diff profile (size/paths/security), so a
1488 # trivial diff runs shallow and a risky one runs full. Off by default; the
1489 # panel is never trimmed; explicit --rounds/--verify/--early-stop override it.
1490 auto_depth: bool = False
1491 # Full-transcript output (issue: full transcript). When True, the markdown
1492 # report defaults to the chronological play-by-play (each agent's raw review,
1493 # the debate, and the chair's reasoning) instead of the consensus-first
1494 # summary. Rendering-only: it does NOT affect orchestration, so it is
1495 # deliberately excluded from ``config_hash`` and the cache key. The CLI
1496 # ``--transcript``/``--no-transcript`` override it; ``--verbose`` is summary +
1497 # transcript in one document.
1498 transcript: bool = False
1499 # Final-verdict mode (issue #220): "chair" = the chair's synthesis is the
1500 # verdict (default, historical); "vote" = the panel verdict is a tally of the
1501 # reviewers (each votes from the worst finding they raised). Rendering-only —
1502 # it does not change orchestration, so it is excluded from ``config_hash`` and
1503 # the cache key. The chair still runs (its reasoning is shown as supporting
1504 # narrative), and the severity-based CI gate is unaffected. CLI: ``--decision``.
1505 decision: str = "chair"
1506 # Animated theater view defaults (issue #364). Rendering-only side channel —
1507 # excluded from ``config_hash`` and the cache key (it never touches the
1508 # outcome). ``theater`` defaults the scene on; ``theater_style`` is "flat"
1509 # (ANSI line scene) or "pixel" (pixel-art room). The CLI ``--theater`` /
1510 # ``--theater-style`` flags override these per run. Theater is TTY-only, so
1511 # even when defaulted on it falls back to ``--live`` off an interactive
1512 # terminal (and ``pixel`` falls back to ``flat`` without truecolor/unicode).
1513 theater: bool = False
1514 theater_style: str = "flat"
1515 # Risk-aware tiered model routing (issue #524): "standard" (uniform panel) |
1516 # "tiered" (cost-optimized with frontier anchor). Normalised on construction
1517 # like `AgentSpec.tier`, its companion key; an unknown value is refused by
1518 # `validate_config` and falls back to the default here so no reader carries a
1519 # spelling the vocabulary lacks (#747).
1520 routing: str = DEFAULT_ROUTING
1521 # Pre-pass static analysis hints (issue #523): inject linter hints into prompt context
1522 hints: bool = False
1524 def __post_init__(self) -> None:
1525 # The single normalisation point for `routing`, for the reason
1526 # `AgentSpec.__post_init__` is one for `tier`: `_from_dict` is not the
1527 # only way a config is built — `--doctor` and `jury run-agent` load
1528 # without validating, and tests and programmatic callers construct
1529 # `JuryConfig` directly — and every one of those readers compares against
1530 # the literal "tiered". A value that survives to a reader unnormalised
1531 # takes the `standard` path while claiming to be something else.
1532 routing = str(self.routing).strip().lower() if isinstance(self.routing, str) else ""
1533 self.routing = routing if routing in KNOWN_ROUTINGS else DEFAULT_ROUTING
1535 @property
1536 def effective_max_rounds(self) -> int:
1537 """Round ceiling for adaptive mode: ``max_rounds`` or ``rounds``."""
1538 return self.max_rounds if self.max_rounds is not None else self.rounds
1540 @property
1541 def enabled_agents(self) -> list[AgentSpec]:
1542 return [a for a in self.agents if a.enabled]
1545def default_chair(agents) -> str:
1546 """The chair a config that names none gets: its first ENABLED agent (pure).
1548 The one reader of the default (docs audit 2026-09-29), so validation and the
1549 run cannot disagree about it. A disabled first seat is skipped because it
1550 cannot chair; with no enabled seat the first seat stands (the run then falls
1551 back to the first usable agent), and with no seat at all ``claude``.
1552 """
1553 for agent in agents:
1554 if agent.enabled:
1555 return agent.name
1556 return agents[0].name if agents else "claude"
1559def _ci_from_dict(data: dict) -> CiConfig:
1560 # `ConfigError`, not the `AttributeError` a `.get` on a string would raise:
1561 # materialisation is reached WITHOUT validation whenever a caller uses
1562 # `load_config`'s default `validate=False` (a Python caller; `jury run-agent`
1563 # validates since #903), and such a caller can handle `ConfigError` where a
1564 # traceback would just be a crash (issue #729, following #716).
1565 if not isinstance(data, dict):
1566 raise ConfigError(_nested_table_message("ci"))
1567 fail_on = data.get("fail_on", ["critical", "major"])
1568 if not isinstance(fail_on, list):
1569 fail_on = [fail_on]
1570 fail_on = [str(s).strip().lower() for s in fail_on if str(s).strip()]
1571 return CiConfig(
1572 fail_on=fail_on,
1573 ignore_unverified=bool(data.get("ignore_unverified", True)),
1574 min_vendors=_non_negative_int(data.get("min_vendors"), DEFAULT_MIN_VENDORS),
1575 min_reviews=_non_negative_int(data.get("min_reviews"), 0),
1576 )
1579def _context_from_dict(data: dict) -> ContextConfig:
1580 if not isinstance(data, dict):
1581 raise ConfigError(_nested_table_message("context"))
1582 mode = str(data.get("mode", "diff-only")).strip().lower()
1583 if mode not in KNOWN_CONTEXT_MODES:
1584 mode = KNOWN_CONTEXT_MODES[0]
1585 return ContextConfig(mode=mode, redact_secrets=bool(data.get("redact_secrets", True)))
1588def _output_from_dict(data: dict) -> OutputConfig:
1589 if not isinstance(data, dict):
1590 raise ConfigError(_nested_table_message("output"))
1591 return OutputConfig(attribution=bool(data.get("attribution", True)))
1594def _str_list(value) -> list[str]:
1595 """Coerce a config value into a clean list of non-empty strings."""
1596 if isinstance(value, str):
1597 value = [value]
1598 if not isinstance(value, list):
1599 return []
1600 return [str(v).strip() for v in value if str(v).strip()]
1603def _diff_from_dict(data: dict) -> DiffConfig:
1604 if not isinstance(data, dict):
1605 raise ConfigError(_nested_table_message("diff"))
1606 default = DiffConfig()
1607 return DiffConfig(
1608 max_bytes=_opt_positive_int(data.get("max_bytes")) or default.max_bytes,
1609 chunk=bool(data.get("chunk", default.chunk)),
1610 chunk_max_bytes=_opt_positive_int(data.get("chunk_max_bytes")),
1611 exclude_generated=bool(data.get("exclude_generated", default.exclude_generated)),
1612 exclude=_str_list(data.get("exclude", [])),
1613 include=_str_list(data.get("include", [])),
1614 )
1617def _seed_from_dict(jury: dict) -> int | None:
1618 """Parse ``[jury] seed`` into an int, or None when absent/invalid.
1620 A non-integer or boolean seed is treated as "no seed" rather than an error:
1621 the seed only governs orchestration randomness, so a malformed value should
1622 degrade gracefully to the unseeded (still deterministic-orchestration) path.
1623 """
1624 raw = jury.get("seed")
1625 if raw is None or isinstance(raw, bool):
1626 return None
1627 try:
1628 return int(raw)
1629 except (TypeError, ValueError):
1630 return None
1633def _opt_positive_int(raw) -> int | None:
1634 """Coerce an optional positive-int config value, else None.
1636 Used for the optional execution budgets (issue #30) and ``max_rounds``
1637 (issue #40). A missing, boolean, non-numeric, or non-positive value degrades
1638 to None (uncapped) rather than raising, so ``_from_dict`` stays tolerant when
1639 called without validation; :func:`validate_config` is what reports the hard
1640 error for an explicit bad value.
1641 """
1642 if raw is None or isinstance(raw, bool):
1643 return None
1644 try:
1645 value = int(raw)
1646 except (TypeError, ValueError):
1647 return None
1648 return value if value > 0 else None
1651def _from_dict(data: dict) -> JuryConfig:
1652 jury = data.get("jury", {})
1653 # The top-level tables get the same treatment the nested ones got in #731: this
1654 # function is reached without validation on real paths (`jury run-agent`,
1655 # `load_config`'s default), and a scalar where a table was meant used to fall
1656 # through to `AttributeError` here instead of the `ConfigError` those callers
1657 # already handle (issue #732).
1658 if not isinstance(jury, dict):
1659 raise ConfigError(_JURY_NOT_A_TABLE)
1660 agents_raw = data.get("agent", [])
1661 if not isinstance(agents_raw, list):
1662 raise ConfigError(_AGENTS_NOT_AN_ARRAY)
1663 default_timeout = int(jury.get("timeout", 600))
1664 agents: list[AgentSpec] = []
1665 for idx, raw in enumerate(agents_raw):
1666 if not isinstance(raw, dict):
1667 raise ConfigError(_agent_not_a_table_message(idx))
1668 # `headers` no longer coerces a non-table to `{}` (issue #716). That
1669 # coercion is what made the mistake invisible: the seat materialised
1670 # cleanly carrying no headers, and only the remote API — or nobody —
1671 # ever noticed.
1672 #
1673 # It raises `ConfigError` rather than falling through to an
1674 # `AttributeError` from `.items()`, because materialisation is reached
1675 # WITHOUT validation whenever a caller uses `load_config`'s default
1676 # `validate=False` (review r1) — a Python caller; `jury run-agent`
1677 # validates since #903. Such a caller can handle `ConfigError`, and an
1678 # uncaught traceback would be a regression for it, not a fix.
1679 # The messages are the shared builders, so the two paths say the same
1680 # thing and neither echoes the value.
1681 #
1682 # A non-string VALUE is coerced, as before, and warned about in
1683 # `validate_config`: `X-Retries = 3` is a header that works.
1684 headers_label = raw.get("name") or f"agent[{len(agents)}]"
1685 raw_headers = raw.get("headers", {})
1686 if not isinstance(raw_headers, dict):
1687 raise ConfigError(_headers_not_a_table_message(headers_label, raw_headers))
1688 for key in raw_headers:
1689 if not isinstance(key, str):
1690 raise ConfigError(_headers_bad_key_message(headers_label, key))
1691 headers_dict = {k: str(v) for k, v in raw_headers.items()}
1692 api_key_env_val = str(raw["api_key_env"]) if raw.get("api_key_env") else None
1693 prompt_mode_val = str(raw["prompt_mode"]) if raw.get("prompt_mode") else None
1694 raw_effort = raw.get("effort")
1695 effort_val = str(raw_effort).strip().lower() if isinstance(raw_effort, str) else None
1696 raw_tier = raw.get("tier")
1697 tier_val = str(raw_tier).strip().lower() if isinstance(raw_tier, str) else ""
1698 # An invalid value is refused by `validate_config`; here it reads as
1699 # unset, like an unknown `effort`, so this reader never carries it.
1700 raw_temperature = raw.get("temperature")
1701 temperature_val = float(raw_temperature) if is_valid_temperature(raw_temperature) else None
1702 agents.append(
1703 AgentSpec(
1704 name=raw["name"],
1705 # Passed through raw on purpose: `AgentSpec.__post_init__` is the
1706 # one place that normalises a vendor, so normalising here as well
1707 # would create a second place to keep in step (issue #701, r3).
1708 vendor=raw.get("vendor", "unknown"),
1709 # ``command`` is optional for local/HTTP agents (issue #43).
1710 command=raw.get("command", ""),
1711 model=raw.get("model"),
1712 timeout=int(raw.get("timeout", default_timeout)),
1713 enabled=bool(raw.get("enabled", True)),
1714 extra_args=list(raw.get("extra_args", [])),
1715 endpoint=raw.get("endpoint"),
1716 api_key_env=api_key_env_val,
1717 prompt_mode=prompt_mode_val,
1718 headers=headers_dict,
1719 effort=effort_val or None,
1720 tier=tier_val or DEFAULT_TIER,
1721 # Raw, like `vendor`: `AgentSpec.__post_init__` normalises both,
1722 # and a second normalisation here would be a second place to keep
1723 # in step (issue #701 r3, extended by #705).
1724 adapter=raw.get("adapter"),
1725 temperature=temperature_val,
1726 )
1727 )
1728 return JuryConfig(
1729 rounds=int(jury.get("rounds", 2)),
1730 chair=jury.get("chair", default_chair(agents)),
1731 timeout=default_timeout,
1732 parallel=bool(jury.get("parallel", True)),
1733 verify=bool(jury.get("verify", True)),
1734 agents=agents,
1735 ci=_ci_from_dict(jury.get("ci", {})),
1736 context=_context_from_dict(jury.get("context", {})),
1737 diff=_diff_from_dict(jury.get("diff", {})),
1738 output=_output_from_dict(jury.get("output", {})),
1739 seed=_seed_from_dict(jury),
1740 anonymize_debate=bool(jury.get("anonymize_debate", True)),
1741 prefer_non_reviewer_chair=bool(jury.get("prefer_non_reviewer_chair", False)),
1742 demote_local_only=bool(jury.get("demote_local_only", False)),
1743 total_timeout=_opt_positive_int(jury.get("total_timeout")),
1744 phase_timeout=_opt_positive_int(jury.get("phase_timeout")),
1745 retries=max(0, int(jury.get("retries", 0) or 0)),
1746 max_rounds=_opt_positive_int(jury.get("max_rounds")),
1747 early_stop=bool(jury.get("early_stop", False)),
1748 auto_depth=bool(jury.get("auto_depth", False)),
1749 transcript=bool(jury.get("transcript", False)),
1750 decision=(str(jury.get("decision", "chair")).strip().lower() or "chair"),
1751 theater=bool(jury.get("theater", False)),
1752 theater_style=(str(jury.get("theater_style", "flat")).strip().lower() or "flat"),
1753 # Passed through raw on purpose, like `AgentSpec`'s `vendor`:
1754 # `JuryConfig.__post_init__` is the one place that normalises a routing
1755 # kind, so normalising here as well would create a second place to keep
1756 # in step (#747).
1757 routing=jury.get("routing", DEFAULT_ROUTING),
1758 hints=bool(jury.get("hints", False)),
1759 )
1762def config_hash(config: JuryConfig) -> str:
1763 """Return a stable SHA-256 hash of the EFFECTIVE jury configuration.
1765 The hash is a function of the resolved configuration only (no timestamps,
1766 no diff text), so the same config always produces the same digest and a
1767 changed config produces a different one. This anchors reproducibility
1768 metadata: two runs with an identical config hash were orchestrated under
1769 identical settings.
1771 The seed is intentionally excluded so the hash describes the *configuration*
1772 independent of which run seed was chosen; the seed is recorded separately in
1773 run metadata.
1774 """
1775 import hashlib
1776 import json
1778 canonical = {
1779 "rounds": config.rounds,
1780 "chair": config.chair,
1781 "timeout": config.timeout,
1782 "parallel": config.parallel,
1783 "verify": config.verify,
1784 "total_timeout": config.total_timeout,
1785 "phase_timeout": config.phase_timeout,
1786 "retries": config.retries,
1787 "max_rounds": config.max_rounds,
1788 "early_stop": config.early_stop,
1789 "auto_depth": config.auto_depth,
1790 # Orchestration-affecting toggles (issue #122): both change how a run is
1791 # conducted, so the "same hash ⇒ same orchestration" promise must include
1792 # them.
1793 "anonymize_debate": config.anonymize_debate,
1794 "prefer_non_reviewer_chair": config.prefer_non_reviewer_chair,
1795 "demote_local_only": config.demote_local_only,
1796 # The static-analysis pre-pass changes what Round 1 is shown, and the
1797 # routing mode changes which models the panel is built from, so two runs
1798 # that disagree about either are not the same run and must not share a
1799 # cache entry (#715). Like `min_vendors` before them, adding these keys
1800 # invalidates every existing review cache entry ONCE on upgrade — noted
1801 # in the CHANGELOG, since a user's first run after the bump is a full one.
1802 "hints": config.hints,
1803 "routing": config.routing,
1804 "ci": {
1805 "fail_on": list(config.ci.fail_on),
1806 "ignore_unverified": config.ci.ignore_unverified,
1807 # `min_vendors` belongs here for the same reason the other two do: it
1808 # decides the outcome of a run, so two runs that disagree about it are
1809 # not the same run. Adding it invalidates every existing review cache
1810 # entry ONCE on upgrade — noted in the CHANGELOG, since a user's first
1811 # run after the bump is a full one.
1812 "min_vendors": config.ci.min_vendors,
1813 },
1814 "context": {
1815 "mode": config.context.mode,
1816 "redact_secrets": config.context.redact_secrets,
1817 },
1818 "diff": {
1819 "max_bytes": config.diff.max_bytes,
1820 "chunk": config.diff.chunk,
1821 "chunk_max_bytes": config.diff.chunk_max_bytes,
1822 "exclude_generated": config.diff.exclude_generated,
1823 "exclude": list(config.diff.exclude),
1824 "include": list(config.diff.include),
1825 },
1826 "agents": [
1827 {
1828 "name": a.name,
1829 "vendor": a.vendor,
1830 "command": a.command,
1831 "endpoint": a.endpoint,
1832 "model": a.model,
1833 "timeout": a.timeout,
1834 "enabled": a.enabled,
1835 "extra_args": list(a.extra_args),
1836 # Effort changes the model id / request body an agent sends, so
1837 # it is orchestration-affecting and must split the cache key.
1838 "effort": a.effort,
1839 # The adapter decides which command line is built, so two seats
1840 # that differ only in it are not the same run and must not share
1841 # a cache entry (issue #705). Present ONLY when it differs from
1842 # the vendor: a seat with no `adapter`, or one naming its own
1843 # vendor, means exactly what it meant before this key existed, so
1844 # its canonical payload — and therefore every existing cache
1845 # entry — stays byte-identical. The conditional is the point,
1846 # not an optimisation: this change invalidates no one's cache.
1847 **({"adapter": a.adapter_key} if a.adapter_key != a.vendor else {}),
1848 # The tier decides which seats a routed panel keeps, so two
1849 # benches that differ in it are not the same run (issue #714).
1850 # Conditional for the reason `adapter` is: a seat with no `tier`
1851 # means exactly what it meant before the key existed, so every
1852 # existing cache entry stays byte-identical.
1853 **({"tier": a.tier} if a.tier != DEFAULT_TIER else {}),
1854 # How the prompt reaches the seat: piped to stdin, or appended to
1855 # argv as the last argument (issue #746). That is the invocation
1856 # protocol, so a cached outcome produced under one must not be
1857 # served for a run configured with the other — the rule that put
1858 # `adapter` here, one field along. Conditional for the reason
1859 # `adapter` is: `None` and `""` both mean the `stdin` fallback in
1860 # `GenericCLIAdapter._prompt_mode`, which is what a config that never
1861 # named the key has always meant, so its canonical payload — and
1862 # therefore every existing cache entry — stays byte-identical.
1863 #
1864 # The test is "was it written", not "does it resolve to something
1865 # other than the default", which is where this parts company with
1866 # `adapter` and `tier`: an explicit `prompt_mode = "stdin"`, or an
1867 # `"ARG"` beside an `"arg"`, splits the key even though the seat is
1868 # invoked identically. Deliberate. Unlike those two, this field is
1869 # neither validated nor normalised on construction, so folding
1870 # spellings together here would put a second, more precise reader
1871 # of the vocabulary next to the adapter's own — the duplication
1872 # #701 r3 removed. The cost is one extra run for a config that
1873 # writes the default out longhand; the alternative is a stale hit.
1874 **({"prompt_mode": a.prompt_mode} if a.prompt_mode else {}),
1875 # Where the request goes and under whose key (issue #716). A
1876 # provider-routing header — `X-Route: premium`, an OpenRouter
1877 # `HTTP-Referer`, an Azure deployment selector — can put a
1878 # byte-identical seat in front of a different model, and
1879 # `api_key_env` can put it in front of a different account. Both
1880 # are therefore orchestration-affecting under the same rule as
1881 # `min_vendors` above, and two configs that disagree about
1882 # either are not the same run.
1883 #
1884 # Unconditional, unlike `adapter`: there is no spelling of these
1885 # keys that means "what it meant before", so this DOES invalidate
1886 # every existing review cache entry once on upgrade. That is the
1887 # honest price of a key that was wrong, and it is noted in the
1888 # CHANGELOG — the user's first run after the bump is a full one.
1889 # Sorted so the digest depends on the mapping, not on the order
1890 # the table happened to be written in.
1891 "api_key_env": a.api_key_env,
1892 "headers": sorted(a.headers.items()),
1893 # The sampling temperature changes what a local seat answers —
1894 # at 0 gpt-oss returns nothing, at 1 a review — so a cached
1895 # outcome from one must not serve the other. Conditional for the
1896 # reason `tier` is: an unset key means what it meant before it
1897 # existed, so every existing cache entry stays byte-identical.
1898 **({"temperature": a.temperature} if a.temperature is not None else {}),
1899 }
1900 for a in config.agents
1901 ],
1902 }
1903 payload = json.dumps(canonical, sort_keys=True, separators=(",", ":"))
1904 return hashlib.sha256(payload.encode("utf-8")).hexdigest()
1907def load_raw_config(path: str | Path | None = None) -> dict:
1908 """Return the raw config dict for *path*, or the built-in default.
1910 If *path* is None, look for ``jury.toml`` in the current directory and
1911 fall back to :data:`DEFAULT_CONFIG` when it is absent. An explicit *path*
1912 that does not exist raises ``FileNotFoundError``.
1913 """
1914 if path is None:
1915 candidate = Path("jury.toml")
1916 if not candidate.exists():
1917 return DEFAULT_CONFIG
1918 path = candidate
1919 path = Path(path)
1920 if not path.exists():
1921 raise FileNotFoundError(f"Config not found: {path}")
1922 return _read_toml_bounded(path)
1925def load_config(
1926 path: str | Path | None = None,
1927 validate: bool = False,
1928 strict: bool = False,
1929) -> JuryConfig:
1930 """Load jury config from *path*, or fall back to the built-in default.
1932 If *path* is None, look for ``jury.toml`` in the current directory.
1934 When *validate* is True, the resolved config dict is checked with
1935 :func:`validate_config` before being materialized; a ``ConfigError`` is
1936 raised on hard-invalid input (and on warnings when *strict* is True).
1937 Validation is opt-in so existing callers stay unaffected.
1938 """
1939 data = load_raw_config(path)
1940 if validate:
1941 validate_config(data, strict=strict)
1942 return _from_dict(data)