Coverage for src/ai_jury/cli.py: 99%
1270 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Command-line entry point: ``jury``.
3Examples:
4 jury --pr 123 # review a GitHub PR
5 jury --pr 123 --post # ...and post the verdict as a comment
6 jury --diff-file changes.diff # review a local diff file
7 jury --diff-file - # read a diff from stdin
8 jury --mock # offline pipeline demo (no live CLIs)
9 jury --doctor # local readiness diagnostics
10 jury --config-validate # validate jury.toml and exit
11"""
13from __future__ import annotations
15import argparse
16import contextlib
17import io
18import json
19import os
20import shutil
21import sys
22from importlib import resources
23from pathlib import Path
25from . import __version__, configtrust, panel
26from . import doctor as doctor_module
27from .adapters import EFFORT_LEVELS, effort_warnings, make_adapter
28from .ci import evaluate_ci, fail_on_error
29from .classification import classify, label_strings
30from .config import (
31 ConfigError,
32 bound_error,
33 load_config,
34 load_raw_config,
35 validate_config,
36)
37from .github import (
38 apply_labels,
39 issue_body,
40 post_inline_comments,
41 post_issue_comment,
42 post_pr_comment,
43 pr_context,
44 pr_diff,
45)
46from .metadata import (
47 build_run_metadata,
48 claimed_vendors,
49 collapse_reason,
50 resolve_min_vendors,
51)
52from .orchestrator import review_diff, run_jury
53from .policy import PolicyError, load_policy
54from .redaction import redact
55from .report import render, render_footer, render_live_step, render_transcript
57# Hard ceiling on raw diff ingestion. The per-run ``diff.max_bytes`` budget is
58# only applied *after* the full diff is read and split, so an unbounded
59# ``stdin``/``--diff-file`` read could OOM the process before that cap engages
60# (security audit 2026-06-13). This ceiling sits far above any realistic review
61# budget; it exists solely to bound memory against a hostile/huge input.
62_MAX_DIFF_INGEST_BYTES = 64 * 1024 * 1024 # 64 MiB
65def _read_capped(fh, source: str) -> str:
66 """Read from ``fh``, refusing inputs above the ingest ceiling.
68 The cap is enforced on **bytes**, not characters: a text read of N chars can
69 hold up to 4N bytes for multi-byte UTF-8, so a char ceiling would admit
70 several times the intended memory (security audit 2026-06-13, red-team).
71 Callers pass a binary stream for real input (``sys.stdin.buffer`` / a file
72 opened ``"rb"``); a text stream is also accepted (its read is measured by its
73 UTF-8 byte length) so test doubles and unusual streams still work.
74 """
75 data = fh.read(_MAX_DIFF_INGEST_BYTES + 1)
76 if isinstance(data, str):
77 if len(data.encode("utf-8", "replace")) > _MAX_DIFF_INGEST_BYTES:
78 raise SystemExit(
79 f"error: {source} exceeds the {_MAX_DIFF_INGEST_BYTES}-byte ingest limit"
80 )
81 return data
82 if len(data) > _MAX_DIFF_INGEST_BYTES:
83 raise SystemExit(f"error: {source} exceeds the {_MAX_DIFF_INGEST_BYTES}-byte ingest limit")
84 return data.decode("utf-8", errors="replace")
87def _checked_revision(value: str, flag: str) -> str:
88 """Reject a revision that cannot safely reach ``git``'s argv (issue #367).
90 ``run`` uses argv, never a shell, so quoting is not the risk — a value starting
91 with ``-`` is: git would read it as an option rather than a revision. Refused
92 rather than escaped, and ``--`` is passed at the call site as a second guard.
93 Empty is refused too, since it would silently widen the diff.
94 """
95 revision = (value or "").strip()
96 if not revision:
97 raise SystemExit(f"error: {flag} needs a revision")
98 if revision.startswith("-"):
99 raise SystemExit(
100 f"error: {flag} revision {revision!r} may not start with '-' "
101 "(git would read it as an option)"
102 )
103 return revision
106def _git_diff(argv: list[str], label: str) -> str:
107 """Run a read-only git command and return its stdout, or exit with its error."""
108 import subprocess # local: keeps the module importable where git is absent
110 try:
111 proc = subprocess.run(argv, capture_output=True, text=True, timeout=120)
112 except (OSError, subprocess.SubprocessError) as exc:
113 raise SystemExit(f"error: could not run git for {label}: {redact(str(exc))[0]}") from None
114 if proc.returncode != 0:
115 detail = (proc.stderr or "").strip().splitlines()
116 raise SystemExit(
117 f"error: git could not resolve {label}"
118 + (f": {redact(detail[0])[0]}" if detail else "")
119 )
120 if not proc.stdout.strip():
121 raise SystemExit(f"error: {label} produced an empty diff — nothing to review")
122 # Same ingest ceiling every other source honours; _read_capped wants a handle.
123 return _read_capped(io.StringIO(proc.stdout), label)
126def _bundled_sample_diff() -> str:
127 """The offline-demo diff shipped inside the package (issue #21).
129 Read from ``ai_jury/data/sample.diff`` via ``importlib.resources`` so it
130 resolves the same way from a wheel, a zipapp, or a source checkout — no
131 ``examples/`` directory and no repo needed. Kept byte-identical to
132 ``examples/sample.diff`` by a test.
133 """
134 return resources.files("ai_jury").joinpath("data/sample.diff").read_text(encoding="utf-8")
137def _read_diff(args) -> tuple[str, str]:
138 """Return (diff, context)."""
139 if getattr(args, "commit", None):
140 rev = _checked_revision(args.commit, "--commit")
141 # `git show` of a merge commit prints no diff by default; -m picks the
142 # first-parent view so a merge is reviewable rather than silently empty.
143 return _git_diff(
144 ["git", "show", "--format=", "--patch", "-m", "--first-parent", rev, "--"],
145 f"commit {rev}",
146 ), ""
147 if getattr(args, "commits", None):
148 rev = _checked_revision(args.commits, "--commits")
149 return _git_diff(["git", "diff", rev, "--"], f"range {rev}"), ""
150 if args.pr:
151 return pr_diff(args.pr, args.repo), pr_context(args.pr, args.repo)
152 if args.issue:
153 # Issue mode (issue #221): the issue's rendered text takes the diff slot;
154 # there is no separate context block (title/labels are folded into it).
155 return issue_body(args.issue, args.repo), ""
156 if args.diff_file:
157 if args.diff_file == "-":
158 # Prefer the byte stream so the cap is exact; fall back to the text
159 # stream (e.g. a StringIO test double) which lacks ``.buffer``.
160 return _read_capped(getattr(sys.stdin, "buffer", sys.stdin), "stdin"), ""
161 try:
162 with Path(args.diff_file).open("rb") as fh:
163 return _read_capped(fh, args.diff_file), ""
164 except (OSError, UnicodeDecodeError) as exc:
165 raise SystemExit(
166 f"error reading diff file '{args.diff_file}': {redact(str(exc))[0]}"
167 ) from None
168 if getattr(args, "mock", False):
169 # Offline demo (issue #21): `--mock` with no diff source reviews a diff
170 # bundled in the package, so `jury --mock` / `jury --mock --theater`
171 # deliver the deliberation the docs promise on a fresh install — no
172 # checkout, no PR, no network. A real (non-`--mock`) run still requires
173 # an explicit source. Announced on stderr so the payments.py sample is
174 # never mistaken for the caller's own change.
175 print(
176 "no diff source given; reviewing the bundled offline-demo diff. "
177 "Pass --diff-file/--pr/--commit to review your own change.",
178 file=sys.stderr,
179 )
180 return _bundled_sample_diff(), ""
181 raise SystemExit(
182 "error: provide one of --pr, --issue, --diff-file, --commit, --commits "
183 "(or --diff-file - for stdin)"
184 )
187def build_parser() -> argparse.ArgumentParser:
188 p = argparse.ArgumentParser(
189 prog="jury",
190 description="Cross-vendor multi-agent PR review jury.",
191 # The subcommands are argv-intercepts (handled before this parser runs),
192 # so argparse cannot list them on its own — and until #661 nothing in
193 # `--help` mentioned that they exist at all.
194 epilog=(
195 "Subcommands (each takes its own --help): init, config, comment, "
196 "apply, replay, run-agent, cache clear, examples, guide."
197 ),
198 )
199 src = p.add_argument_group("input")
200 src.add_argument("--pr", help="GitHub PR number/URL to review (uses `gh`)")
201 src.add_argument(
202 "--issue",
203 help="GitHub issue number/URL to review for completeness/clarity (uses "
204 "`gh`); runs the full jury with an issue-quality rubric",
205 )
206 src.add_argument("--repo", help="owner/name for --pr/--issue (defaults to current repo)")
207 src.add_argument("--diff-file", help="path to a diff file, or '-' for stdin")
208 src.add_argument("--commit", help="review the diff one commit introduces (needs a git repo)")
209 src.add_argument(
210 "--commits",
211 help="review a commit range, e.g. origin/main..HEAD or HEAD~5..HEAD (needs a git repo)",
212 )
214 p.add_argument("--config", help="path to jury.toml (default: ./jury.toml or built-in)")
215 p.add_argument(
216 "--policy",
217 type=Path,
218 default=None,
219 help="path to an optional repository review policy file (default: "
220 "auto-discover .jury/policy.toml or jury-policy.toml); "
221 "missing policy files are allowed",
222 )
223 p.add_argument(
224 "--context-mode",
225 choices=["diff-only", "expanded"],
226 default=None,
227 help="context policy: diff-only sends only the diff; expanded includes PR context",
228 )
229 p.add_argument(
230 "--redact",
231 dest="redact",
232 action="store_true",
233 default=None,
234 help="redact secrets from prompt text before sending (default: from config)",
235 )
236 p.add_argument(
237 "--no-redact",
238 dest="redact",
239 action="store_false",
240 help="do not redact secrets before sending",
241 )
242 p.add_argument(
243 "--rounds",
244 type=int,
245 help="override number of rounds (1=review, 2=+debate); a fixed value "
246 "disables early-stop for reproducible benchmarking",
247 )
248 p.add_argument(
249 "--max-rounds",
250 type=int,
251 help="ceiling on adaptive rounds when early-stop is on",
252 )
253 p.add_argument(
254 "--early-stop",
255 dest="early_stop",
256 action="store_true",
257 default=None,
258 help="stop after round 1 when reviewers agree; debate only on disagreement",
259 )
260 p.add_argument(
261 "--no-early-stop",
262 dest="early_stop",
263 action="store_false",
264 help="disable adaptive early-stop (honour a fixed number of rounds)",
265 )
266 p.add_argument(
267 "--auto",
268 dest="auto",
269 action="store_true",
270 default=None,
271 help="risk-aware auto-depth: scale rounds/verify to the diff",
272 )
273 p.add_argument(
274 "--no-auto",
275 dest="auto",
276 action="store_false",
277 help="disable auto-depth (use configured/fixed rounds)",
278 )
279 p.add_argument(
280 "--total-timeout",
281 type=int,
282 help="overall wall-clock budget (seconds) for the whole run",
283 )
284 p.add_argument(
285 "--phase-timeout",
286 type=int,
287 help="per-phase wall-clock budget (seconds)",
288 )
289 p.add_argument(
290 "--retries",
291 type=int,
292 help="extra attempts for transient (timeout/rate-limit/spawn) failures",
293 )
294 p.add_argument(
295 "--max-diff-bytes",
296 type=int,
297 help="size budget for the (filtered) diff before chunking/too-large",
298 )
299 p.add_argument(
300 "--chunk",
301 dest="chunk",
302 action="store_true",
303 default=None,
304 help="chunk an over-budget diff by file instead of failing",
305 )
306 p.add_argument(
307 "--no-chunk",
308 dest="chunk",
309 action="store_false",
310 help="disable diff chunking (fail clearly when over budget)",
311 )
312 p.add_argument(
313 "--exclude",
314 action="append",
315 metavar="GLOB",
316 default=None,
317 help="exclude files matching this path glob (repeatable)",
318 )
319 p.add_argument(
320 "--include",
321 action="append",
322 metavar="GLOB",
323 default=None,
324 help="only review files matching this path glob (repeatable)",
325 )
326 p.add_argument(
327 "--seed",
328 type=int,
329 help="run seed for reproducible orchestration; mock runs with the same seed "
330 "produce byte-identical reports (overrides [jury] seed)",
331 )
332 p.add_argument("--chair", help="override the synthesizing chair agent")
333 p.add_argument(
334 "--mock", action="store_true", help="offline demo: use deterministic mock agents"
335 )
336 p.add_argument(
337 "--strict", action="store_true", help="fail if any configured agent CLI is missing"
338 )
339 p.add_argument(
340 "--min-vendors",
341 type=int,
342 default=None,
343 metavar="N",
344 help=(
345 "fail (exit 3) unless at least N distinct vendors contributed a "
346 "review; 0 disables. Default from config ([jury.ci] min_vendors, "
347 "shipped as 2). --strict checks availability at startup, this "
348 "checks participation at the end"
349 ),
350 )
351 p.add_argument(
352 "--no-min-vendors",
353 dest="min_vendors",
354 action="store_const",
355 const=0,
356 help=(
357 "opt out of the cross-vendor guard: accept a run whose panel "
358 "collapsed to a single vendor (same as --min-vendors 0)"
359 ),
360 )
361 p.add_argument(
362 "--min-reviews",
363 type=int,
364 default=None,
365 metavar="N",
366 help=(
367 "require at least N reviews for a downstream consumer (a panel ballot "
368 "that names what it read and votes; neither the chair's synthesis "
369 "record nor an abstaining ballot is one); 0 disables. Checked before "
370 "the panel runs and again on the result (exit 3). "
371 "Default from config ([jury.ci] min_reviews, shipped as 0)"
372 ),
373 )
374 p.add_argument(
375 "--verify",
376 dest="verify",
377 action="store_true",
378 default=None,
379 help="run the verification round (default: from config)",
380 )
381 p.add_argument(
382 "--no-verify",
383 dest="verify",
384 action="store_false",
385 help="skip the verification round",
386 )
387 p.add_argument(
388 "--doctor",
389 action="store_true",
390 help="print a local readiness diagnostics report and exit (no telemetry is collected or sent)",
391 )
392 p.add_argument(
393 "--json",
394 action="store_true",
395 help=(
396 "with --doctor, print the machine-readable provider export "
397 "(schema ai-jury.doctor.v1) as the only thing on stdout"
398 ),
399 )
400 p.add_argument(
401 "--write",
402 help="with --doctor, also write the diagnostics as JSON to this path (secrets redacted)",
403 )
404 p.add_argument(
405 "--effort",
406 choices=list(EFFORT_LEVELS),
407 default=None,
408 help=(
409 "reasoning effort for every agent this run (overrides [[agent]] effort); "
410 "ignored with a warning for agents whose vendor has no effort control"
411 ),
412 )
413 p.add_argument("-o", "--output", help="write the report to a file instead of stdout")
414 p.add_argument(
415 "--metadata-json",
416 metavar="PATH",
417 help="write machine-readable run metadata (durations, status, rounds) as JSON",
418 )
419 p.add_argument(
420 "--format",
421 choices=["markdown", "json", "sarif", "keel-reviews"],
422 default="markdown",
423 help="output format for stdout/--output (default: markdown); "
424 "'keel-reviews' emits one review record per panelist plus the chair",
425 )
426 p.add_argument(
427 "--decision",
428 choices=["chair", "vote"],
429 default=None,
430 help="final verdict: 'chair' synthesis (default) or panel 'vote' (tally "
431 "the reviewers); overrides [jury] decision",
432 )
433 p.add_argument(
434 "--transcript",
435 dest="transcript",
436 action="store_true",
437 default=None,
438 help="render the full play-by-play transcript (each agent's review, the "
439 "debate, and the chair's reasoning) instead of the summary report",
440 )
441 p.add_argument(
442 "--no-transcript",
443 dest="transcript",
444 action="store_false",
445 help="force the summary report even if [jury] transcript is set",
446 )
447 p.add_argument(
448 "--verbose",
449 dest="verbose",
450 action="store_true",
451 help="summary report followed by the full transcript, in one document",
452 )
453 p.add_argument(
454 "--live",
455 dest="live",
456 action="store_true",
457 help="stream each step (review, debate, verdict) to stdout as it happens; "
458 "add --pr --post to also post each step as its own PR comment",
459 )
460 p.add_argument(
461 "--theater",
462 dest="theater",
463 action="store_true",
464 default=None,
465 help="animated deliberation view of the live run (each model seated "
466 "around a table, speaking per phase, panel-vote/chair finale); needs an "
467 "interactive terminal, else falls back to --live. Can be defaulted on in "
468 "jury.toml ([jury] theater = true)",
469 )
470 p.add_argument(
471 "--no-theater",
472 dest="theater",
473 action="store_false",
474 help="disable the theater scene even if jury.toml enables it",
475 )
476 p.add_argument(
477 "--theater-style",
478 dest="theater_style",
479 choices=("flat", "pixel"),
480 default=None,
481 help="--theater scene style: 'flat' (ANSI line scene, default) or "
482 "'pixel' (pixel-art room; needs a truecolor+unicode terminal). Defaults "
483 "from jury.toml ([jury] theater_style)",
484 )
485 p.add_argument(
486 "--post-summary",
487 "--post",
488 dest="post_summary",
489 action="store_true",
490 help="post the report as a single summary comment on --pr",
491 )
492 p.add_argument(
493 "--no-attribution",
494 dest="attribution",
495 action="store_false",
496 default=None,
497 help="leave the ai-jury footer naming the seats that reviewed off the "
498 "markdown report and posted comments (default from jury.toml "
499 "[jury.output] attribution, on)",
500 )
501 p.add_argument(
502 "--post-inline",
503 dest="post_inline",
504 action="store_true",
505 help="post inline review comments for located findings on --pr",
506 )
507 p.add_argument(
508 "--post-progress",
509 dest="post_progress",
510 action="store_true",
511 help="keep a live, sticky status comment on --pr updated per round/chunk",
512 )
513 p.add_argument(
514 "--post-mode",
515 choices=["single", "phased"],
516 # None is "not passed" (read as 'single'), so a `--post-mode` with no
517 # summary to shape can be told apart from the default and refused.
518 default=None,
519 help="with --post-summary: 'single' (one comment) or 'phased' (separate "
520 "Round 1 / debate / decision comments)",
521 )
522 p.add_argument(
523 "--dry-run",
524 dest="dry_run",
525 action="store_true",
526 help="with --post-inline, print what would be posted without calling GitHub",
527 )
528 p.add_argument(
529 "--label",
530 dest="label",
531 action="store_true",
532 help="apply classification labels (review effort / risk / security) to "
533 "--pr (off by default; never applied automatically)",
534 )
535 p.add_argument(
536 "--ci",
537 action="store_true",
538 help="CI mode: exit non-zero when blocking findings remain",
539 )
540 p.add_argument(
541 "--fail-on",
542 help="comma-separated severities that fail CI (overrides config)",
543 )
544 p.add_argument(
545 "--cache",
546 action="store_true",
547 help="use the local result cache: reuse a cached outcome for an unchanged "
548 "diff+config, else run and store it (off by default)",
549 )
550 p.add_argument(
551 "--clear-cache",
552 action="store_true",
553 help="delete all local cache entries and exit (also: `jury cache clear`)",
554 )
555 p.add_argument(
556 "--cache-dir",
557 help="override the cache directory (default: $JURY_CACHE_DIR or ~/.cache/ai-jury)",
558 )
559 p.add_argument(
560 "--suggest-patches",
561 dest="suggest_patches",
562 action="store_true",
563 help="emit a separate, opt-in suggested-patches section for VERIFIED "
564 "findings (read-only; never applied automatically)",
565 )
566 p.add_argument(
567 "--patches-out",
568 metavar="PATH",
569 help="with --suggest-patches, write the patches to this file instead of "
570 "appending them after the report",
571 )
572 p.add_argument(
573 "--incremental",
574 action="store_true",
575 help="review only the diff since the last jury run on --pr when a prior "
576 "marker exists, else fall back to a full review",
577 )
578 p.add_argument("-q", "--quiet", action="store_true", help="suppress progress logs on stderr")
579 p.add_argument(
580 "--config-validate",
581 action="store_true",
582 help="validate the resolved config and exit (0 valid, 2 invalid)",
583 )
584 p.add_argument(
585 "--strict-config",
586 action="store_true",
587 help="treat configuration warnings as errors",
588 )
589 p.add_argument(
590 "--tiered",
591 action="store_true",
592 help="opt-in risk-aware tiered model routing with frontier anchor (issue #524)",
593 )
594 p.add_argument(
595 "--hints",
596 dest="hints",
597 action="store_true",
598 default=None,
599 help="run local static analysis pre-pass (Ruff/ESLint) to inject hints (issue #523)",
600 )
601 p.add_argument(
602 "--no-hints",
603 dest="hints",
604 action="store_false",
605 help="skip the static analysis pre-pass even if jury.toml enables it",
606 )
607 p.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
608 return p
611def _run_apply(rest: list[str]) -> int:
612 """Handle `jury apply` (issue #521): apply verified patch suggestions."""
613 from .patches import (
614 apply_patch_suggestion,
615 parse_patch_suggestions,
616 preview_patch_suggestion,
617 )
619 sub = argparse.ArgumentParser(
620 prog="jury apply", description="Apply verified suggested patches to the repository."
621 )
622 sub.add_argument(
623 "index",
624 nargs="?",
625 default=None,
626 help="1-indexed patch suggestion number to apply, or 'all'. Required — "
627 "applying every suggestion is the wrong default for a command that "
628 "writes to the working tree",
629 )
630 sub.add_argument(
631 "--report",
632 "-r",
633 help="Path to a markdown report file or patch file (defaults to stdin)",
634 )
635 sub.add_argument(
636 "--dry-run",
637 action="store_true",
638 help="Print the paths each suggestion would touch and write nothing",
639 )
640 sub.add_argument(
641 "--yes",
642 "-y",
643 action="store_true",
644 help="Skip the confirmation prompt; required when stdin is not a terminal",
645 )
646 ns = sub.parse_args(rest)
648 content = ""
649 if ns.report:
650 p = Path(ns.report)
651 try:
652 if not p.is_file():
653 print(f"Error: report file not found: {ns.report}", file=sys.stderr)
654 return 2
655 content = p.read_text(encoding="utf-8")
656 except (OSError, UnicodeDecodeError) as exc:
657 print(
658 f"Error reading report file '{ns.report}': {redact(str(exc))[0]}", file=sys.stderr
659 )
660 return 2
661 elif sys.stdin is not None and not sys.stdin.isatty():
662 content = sys.stdin.read()
663 else:
664 print(
665 "Error: provide a report file via --report <file> or pipe a report via stdin",
666 file=sys.stderr,
667 )
668 return 2
670 suggestions = parse_patch_suggestions(content)
671 if not suggestions:
672 print("No verified patch suggestions found in the provided report.", file=sys.stderr)
673 return 1
675 if ns.index is None:
676 print(
677 f"Error: choose what to apply — an index from 1 to {len(suggestions)}, or 'all'.\n"
678 " `jury apply --dry-run all` shows what each one would touch.",
679 file=sys.stderr,
680 )
681 return 2
683 if ns.index.lower() == "all":
684 selected = list(enumerate(suggestions, 1))
685 else:
686 try:
687 target_idx = int(ns.index) - 1
688 except ValueError:
689 print(f"Error: invalid index '{ns.index}'", file=sys.stderr)
690 return 2
691 if not (0 <= target_idx < len(suggestions)):
692 print(
693 f"Error: patch index {ns.index} out of range (found {len(suggestions)} suggestions)",
694 file=sys.stderr,
695 )
696 return 2
697 selected = [(target_idx + 1, suggestions[target_idx])]
699 # Preview before writing, always. The report is derived from a diff that may be
700 # attacker-influenced and its suggestions were written by an LLM, so the
701 # operator gets git's own answer about what would change *before* anything
702 # changes. The old output was printed after the write had happened (#605).
703 print("These suggestions would touch:", file=sys.stderr)
704 for number, suggestion in selected:
705 paths, refusal = preview_patch_suggestion(suggestion)
706 listed = ", ".join(paths) if paths else "(nothing git could read)"
707 print(f" [{number}] {suggestion.file}: {listed}", file=sys.stderr)
708 if refusal is not None:
709 print(f" would be refused: {refusal}", file=sys.stderr)
711 if ns.dry_run:
712 print("Dry run: nothing was written.", file=sys.stderr)
713 return 0
715 if not ns.yes:
716 # Piping a report in is exactly the unattended case, and stdin may already
717 # be consumed by the report itself — so silence is refused rather than read
718 # as consent.
719 if sys.stdin is None or not sys.stdin.isatty():
720 print(
721 "Error: refusing to write without confirmation. Re-run with --yes to "
722 "confirm, or --dry-run to preview.",
723 file=sys.stderr,
724 )
725 return 2
726 answer = input(f"Apply {len(selected)} suggestion(s)? [y/N] ").strip().lower()
727 if answer not in ("y", "yes"):
728 print("Aborted; nothing was written.", file=sys.stderr)
729 return 1
731 success_count = 0
732 total = len(selected)
733 for number, suggestion in selected:
734 ok, msg = apply_patch_suggestion(suggestion)
735 if ok:
736 success_count += 1
737 print(f"✓ [{number}/{total}] {msg}")
738 else:
739 print(f"✗ [{number}/{total}] {msg}", file=sys.stderr)
740 return 0 if success_count > 0 else 1
743def _run_comment_command(rest: list[str]) -> int:
744 """Handle ``jury comment`` (issue #11): parse an allowlisted PR-comment
745 command and either print the resolved jury args or dispatch the run.
747 Returns 2 on a rejected/invalid command (so a workflow can ignore it), else
748 the dispatched run's exit code (or 0 with --print-args).
749 """
750 import shlex
752 from .commands import CommandError, parse_comment
754 sub = argparse.ArgumentParser(prog="jury comment", add_help=True)
755 sub.add_argument("--text", required=True, help="the PR comment body to parse")
756 sub.add_argument("--pr", help="PR number/URL to review and post back to")
757 sub.add_argument("--repo", help="owner/name (defaults to current repo)")
758 sub.add_argument(
759 "--print-args",
760 dest="print_args",
761 action="store_true",
762 help="print the resolved jury args instead of running",
763 )
764 sub.add_argument(
765 "--no-post",
766 dest="no_post",
767 action="store_true",
768 help="do not post the result back as a summary comment",
769 )
770 ns = sub.parse_args(rest)
772 try:
773 parsed = parse_comment(ns.text)
774 except CommandError as exc:
775 print(f"comment command rejected: {redact(str(exc))[0]}", file=sys.stderr)
776 return 2
778 inner = parsed.to_cli_args()
779 if ns.pr:
780 inner += ["--pr", ns.pr]
781 if not ns.no_post:
782 inner += ["--post-summary"]
783 if ns.repo:
784 inner += ["--repo", ns.repo]
786 if ns.print_args:
787 print(" ".join(shlex.quote(a) for a in inner))
788 return 0
789 return main(inner)
792_AGENT_BLURB = {
793 "claude": "Claude Code (Anthropic)",
794 "codex": "Codex CLI (OpenAI)",
795 "agy": "Antigravity (Google) — opt-in only, cannot be confined",
796 "qwen": "local / open-weight via Ollama (free, offline)",
797 "claude-api": "hosted Anthropic API (ANTHROPIC_API_KEY, no CLI needed)",
798 "codex-api": "hosted OpenAI API (OPENAI_API_KEY, no CLI needed)",
799 "gemini-api": "hosted Google Gemini API (GEMINI_API_KEY, no CLI needed)",
800 "openrouter": "hosted OpenRouter API (OPENROUTER_API_KEY)",
801 "deepseek": "hosted DeepSeek API (DEEPSEEK_API_KEY)",
802 "groq": "hosted Groq API (GROQ_API_KEY)",
803 "aider": "generic CLI coding agent (Aider)",
804}
807def _default_init_agents(available: dict) -> list[str]:
808 """The agents an interactive `jury init` pre-fills: detected, never opt-in-only.
810 agy stays listed and can be typed in, but pressing Enter never seats it.
811 """
812 from .scaffold import KNOWN_AGENTS, implicit_choices
814 detected = implicit_choices(n for n in KNOWN_AGENTS if available.get(n))
815 return detected or implicit_choices(KNOWN_AGENTS)
818def _init_available() -> dict:
819 """Map each known agent name to whether it is reachable right now."""
820 from .config import AgentSpec
821 from .scaffold import KNOWN_AGENTS, agent_templates
823 templates = agent_templates()
824 out = {}
825 for name in KNOWN_AGENTS:
826 try:
827 out[name] = make_adapter(AgentSpec(**templates[name])).available()
828 except Exception: # noqa: BLE001 - detection is best-effort
829 out[name] = False
830 return out
833def _stdin_is_terminal() -> bool:
834 """Whether ``jury init`` can prompt: stdin exists and is a terminal.
836 ``sys.stdin`` is ``None`` when the process has no stdin at all (a detached
837 or daemonised launch), and ``None.isatty()`` is an ``AttributeError`` (#897).
838 """
839 return sys.stdin is not None and sys.stdin.isatty()
842def _init_interactive(available: dict, input_fn=input, local_endpoint=None, models_fn=None) -> dict:
843 """Prompt for jury settings; returns kwargs for scaffold.build_config.
845 ``input_fn`` and ``models_fn`` are injectable for testing (the latter lists
846 local models). Defaults are pre-filled from the detected agents/models so
847 pressing Enter accepts a sensible config.
848 """
849 from .scaffold import KNOWN_AGENTS
851 if models_fn is None:
852 from .adapters import list_local_models as models_fn
854 print("Configure a review jury (jury.toml).\n", file=sys.stderr)
855 for name in KNOWN_AGENTS:
856 mark = "available" if available.get(name) else "not found"
857 print(f" - {name}: {_AGENT_BLURB[name]} [{mark}]", file=sys.stderr)
858 default_agents = _default_init_agents(available)
859 raw_agents = input_fn(f"\nAgents to include [default: {','.join(default_agents)}]: ").strip()
860 agents = [a.strip() for a in raw_agents.split(",") if a.strip()] or default_agents
862 rounds_raw = input_fn("Rounds — 1=review, 2=+debate [2]: ").strip()
863 rounds = int(rounds_raw) if rounds_raw.isdigit() else 2
865 chair_default = agents[0] if agents else "claude"
866 chair = input_fn(f"Chair agent [{chair_default}]: ").strip() or chair_default
868 verify = (input_fn("Run verification round? [Y/n]: ").strip().lower() or "y") != "n"
870 local_model = None
871 has_local = any(a in agents for a in ("qwen", "local"))
872 if has_local:
873 from .scaffold import pick_default_model
875 models = models_fn(local_endpoint or "http://localhost:11434/v1")
876 if models:
877 default = pick_default_model(models)
878 print("\nLocal models available on the server:", file=sys.stderr)
879 for i, m in enumerate(models, 1):
880 star = " (default)" if m == default else ""
881 print(f" {i}. {m}{star}", file=sys.stderr)
882 raw = input_fn(f"Pick a local model [number or name, default: {default}]: ").strip()
883 if raw.isdigit() and 1 <= int(raw) <= len(models):
884 local_model = models[int(raw) - 1]
885 elif raw:
886 local_model = raw
887 else:
888 local_model = default
889 else:
890 print(
891 "\n(could not reach the local server to list models; using the default)",
892 file=sys.stderr,
893 )
894 local_model = input_fn("Local model name [qwen2.5-coder:7b]: ").strip() or None
896 # Reasoning effort (issue #662). Skippable: Enter leaves it unset, so
897 # nothing is written and every agent keeps its vendor default. Asked last so
898 # the established question order is unchanged.
899 effort_raw = input_fn("Reasoning effort — low/medium/high [skip]: ").strip().lower()
900 effort = effort_raw if effort_raw in EFFORT_LEVELS else None
902 return {
903 "agents": agents,
904 "rounds": rounds,
905 "chair": chair,
906 "verify": verify,
907 "local_model": local_model,
908 "effort": effort,
909 }
912def _init_wizard(available: dict, input_fn=input, local_endpoint=None, models_fn=None) -> dict:
913 """Guided, numbered-option setup for ``jury init --wizard`` (issue #231).
915 Mirrors :func:`_init_interactive`'s injectable params for offline testing.
916 Every question is SKIPPABLE: pressing Enter leaves the setting unset, so it
917 falls back to the built-in default and is NOT written to ``jury.toml`` (which
918 keeps the generated file minimal). Returns kwargs for ``scaffold.build_config``
919 containing only the values the user explicitly chose.
920 """
921 from .scaffold import KNOWN_AGENTS
923 if models_fn is None:
924 from .adapters import list_local_models as models_fn
926 def ask(prompt: str) -> str:
927 return input_fn(prompt).strip()
929 def choose(prompt: str, options: list[str], default_idx: int) -> int | None:
930 """Print numbered options and read a 1-based pick. Enter -> None (skip)."""
931 print(prompt, file=sys.stderr)
932 for i, label in enumerate(options, 1):
933 star = " (default)" if i - 1 == default_idx else ""
934 print(f" {i}. {label}{star}", file=sys.stderr)
935 raw = ask("Pick a number [Enter to keep default]: ")
936 if not raw:
937 return None
938 if raw.isdigit() and 1 <= int(raw) <= len(options):
939 return int(raw) - 1
940 return None
942 print(
943 "jury init --wizard — guided setup (writes jury.toml).\n"
944 "Every question is optional: press Enter to keep the default and skip it;\n"
945 "skipped settings are left at their built-in defaults (not written).\n",
946 file=sys.stderr,
947 )
949 # Reviewers (always written — like plain init).
950 for name in KNOWN_AGENTS:
951 mark = "available" if available.get(name) else "not found"
952 print(f" - {name}: {_AGENT_BLURB[name]} [{mark}]", file=sys.stderr)
953 default_agents = _default_init_agents(available)
954 raw_agents = ask(f"\nReviewers to include [default: {','.join(default_agents)}]: ")
955 agents = [a.strip() for a in raw_agents.split(",") if a.strip()] or default_agents
957 kwargs: dict = {"agents": agents}
959 # Depth -> rounds / early_stop / auto_depth.
960 depth = choose(
961 "\nDepth:",
962 [
963 "1 round (review only)",
964 "2 rounds + debate",
965 "adaptive (early-stop)",
966 "auto-depth (scale to the diff)",
967 ],
968 default_idx=1,
969 )
970 if depth == 0:
971 kwargs["rounds"] = 1
972 elif depth == 1:
973 kwargs["rounds"] = 2
974 elif depth == 2:
975 kwargs["rounds"] = 2
976 kwargs["early_stop"] = True
977 elif depth == 3:
978 kwargs["auto_depth"] = True
980 # Decision: chair (default) or panel vote. Only written on a non-default.
981 decision = choose("\nDecision:", ["chair synthesis", "panel vote"], default_idx=0)
982 if decision == 1:
983 kwargs["decision"] = "vote"
985 # Verification (always written — like plain init).
986 verify_raw = ask("\nRun verification round? [Y/n]: ").lower()
987 if verify_raw:
988 kwargs["verify"] = verify_raw != "n"
990 # Context: diff-only (default) or expanded; redact secrets Y/n.
991 ctx = choose(
992 "\nContext sent to reviewers:",
993 ["diff-only", "expanded (include PR context)"],
994 default_idx=0,
995 )
996 if ctx == 1:
997 kwargs["context_mode"] = "expanded"
998 redact_raw = ask("Redact secrets before sending? [Y/n]: ").lower()
999 if redact_raw == "n":
1000 kwargs["redact_secrets"] = False
1002 # CI gate fail-on. Only write [jury.ci] on a non-default pick.
1003 gate = choose(
1004 "\nCI gate — fail on which severities?",
1005 ["critical,major", "critical only", "skip (never fail CI)"],
1006 default_idx=0,
1007 )
1008 if gate == 1:
1009 kwargs["ci_fail_on"] = ["critical"]
1010 elif gate == 2:
1011 kwargs["ci_fail_on"] = []
1013 # Chair (always written — like plain init; default = first reviewer).
1014 chair_default = agents[0] if agents else "claude"
1015 chair = ask(f"\nChair agent [{chair_default}]: ") or chair_default
1016 kwargs["chair"] = chair
1018 # Local model pick when a local reviewer is chosen (reuse init's logic).
1019 if any(a in agents for a in ("qwen", "local")):
1020 from .scaffold import pick_default_model
1022 models = models_fn(local_endpoint or "http://localhost:11434/v1")
1023 if models:
1024 default = pick_default_model(models)
1025 print("\nLocal models available on the server:", file=sys.stderr)
1026 for i, m in enumerate(models, 1):
1027 star = " (default)" if m == default else ""
1028 print(f" {i}. {m}{star}", file=sys.stderr)
1029 raw = ask(f"Pick a local model [number or name, default: {default}]: ")
1030 if raw.isdigit() and 1 <= int(raw) <= len(models):
1031 kwargs["local_model"] = models[int(raw) - 1]
1032 elif raw:
1033 kwargs["local_model"] = raw
1034 else:
1035 kwargs["local_model"] = default
1036 else:
1037 print(
1038 "\n(could not reach the local server to list models; using the default)",
1039 file=sys.stderr,
1040 )
1041 typed = ask("Local model name [qwen2.5-coder:7b]: ")
1042 if typed:
1043 kwargs["local_model"] = typed
1045 return kwargs
1048def _run_init(rest: list[str]) -> int:
1049 """Handle ``jury init`` (issue #107): scaffold a jury.toml."""
1050 from .config import AGY_OPT_IN_NOTE, ConfigError, validate_config
1051 from .scaffold import (
1052 KNOWN_AGENTS,
1053 OPT_IN_AGENTS,
1054 PRESETS,
1055 agent_templates,
1056 agents_needing_remote_opt_in,
1057 build_config,
1058 implicit_choices,
1059 render_toml,
1060 seat_local_agents,
1061 )
1063 sub = argparse.ArgumentParser(prog="jury init")
1064 sub.add_argument(
1065 "--preset",
1066 choices=sorted(PRESETS),
1067 help="setup preset: offline (local-only), fast (1 round), balanced "
1068 "(debate + early-stop), thorough (all agents + debate + verify)",
1069 )
1070 sub.add_argument(
1071 "--agents",
1072 help="comma-separated: claude,codex,qwen (agy only by name: it cannot be confined)",
1073 )
1074 sub.add_argument("--rounds", type=int, default=None)
1075 sub.add_argument("--chair")
1076 sub.add_argument("--verify", dest="verify", action="store_true", default=None)
1077 sub.add_argument("--no-verify", dest="verify", action="store_false")
1078 sub.add_argument("--local-model", help="model id for a local agent (qwen)")
1079 sub.add_argument("--local-endpoint", help="OpenAI-compatible base URL for a local agent")
1080 sub.add_argument("-o", "--output", default="jury.toml")
1081 sub.add_argument("--force", action="store_true", help="overwrite an existing file")
1082 sub.add_argument("--interactive", action="store_true", help="force interactive prompts")
1083 sub.add_argument(
1084 "--wizard",
1085 action="store_true",
1086 help="guided, numbered-option setup; every question is skippable (Enter "
1087 "keeps the built-in default) and only chosen keys are written",
1088 )
1089 sub.add_argument(
1090 "--list-agents", action="store_true", help="list known agents + availability and exit"
1091 )
1092 sub.add_argument(
1093 "--list-models", action="store_true", help="list local models on the server and exit"
1094 )
1095 ns = sub.parse_args(rest)
1097 # The wizard reads every answer from a terminal. With none — CI, a pipe, an
1098 # agent's shell — its first prompt died as an `EOFError` traceback (#865), so it
1099 # is refused here, before the availability probes spend anything. The listing
1100 # flags answer before the wizard would ever ask, so they are left to do so.
1101 listing = ns.list_models or ns.list_agents
1102 if ns.wizard and not listing and not _stdin_is_terminal():
1103 print(
1104 "error: the wizard needs a terminal; use `jury init --preset <name>`",
1105 file=sys.stderr,
1106 )
1107 return 2
1108 # `--interactive` is the same promise and died the same way (#897): an
1109 # `EOFError` on its first prompt from a pipe, and with no stdin at all an
1110 # `AttributeError` from the TTY test below. Refused where it would prompt —
1111 # `--agents`/`--preset` answer the questions, so it never asks then.
1112 if ns.interactive and not (listing or ns.agents or ns.preset) and not _stdin_is_terminal():
1113 print(
1114 "error: --interactive needs a terminal; use `jury init --preset <name>` "
1115 "or `jury init --agents <list>`",
1116 file=sys.stderr,
1117 )
1118 return 2
1120 from .adapters import list_local_models, local_model_listing
1121 from .redaction import redact_url_userinfo
1123 endpoint = ns.local_endpoint or "http://localhost:11434/v1"
1124 # Strip any userinfo credentials before echoing the endpoint to stdout/CI
1125 # logs (issue #316/L-7, completed in v1.5.0/L-1: structural strip catches
1126 # short and colon-less userinfo the regex missed), mirroring doctor.py.
1127 endpoint_disp = redact_url_userinfo(endpoint)
1129 if ns.list_models:
1130 models = list_local_models(endpoint)
1131 if not models:
1132 print(f"No local models found (is a server reachable at {endpoint_disp}?).")
1133 return 0
1134 print(f"Local models at {endpoint_disp}:")
1135 for m in models:
1136 print(f" - {m}")
1137 return 0
1139 available = _init_available()
1141 if ns.list_agents:
1142 for name in KNOWN_AGENTS:
1143 mark = "available" if available.get(name) else "not found"
1144 print(f"{name:8} {_AGENT_BLURB[name]:45} [{mark}]")
1145 # Show discovered local models so the user sees what they can pick.
1146 models = list_local_models(endpoint)
1147 if models:
1148 print(f"\nlocal models at {endpoint_disp}: {', '.join(models)}")
1149 return 0
1151 preset = PRESETS.get(ns.preset, {})
1152 templates = agent_templates()
1153 # Seats `jury init` writes commented out, under a hint (#864). Only plain init
1154 # fills it; the prompts and the wizard ask the operator instead.
1155 commented: list[dict] = []
1157 def _detected_agents():
1158 # Never an opt-in-only agent (agy): "detected" is a default, not a choice.
1159 return implicit_choices(n for n in KNOWN_AGENTS if available.get(n))
1161 def _selectable_agents():
1162 """Every known agent whose template scaffolds to a *valid* config here.
1164 Three hosted templates point at real vendor hosts, and `config` refuses
1165 a non-loopback endpoint unless `JURY_ALLOW_REMOTE_ENDPOINT` is set. A
1166 preset that silently includes them writes a config `jury init` then
1167 rejects — so `--preset thorough` failed outright rather than producing
1168 something usable. They stay in `--list-agents` and remain selectable by
1169 name; they are only excluded from "all" until the opt-in is present.
1170 """
1171 if os.environ.get("JURY_ALLOW_REMOTE_ENDPOINT"): 1171 ↛ 1172line 1171 didn't jump to line 1172 because the condition on line 1171 was never true
1172 return implicit_choices(KNOWN_AGENTS)
1173 needs_opt_in = set(agents_needing_remote_opt_in())
1174 return implicit_choices(n for n in KNOWN_AGENTS if n not in needs_opt_in)
1176 def _resolve_preset_agents(spec):
1177 if spec == "all":
1178 return _selectable_agents()
1179 if spec == "detected":
1180 return _detected_agents() or _selectable_agents()
1181 return list(spec)
1183 # rounds / verify / early_stop: explicit flag > preset > built-in default.
1184 rounds = ns.rounds if ns.rounds is not None else preset.get("rounds", 2)
1185 verify = ns.verify if ns.verify is not None else preset.get("verify", True)
1186 early_stop = preset.get("early_stop")
1188 # Guided wizard (issue #231): opt-in via --wizard. A numbered-option flow
1189 # where every question is skippable; only explicitly-chosen settings are
1190 # written, so the file stays minimal. Explicit, so it skips the TTY test the
1191 # prompts below use; a missing terminal was refused above instead (#865).
1192 if ns.wizard:
1193 kwargs = _init_wizard(available, local_endpoint=ns.local_endpoint)
1194 kwargs["local_endpoint"] = ns.local_endpoint
1195 if ns.local_model:
1196 kwargs["local_model"] = ns.local_model
1197 # Interactive only when neither --agents nor --preset was given and we're on a
1198 # TTY (or --interactive, refused above without one). Presets/flags are
1199 # non-interactive by design. With no stdin at all, plain init detects (#897).
1200 elif not ns.agents and not ns.preset and (ns.interactive or _stdin_is_terminal()):
1201 kwargs = _init_interactive(available, local_endpoint=ns.local_endpoint)
1202 kwargs["local_endpoint"] = ns.local_endpoint
1203 if ns.local_model:
1204 kwargs["local_model"] = ns.local_model
1205 else:
1206 if ns.agents:
1207 agents = [a.strip() for a in ns.agents.split(",") if a.strip()]
1208 elif ns.preset:
1209 agents = _resolve_preset_agents(preset["agents"])
1210 else:
1211 agents = _detected_agents()
1212 if not agents:
1213 print(
1214 "error: no agents detected and none specified; pass --agents "
1215 "or --preset (e.g. --preset offline), or run interactively.",
1216 file=sys.stderr,
1217 )
1218 if available.get("agy"):
1219 print(f"note: {AGY_OPT_IN_NOTE}.", file=sys.stderr)
1220 return 2
1221 # A local seat names a model its server lists, as the prompts and the wizard
1222 # already did (#864); plain init wrote the template's model and then warned
1223 # about it. Asked only when a local seat is chosen and no model was named.
1224 # `local_model_listing`, not `list_local_models`: only a server that answered
1225 # with no model gets its seat commented out, and a failed listing (None)
1226 # leaves the seat as it was — no evidence is not evidence of a fault (#849).
1227 local_model = ns.local_model
1228 if not local_model and any(templates.get(a, {}).get("vendor") == "local" for a in agents):
1229 agents, local_model, left_out = seat_local_agents(agents, local_model_listing(endpoint))
1230 if left_out and ns.chair in left_out:
1231 # The operator named this chair. Writing it over a commented seat warns
1232 # on every run, and picking another chair overrides them, so neither.
1233 print(
1234 f"error: the chair {ns.chair} is the local seat, but the server at "
1235 f"{endpoint_disp} lists no model; pass --local-model <model> or choose "
1236 "another --chair",
1237 file=sys.stderr,
1238 )
1239 return 2
1240 if left_out:
1241 commented = build_config(left_out, local_endpoint=ns.local_endpoint)["agent"]
1242 print(
1243 f"note: the local server at {endpoint_disp} lists no model, so "
1244 f"{', '.join(repr(a) for a in left_out)} is written commented out; "
1245 "pull one and uncomment it, or pass --local-model.",
1246 file=sys.stderr,
1247 )
1248 kwargs = {
1249 "agents": agents,
1250 "rounds": rounds,
1251 "chair": ns.chair,
1252 "verify": verify,
1253 "early_stop": early_stop,
1254 "local_model": local_model,
1255 "local_endpoint": ns.local_endpoint,
1256 }
1258 try:
1259 config = build_config(**kwargs)
1260 except ValueError as exc:
1261 print(f"error: {redact(str(exc))[0]}", file=sys.stderr)
1262 return 2
1264 # The scaffolded config must itself be valid (fail loudly if a template drifts).
1265 try:
1266 validate_config(config)
1267 except ConfigError as exc:
1268 print(f"error: generated config is invalid: {redact(str(exc))[0]}", file=sys.stderr)
1269 return 2
1271 out_path = Path(ns.output)
1272 if out_path.exists() and not ns.force:
1273 print(
1274 f"error: {out_path} already exists; pass --force to overwrite.",
1275 file=sys.stderr,
1276 )
1277 return 2
1279 out_path.write_text(render_toml(config, commented_agents=commented), encoding="utf-8")
1280 # This config is the operator's own deliberate act, so record it as trusted: the
1281 # ordinary `jury init` → `jury` path then never has to ask (#831).
1282 configtrust.record_trust(out_path, configtrust.content_digest(out_path.read_bytes()))
1283 chosen = ", ".join(a["name"] for a in config["agent"])
1284 print(f"Wrote {out_path} — panel: {chosen} · rounds: {config['jury']['rounds']}")
1285 # Seated by name only, so the operator chose it — say what they chose.
1286 if any(a["name"] in OPT_IN_AGENTS for a in config["agent"]):
1287 print(
1288 "warning: agy cannot be confined (it reads, writes and reaches the network "
1289 "even with --sandbox); do not use it on untrusted diffs. A review run warns "
1290 "about this seat, and `--strict` fails on it.",
1291 file=sys.stderr,
1292 )
1293 # A local seat whose server does not list its model is still a valid config,
1294 # so it is written — but said now, not discovered by the first review as an
1295 # `HTTP 404` (#849). `--list-models` already knew; init never asked. The
1296 # doctor's own diagnosis is reused so the two cannot disagree about a seat.
1297 from types import SimpleNamespace
1299 from .doctor import _local_model_gap
1301 for agent in config["agent"]:
1302 fields = ("name", "vendor", "adapter", "endpoint", "model")
1303 gap = _local_model_gap(SimpleNamespace(**{k: agent.get(k) for k in fields}))
1304 if gap:
1305 print(f"warning: agent '{agent['name']}': {gap[1]}", file=sys.stderr)
1306 print(f"Next: jury --config-validate --config {out_path}")
1307 print("Then: git diff main... | jury --diff-file -")
1308 return 0
1311def _config_read_error(exc: OSError, config_arg) -> str:
1312 """The one line a config that exists but cannot be read is reported with (#893).
1314 Every config load site catches ``ConfigError`` and ``FileNotFoundError``, but an
1315 unreadable ``./jury.toml`` raises ``PermissionError`` and ``--config <directory>``
1316 raises ``IsADirectoryError`` (``PermissionError`` on Windows), and both escaped as a
1317 traceback. The path comes from the exception, so an auto-discovered ``jury.toml`` is
1318 named as well as an explicit one; ``strerror`` is the reason without the errno and
1319 path ``str(exc)`` would repeat.
1320 """
1321 path = exc.filename or config_arg or "jury.toml"
1322 reason = exc.strerror or str(exc)
1323 return redact(f"error: cannot read config {path}: {reason}")[0]
1326def _config_source(config_arg) -> str:
1327 """Human-readable source of the config the jury would load."""
1328 if config_arg:
1329 return str(config_arg)
1330 return "jury.toml" if Path("jury.toml").exists() else "(built-in defaults)"
1333def _render_effective_config(cfg) -> str:
1334 """Render the EFFECTIVE resolved config as a readable summary (config show)."""
1335 on = lambda b: "on" if b else "off" # noqa: E731
1336 lines = []
1337 lines.append(
1338 f"[jury] rounds={cfg.rounds} chair={cfg.chair} verify={on(cfg.verify)} "
1339 f"parallel={on(cfg.parallel)} timeout={cfg.timeout}s"
1340 )
1341 adaptive = f"early_stop={on(cfg.early_stop)} max_rounds={cfg.effective_max_rounds}"
1342 budget = (
1343 f"total_timeout={cfg.total_timeout or '—'} "
1344 f"phase_timeout={cfg.phase_timeout or '—'} retries={cfg.retries}"
1345 )
1346 lines.append(
1347 f" {adaptive} · {budget} · seed={cfg.seed if cfg.seed is not None else '—'}"
1348 )
1349 lines.append(
1350 f"[jury.ci] fail_on={cfg.ci.fail_on} ignore_unverified={on(cfg.ci.ignore_unverified)}"
1351 )
1352 lines.append(
1353 f"[jury.context] mode={cfg.context.mode} redact_secrets={on(cfg.context.redact_secrets)}"
1354 )
1355 d = cfg.diff
1356 lines.append(
1357 f"[jury.diff] max_bytes={d.max_bytes} chunk={on(d.chunk)} "
1358 f"exclude_generated={on(d.exclude_generated)} "
1359 f"exclude={d.exclude or '[]'} include={d.include or '[]'}"
1360 )
1361 lines.append("agents:")
1362 for a in cfg.agents:
1363 flag = "" if a.enabled else " (disabled)"
1364 target = a.endpoint if a.adapter_key == "local" else (a.command or "—")
1365 model = f" model={a.model}" if a.model else ""
1366 # The adapter is shown only when it is not the vendor's own (#705), so
1367 # this line keeps its shape for every configuration that predates the key.
1368 via = f" via {a.adapter_key}" if a.adapter_key != a.vendor else ""
1369 lines.append(f" - {a.name} ({a.vendor}{via}) → {target}{model}{flag}")
1370 return "\n".join(lines)
1373def _run_config(rest: list[str]) -> int:
1374 """Handle ``jury config show|path``."""
1375 from .config import ConfigError, load_config
1377 sub = argparse.ArgumentParser(prog="jury config")
1378 sub.add_argument("action", choices=["show", "path"])
1379 sub.add_argument("--config", help="path to jury.toml (default: ./jury.toml or built-in)")
1380 ns = sub.parse_args(rest)
1382 source = _config_source(ns.config)
1383 if ns.action == "path":
1384 print(source)
1385 return 0
1387 try:
1388 cfg = load_config(ns.config, validate=True)
1389 except (ConfigError, FileNotFoundError) as exc:
1390 print(f"error: {redact(str(exc))[0]}", file=sys.stderr)
1391 return 2
1392 except OSError as exc:
1393 print(_config_read_error(exc, ns.config), file=sys.stderr)
1394 return 2
1395 print(f"source: {source}")
1396 print(_render_effective_config(cfg))
1397 return 0
1400# Cap on a prompt file. An orchestrator's prompt is a few KB of instructions
1401# plus, at most, a diff; this only bounds memory against a pathological file.
1402_MAX_PROMPT_BYTES = 8 * 1024 * 1024
1405def _read_prompt(path: str) -> tuple[str, str | None]:
1406 """Read a prompt file (or stdin for ``-``). Returns ``(text, error)``."""
1407 if path == "-":
1408 stream = sys.stdin.buffer if sys.stdin is not None else None
1409 if stream is None:
1410 return "", "cannot read the prompt from stdin: stdin is not available"
1411 raw = stream.read(_MAX_PROMPT_BYTES + 1)
1412 else:
1413 target = Path(path)
1414 if not target.is_file():
1415 return "", f"prompt file not found: {path}"
1416 try:
1417 raw = target.read_bytes()[: _MAX_PROMPT_BYTES + 1]
1418 except OSError as exc:
1419 return "", f"could not read prompt file '{path}': {redact(str(exc))[0]}"
1420 if len(raw) > _MAX_PROMPT_BYTES:
1421 return "", f"prompt exceeds the {_MAX_PROMPT_BYTES}-byte limit"
1422 text = raw.decode("utf-8", errors="replace")
1423 if not text.strip():
1424 return "", "empty prompt — nothing to send to the agent"
1425 return text, None
1428def _run_agent_parser() -> argparse.ArgumentParser:
1429 """The ``jury run-agent`` flag surface (built separately so it can be tested)."""
1430 from . import runagent
1432 sub = argparse.ArgumentParser(
1433 prog="jury run-agent",
1434 description="Run ONE configured agent for one role and print a JSON "
1435 "result with attribution. The integration point for an orchestrator "
1436 "(keel's `keel delegate run`) or a CI script that needs a single "
1437 "agent rather than the whole panel.",
1438 )
1439 sub.add_argument(
1440 "--agent",
1441 help="agent to run: a [[agent]] name from jury.toml, or a built-in "
1442 f"vendor ({', '.join(runagent.BUILTIN_AGENTS)}); append ':<model>' to "
1443 "override the model (e.g. codex:gpt-5.2)",
1444 )
1445 sub.add_argument(
1446 "--role",
1447 help="what the agent is being asked to do: "
1448 f"{'|'.join(runagent.ROLES)}. "
1449 f"{'/'.join(runagent.READ_ONLY_ROLES)} always run read-only; "
1450 f"{'/'.join(sorted(runagent.WRITE_ROLES))} need --allow-write",
1451 )
1452 sub.add_argument("--prompt-file", help="path to the prompt to send, or '-' for stdin")
1453 sub.add_argument(
1454 "--cwd",
1455 help="directory a write role (implement/fix) runs in (default: the current one); "
1456 "read-only roles on claude/codex/agy start in an empty temporary directory",
1457 )
1458 sub.add_argument(
1459 "--timeout",
1460 type=int,
1461 help="seconds before the AGENT is killed (default: the agent's configured "
1462 "timeout). This bounds the agent, never the wait — see --wait-timeout",
1463 )
1464 sub.add_argument(
1465 "--effort",
1466 choices=list(EFFORT_LEVELS),
1467 help="reasoning effort for vendors that support one (see `jury --doctor --json`)",
1468 )
1469 sub.add_argument(
1470 "--allow-write",
1471 action="store_true",
1472 help="grant the vendor's write/tool mode; required by the implement and "
1473 "fix roles, ignored by the read-only ones",
1474 )
1475 sub.add_argument(
1476 "--format",
1477 choices=["json", "text"],
1478 default="json",
1479 help="json: the full result document (default); text: only the agent's text",
1480 )
1481 sub.add_argument(
1482 "--detach",
1483 action="store_true",
1484 help="start the run in the background and print its run id immediately",
1485 )
1486 sub.add_argument(
1487 "--run-id",
1488 help="id for a detached run: letters, digits, '.', '_' or '-', max 64 "
1489 "characters (default: a random one). Only meaningful with --detach",
1490 )
1491 sub.add_argument("--wait", metavar="RUN_ID", help="block until a detached run finishes")
1492 sub.add_argument(
1493 "--wait-timeout",
1494 type=int,
1495 metavar="SECONDS",
1496 help="seconds --wait will block before giving up (default: the run's own "
1497 f"timeout + {runagent.WAIT_GRACE_S}s, else {runagent.DEFAULT_WAIT_TIMEOUT_S}s)",
1498 )
1499 sub.add_argument("--status", action="store_true", help="list detached runs and exit")
1500 sub.add_argument("--config", help="path to jury.toml (default: ./jury.toml or built-in)")
1501 sub.add_argument(
1502 "--cache-dir",
1503 help="where detached-run state lives (default: $JURY_CACHE_DIR or ~/.cache/ai-jury)",
1504 )
1505 sub.add_argument(
1506 "--mock", action="store_true", help="run the offline mock adapter instead of a real agent"
1507 )
1508 sub.add_argument(
1509 "--strict",
1510 action="store_true",
1511 help="refuse (exit 2) a read-only role whose seat draws a least-privilege "
1512 "warning, as a panel run with --strict does",
1513 )
1514 # The detached child re-enters this same command; the flag only tells it to
1515 # record its result in the run's state file. Hidden: it is not a surface
1516 # anyone should call directly.
1517 sub.add_argument("--_child", dest="child", action="store_true", help=argparse.SUPPRESS)
1518 return sub
1521def _child_argv(ns, run_id: str, python: str) -> list[str]:
1522 """The argv of the background child a ``--detach`` run spawns (pure)."""
1523 argv = [
1524 python,
1525 "-m",
1526 "ai_jury",
1527 "run-agent",
1528 "--agent",
1529 ns.agent,
1530 "--role",
1531 ns.role,
1532 "--prompt-file",
1533 str(Path(ns.prompt_file).resolve()),
1534 "--run-id",
1535 run_id,
1536 "--_child",
1537 ]
1538 if ns.cwd:
1539 argv += ["--cwd", str(Path(ns.cwd).resolve())]
1540 if ns.timeout is not None:
1541 argv += ["--timeout", str(ns.timeout)]
1542 if ns.effort:
1543 argv += ["--effort", ns.effort]
1544 if ns.allow_write:
1545 argv.append("--allow-write")
1546 if ns.config:
1547 argv += ["--config", str(Path(ns.config).resolve())]
1548 if ns.cache_dir:
1549 argv += ["--cache-dir", str(Path(ns.cache_dir).resolve())]
1550 if ns.mock:
1551 argv.append("--mock")
1552 if ns.strict:
1553 argv.append("--strict")
1554 return argv
1557def _run_run_agent(rest: list[str], spawn=None, sleep=None, clock=None) -> int:
1558 """Handle ``jury run-agent`` (issue #661): one agent, one role, one JSON result.
1560 Exit codes: ``0`` the agent ran and produced output, ``1`` it ran and
1561 failed (the JSON says why — same fail-soft vocabulary as a panel run),
1562 ``2`` the request itself was refused (bad role, missing write grant,
1563 unknown agent, unreadable prompt).
1564 """
1565 import dataclasses
1566 import time
1568 from . import runagent
1569 from .adapters import effort_args, make_adapter
1571 ns = _run_agent_parser().parse_args(rest)
1573 # Validate every id the caller supplies BEFORE it can reach a path. The
1574 # library refuses an unsafe one on its own (runagent.run_path), so this is
1575 # about the error the operator sees, not about whether the write is safe.
1576 for flag, value in (("--run-id", ns.run_id), ("--wait", ns.wait)):
1577 if value is not None:
1578 problem = runagent.check_run_id(value)
1579 if problem:
1580 print(f"error: {flag}: {problem}", file=sys.stderr)
1581 return 2
1582 if ns.run_id and not (ns.detach or ns.child):
1583 # It named nothing and recorded nothing — silently ignoring it would let
1584 # a script believe it had a handle it could later --wait on.
1585 print(
1586 "error: --run-id only applies to a detached run; add --detach, or "
1587 "drop --run-id to run in the foreground",
1588 file=sys.stderr,
1589 )
1590 return 2
1592 if ns.status:
1593 print(json.dumps({"runs": runagent.list_runs(ns.cache_dir)}, indent=2))
1594 return 0
1596 if ns.wait:
1597 if ns.wait_timeout is not None and ns.wait_timeout <= 0:
1598 print("error: --wait-timeout must be a positive number of seconds", file=sys.stderr)
1599 return 2
1600 # `--detach` reserves the id by writing the state file BEFORE it returns,
1601 # so a run with no state file was never started — a typo, or the wrong
1602 # --cache-dir. Say so now instead of blocking until a deadline expires.
1603 known = runagent.read_state(ns.wait, ns.cache_dir)
1604 if known is None:
1605 print(f"error: no such run '{ns.wait}'", file=sys.stderr)
1606 return 2
1607 # The deadline comes from what the run was actually given (its own
1608 # timeout plus head-room to write its state file), so a long agent run
1609 # is not cut short and a dead one is not waited on forever.
1610 deadline = ns.wait_timeout
1611 if deadline is None:
1612 deadline = runagent.default_wait_timeout(known)
1613 # When liveness cannot be determined, a run that has already died can
1614 # only be noticed when the deadline expires. Say so once, so a long
1615 # silence reads as a known limitation rather than a hung command. The
1616 # opening is cause-neutral and the reason names which case it is —
1617 # "this platform cannot probe" is false on POSIX for a run that simply
1618 # has not been claimed yet. The deadline always applies regardless, so
1619 # the wait is bounded either way.
1620 unknown = runagent.liveness_unknown_reason(known)
1621 if unknown:
1622 print(
1623 f"warning: cannot tell whether run '{ns.wait}' is still alive — "
1624 f"{unknown}; waiting up to {deadline:g}s for it to report",
1625 file=sys.stderr,
1626 )
1627 state, timed_out = runagent.wait_for_run(
1628 ns.wait,
1629 cache_dir=ns.cache_dir,
1630 timeout=deadline,
1631 sleep=sleep or time.sleep,
1632 clock=clock or time.monotonic,
1633 )
1634 if timed_out:
1635 print(
1636 f"error: run '{ns.wait}' did not finish within {deadline:g}s",
1637 file=sys.stderr,
1638 )
1639 return 2
1640 if state is None:
1641 # The file vanished while we waited (a `jury cache clear`, a manual
1642 # rm). Distinguished from "never existed", which returned above.
1643 print(f"error: run '{ns.wait}' disappeared while waiting", file=sys.stderr)
1644 return 2
1645 if state.get("status") == runagent.STATUS_LOST:
1646 # Returned as soon as the pid was seen gone, not at the deadline:
1647 # there is no result coming, and blocking for one is just delay.
1648 print(
1649 f"error: run '{ns.wait}' was lost — its process is gone and it "
1650 f"never recorded a result (see {runagent.output_path(ns.wait, ns.cache_dir)})",
1651 file=sys.stderr,
1652 )
1653 print(json.dumps(state, indent=2))
1654 return 0 if state.get("ok") else 1
1656 missing = [
1657 flag
1658 for flag, value in (
1659 ("--agent", ns.agent),
1660 ("--role", ns.role),
1661 ("--prompt-file", ns.prompt_file),
1662 )
1663 if not value
1664 ]
1665 if missing:
1666 print(
1667 f"error: {', '.join(missing)} required (or use --wait <run-id> / --status)",
1668 file=sys.stderr,
1669 )
1670 return 2
1672 # Role policy first: an implement run without --allow-write must be refused
1673 # before a prompt is read, a config is loaded, or a child is spawned.
1674 policy = runagent.role_policy(ns.role, ns.allow_write)
1675 if policy.refusal:
1676 print(f"error: {policy.refusal}", file=sys.stderr)
1677 return 2
1678 if policy.warning:
1679 print(f"warning: {policy.warning}", file=sys.stderr)
1681 if ns.cwd and not Path(ns.cwd).is_dir():
1682 print(f"error: --cwd is not a directory: {ns.cwd}", file=sys.stderr)
1683 return 2
1684 if ns.cwd and not policy.write:
1685 # Said, not silently dropped: a read-only role on a native CLI starts in an
1686 # empty directory, so a checkout's project settings cannot reach it.
1687 print(
1688 f"note: --cwd applies to write roles; the read-only role '{policy.role}' "
1689 f"starts in an empty temporary directory (claude/codex/agy).",
1690 file=sys.stderr,
1691 )
1693 # Validated as the panel validates it (#903): an `endpoint` on a CLI seat, a
1694 # `command` path rule, an unknown adapter — each refused here as it is by a
1695 # review, `--config-validate`, `jury config` and `--doctor`, rather than
1696 # reaching the spawner as a config the rest of the tool rejects.
1697 try:
1698 config = load_config(ns.config, validate=True)
1699 except (ConfigError, FileNotFoundError) as exc:
1700 print(redact(f"error: {exc}")[0], file=sys.stderr)
1701 return 2
1702 except OSError as exc:
1703 print(_config_read_error(exc, ns.config), file=sys.stderr)
1704 return 2
1706 # run-agent runs a config-defined command too, and a discovered config can even shadow
1707 # a built-in name (`--agent claude` with `command = "sh"`), so it needs the same trust
1708 # gate as a review (#831).
1709 try:
1710 configtrust.enforce(ns.config, config, mock=ns.mock)
1711 except configtrust.ConfigTrustError as exc:
1712 print(f"error: {exc}", file=sys.stderr)
1713 return 2
1715 spec, error = runagent.resolve_agent(config, ns.agent)
1716 if error is not None:
1717 print(f"error: {error}", file=sys.stderr)
1718 return 2
1719 if ns.timeout is not None:
1720 if ns.timeout <= 0:
1721 print("error: --timeout must be a positive number of seconds", file=sys.stderr)
1722 return 2
1723 spec = dataclasses.replace(spec, timeout=ns.timeout)
1724 if ns.effort:
1725 spec = dataclasses.replace(spec, effort=ns.effort)
1726 warning = effort_args(spec.adapter_key, ns.effort, spec.model).warning
1727 if warning:
1728 print(f"warning: {warning}", file=sys.stderr)
1730 # The panel's least-privilege audit, for this one seat (#903). A read-only
1731 # role is spawned with the argv a panel review uses, so it is audited the
1732 # same way, and `--strict` refuses it the same way. A write role is not: it
1733 # asked for write access with `--allow-write`, and the audit describes the
1734 # read-only argv, which is not the one that role runs. Before `--detach`, so a
1735 # refused seat never starts a background run.
1736 if not policy.write:
1737 from .privilege import audit_agent
1739 privilege_warnings = audit_agent(spec)
1740 for w in privilege_warnings:
1741 print(f"warning: least-privilege: {w}", file=sys.stderr)
1742 if ns.strict and privilege_warnings:
1743 print(
1744 "error: least-privilege check failed (--strict): " + "; ".join(privilege_warnings),
1745 file=sys.stderr,
1746 )
1747 return 2
1749 if ns.detach:
1750 return _detach_run_agent(ns, spec, policy, spawn=spawn)
1752 prompt, error = _read_prompt(ns.prompt_file)
1753 if error is not None:
1754 print(f"error: {error}", file=sys.stderr)
1755 return 2
1757 # The adapter name is also checked where it is used (#708). Unreachable today:
1758 # validation above refuses an unknown adapter (#903), and a built-in agent
1759 # names none. Kept so a future path that skips validation fails with the
1760 # config error, not a traceback.
1761 try:
1762 adapter = make_adapter(spec, mock=ns.mock)
1763 except ConfigError as exc: # pragma: no cover - see above
1764 print(redact(f"error: {exc}")[0], file=sys.stderr)
1765 return 2
1766 print(
1767 f"run-agent: {spec.name} ({spec.vendor}) role={policy.role} "
1768 f"{'write' if policy.write else 'read-only'}",
1769 file=sys.stderr,
1770 )
1771 started = time.time()
1772 record = _child_recorder(ns, spec, policy, started) if ns.child and ns.run_id else None
1773 if record is not None:
1774 # Claim the run with our OWN pid before doing any work. Every write to
1775 # this state file now comes from this process, in order, so nothing can
1776 # overwrite the terminal document the way the parent's post-spawn write
1777 # used to. Until this lands the run has no pid, which reads as
1778 # "liveness unknown" — reported as running, never as lost.
1779 record(runagent.initial_state(ns.run_id, spec, policy.role, started), terminal=False)
1780 try:
1781 if ns.cwd:
1782 with contextlib.chdir(ns.cwd):
1783 result = adapter.run(prompt, phase=policy.role, role_policy=policy)
1784 else:
1785 result = adapter.run(prompt, phase=policy.role, role_policy=policy)
1786 # The second (and last) place a seat is invoked, so it records the id it
1787 # sent exactly as the orchestrator's does (#709).
1788 result.model = adapter.resolved_model()
1789 document = runagent.result_dict(spec, policy.role, result)
1790 except BaseException as exc:
1791 # A detached child that dies without writing leaves its run at
1792 # "running" forever, and a `--wait` on it can only time out. Adapters
1793 # are fail-soft, so reaching here means something unexpected — a
1794 # KeyboardInterrupt, a bug — and the run must still end in a terminal
1795 # state that says so. Re-raised: this records the death, it does not
1796 # swallow it.
1797 if record is not None: 1797 ↛ 1799line 1797 didn't jump to line 1799 because the condition on line 1797 was always true
1798 record(_crash_document(spec, policy.role, started, exc))
1799 raise
1801 if record is not None:
1802 record(document)
1803 if ns.format == "text":
1804 print(document["text"])
1805 else:
1806 print(json.dumps(document, indent=2))
1807 return 0 if document["ok"] else 1
1810def _crash_document(spec, role: str, started: float, exc: BaseException) -> dict:
1811 """A result document for a child that died before producing one (#661)."""
1812 import time
1814 from . import runagent
1815 from .adapters import ERR_UNKNOWN, AgentResult
1817 reason = f"{type(exc).__name__}: {redact(str(exc))[0]}"
1818 return runagent.result_dict(
1819 spec,
1820 role,
1821 AgentResult(
1822 spec.name,
1823 spec.vendor,
1824 False,
1825 "",
1826 max(0.0, time.time() - started),
1827 f"run-agent exited before the agent finished: {reason}",
1828 error_code=ERR_UNKNOWN,
1829 ),
1830 )
1833def _child_recorder(ns, spec, policy, started: float):
1834 """A callable that writes a detached child's state file (issue #661).
1836 Called twice: once at the top with ``terminal=False`` to claim the run with
1837 this process's pid, and once at the end — from the success path or the
1838 crash handler — with the result document. Both writes come from the child,
1839 which is what keeps them ordered: the parent writes only before the spawn,
1840 so no stale copy can land on top of the finished document.
1842 Fail-soft: a child that cannot write its state must still print its
1843 document to the log, since the log is then the only record.
1844 """
1845 from . import runagent
1847 del spec, policy
1849 def record(document: dict, terminal: bool = True) -> None:
1850 state = dict(document)
1851 state.update(
1852 {
1853 "run_id": ns.run_id,
1854 "status": runagent.STATUS_DONE if terminal else runagent.STATUS_RUNNING,
1855 # `started_at` travels with the pid so a reader can see how old
1856 # the claim is — see runagent.pid_alive on why a pid alone is
1857 # not proof of identity.
1858 "started_at": started,
1859 "pid": os.getpid(),
1860 }
1861 )
1862 try:
1863 runagent.write_state(ns.run_id, state, ns.cache_dir)
1864 except (OSError, ValueError) as exc:
1865 print(f"warning: could not record run state: {redact(str(exc))[0]}", file=sys.stderr)
1867 return record
1870def _detach_run_agent(ns, spec, policy, spawn=None) -> int:
1871 """Start a ``run-agent`` call in the background and print its run id."""
1872 import time
1874 from . import runagent
1876 if ns.prompt_file == "-":
1877 print(
1878 "error: --detach needs a prompt FILE; the background run has no stdin to read from",
1879 file=sys.stderr,
1880 )
1881 return 2
1882 if not Path(ns.prompt_file).is_file():
1883 print(f"error: prompt file not found: {ns.prompt_file}", file=sys.stderr)
1884 return 2
1886 # An operator-supplied id was already validated at parse time, and a
1887 # generated one is valid by construction — `runagent.run_path` refuses an
1888 # unsafe id regardless, so there is nothing left to re-check here.
1889 run_id = ns.run_id or runagent.new_run_id()
1890 if runagent.state_path(run_id, ns.cache_dir).exists():
1891 print(
1892 f"error: run '{run_id}' already exists; choose another --run-id",
1893 file=sys.stderr,
1894 )
1895 return 2
1897 state = runagent.initial_state(run_id, spec, policy.role, time.time())
1898 try:
1899 # Reserve the id BEFORE spawning, so a duplicate is refused rather than
1900 # discovered by two children racing for the same state file.
1901 runagent.write_state(run_id, state, ns.cache_dir)
1902 out_path = runagent.output_path(run_id, ns.cache_dir)
1903 argv = _child_argv(ns, run_id, sys.executable)
1904 launch = spawn or _spawn_detached
1905 pid = launch(argv, out_path)
1906 except (OSError, ValueError) as exc:
1907 print(f"error: could not start the detached run: {redact(str(exc))[0]}", file=sys.stderr)
1908 return 2
1910 # The parent writes the state file EXACTLY ONCE, before the spawn. It used
1911 # to write a second time afterwards to record the pid, from the stale dict
1912 # it still held — and a child that finished during launch had its terminal
1913 # document overwritten with `status: running`, losing the answer entirely.
1914 # The child records its own pid instead (see `_child_running_state`), so
1915 # every write after this one comes from the child, in order.
1916 print(
1917 json.dumps(
1918 {
1919 # The pid is the parent's own knowledge, reported for the
1920 # operator's benefit; it is deliberately NOT written to the
1921 # state file from here.
1922 **runagent.run_summary(state),
1923 "pid": pid if isinstance(pid, int) else None,
1924 "state_file": str(runagent.state_path(run_id, ns.cache_dir)),
1925 "output_file": str(runagent.output_path(run_id, ns.cache_dir)),
1926 },
1927 indent=2,
1928 )
1929 )
1930 return 0
1933#: Deliberately-unreaped background children (see :func:`_spawn_detached`). The
1934#: parent is a short-lived CLI invocation, so this holds at most one entry per
1935#: `--detach` in a process that is about to exit.
1936_DETACHED_CHILDREN: list = []
1939def _spawn_detached(argv: list[str], out_path: Path) -> int:
1940 """Start the background child, streaming its stdout+stderr to ``out_path``.
1942 Its own session, so the child outlives the shell that launched it — the
1943 whole point of ``--detach`` for an orchestrator that dispatches work and
1944 comes back for it later. Returns the child's pid, which the caller records
1945 so ``--status`` can later tell a live run from an abandoned one.
1946 """
1947 import subprocess
1949 kwargs: dict = {}
1950 if hasattr(os, "setsid"): 1950 ↛ 1954line 1950 didn't jump to line 1954 because the condition on line 1950 was always true
1951 kwargs["start_new_session"] = True
1952 # 0600 like the state file: the log holds the agent's full output, which is
1953 # derived from the prompt and is no less sensitive than the diff a review sees.
1954 fd = os.open(out_path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600)
1955 with os.fdopen(fd, "wb") as out:
1956 proc = subprocess.Popen( # noqa: S603 - argv is built here, never shell-parsed
1957 argv,
1958 stdin=subprocess.DEVNULL,
1959 stdout=out,
1960 stderr=subprocess.STDOUT,
1961 **kwargs,
1962 )
1963 # Not waiting for this child is the whole point, but a dropped Popen warns
1964 # ("subprocess N is still running") when it is collected, and the parent
1965 # exits moments later so there is nothing to reap. Hold the reference
1966 # instead of lying about the return code or reaching into private state.
1967 _DETACHED_CHILDREN.append(proc)
1968 return proc.pid
1971def _run_replay(rest: list[str]) -> int:
1972 """Handle ``jury replay <outcome.json>`` (issue #449).
1974 Replays a saved run in the deliberation theater — or, off a TTY / without
1975 ``--theater``, as the same plain step stream ``--live`` prints. Pure
1976 presentation: no orchestration, no network, no agents.
1977 """
1978 from .replay import ReplayError, load_outcome, replay_events, replay_into
1980 sub = argparse.ArgumentParser(
1981 prog="jury replay",
1982 description="Replay a saved jury outcome (a result-cache entry or a "
1983 "serialized outcome dict) in the deliberation theater. No agents run.",
1984 )
1985 sub.add_argument(
1986 "outcome",
1987 help="path to a saved outcome JSON (cache entry or outcome dict)",
1988 )
1989 sub.add_argument(
1990 "--theater",
1991 action="store_true",
1992 help="replay in the animated deliberation scene (needs a wide TTY; "
1993 "falls back to plain transcript lines otherwise)",
1994 )
1995 sub.add_argument(
1996 "--theater-style",
1997 choices=["flat", "pixel"],
1998 default="flat",
1999 help="--theater scene style: 'flat' (ANSI line scene, default) or "
2000 "'pixel' (half-block pixel-art room)",
2001 )
2002 sub.add_argument(
2003 "--decision",
2004 choices=["chair", "vote"],
2005 default="chair",
2006 help="finale mode: 'chair' shows the stored synthesis verdict (default); "
2007 "'vote' re-tallies the panel ballots for the vote finale",
2008 )
2009 sub.add_argument(
2010 "--mode",
2011 choices=["code", "issue"],
2012 default="code",
2013 help="vote vocabulary for --decision vote (the serialized outcome does "
2014 "not record the run mode): 'code' (APPROVE/COMMENT/REQUEST CHANGES, "
2015 "default) or 'issue' (READY/UNCLEAR/NEEDS-INFO)",
2016 )
2017 ns = sub.parse_args(rest)
2019 try:
2020 outcome = load_outcome(Path(ns.outcome))
2021 except ReplayError as exc:
2022 print(f"error: {redact(str(exc))[0]}", file=sys.stderr)
2023 return 2
2025 # Panel-vote finale (mirrors the live path): re-tally from the stored
2026 # groups/reviews — deterministic, no agents involved.
2027 vote = None
2028 if ns.decision == "vote":
2029 from .voting import is_abstention, tally_votes
2031 voters = [
2032 r.agent for r in outcome.reviews if r.ok and not is_abstention(getattr(r, "output", ""))
2033 ]
2034 vote = tally_votes(outcome.groups, voters, mode=ns.mode)
2036 # Same TTY gate as the live path: the scene needs a wide TTY, otherwise
2037 # degrade to the plain --live step stream.
2038 court = None
2039 if ns.theater:
2040 from . import theater as _theater
2042 if _theater.supports_scene(sys.stdout):
2043 seats: dict[str, str] = {}
2044 for r in outcome.reviews:
2045 seats.setdefault(r.agent, r.vendor)
2046 court = _theater.Courtroom(
2047 list(seats.items()),
2048 outcome.chair or "chair",
2049 case=Path(ns.outcome).name,
2050 decision=ns.decision,
2051 style=ns.theater_style,
2052 )
2054 if court is not None:
2055 replay_into(court, outcome, vote=vote)
2056 else:
2057 for kind, result, round_no in replay_events(outcome):
2058 title, body = render_live_step(kind, result, round_no)
2059 print(f"## {title}\n\n{body}\n", flush=True)
2060 if vote is not None:
2061 # The vote finale must survive the transcript fallback too (review
2062 # finding: --decision vote was computed then silently dropped here).
2063 print("## Panel vote\n", flush=True)
2064 for ballot in vote.ballots:
2065 print(f"- {ballot.reviewer}: {ballot.vote} ({ballot.reason})", flush=True)
2066 print(f"\nVerdict: {vote.verdict}\n", flush=True)
2067 return 0
2070_PROGRESS_PREFIXES = (
2071 "round ",
2072 "reviewing chunk",
2073 "verification",
2074 "synthesis",
2075 "diff size",
2076 "early stop",
2077 "auto-depth",
2078)
2081def _is_progress_milestone(msg: str) -> bool:
2082 """Whether a log line is a coarse milestone worth a sticky-comment update."""
2083 return msg.startswith(_PROGRESS_PREFIXES)
2086#: Every CLI flag that writes a `[jury]` setting `validate_config` puts a bound
2087#: on, as `(flag, argparse dest, setting)` (issue #748). The settings are keys
2088#: of `config._NUMERIC_BOUNDS`, so the rule and the message are the config
2089#: path's own; only the `where` differs, naming the flag the operator typed
2090#: rather than a key they may never have written.
2091#:
2092#: The flags NOT here were checked against the same table and belong nowhere
2093#: else: `--seed` has no bound to break (a malformed `[jury] seed` is read as
2094#: "no seed" on purpose, not as an error); `--chunk`, `--early-stop`,
2095#: `--verify`, `--redact`, `--auto` and `--hints` are booleans; `--effort`,
2096#: `--decision`, `--context-mode` and `--format` are argparse `choices`, which
2097#: already refuse a value outside the vocabulary; and `--fail-on` has shared its
2098#: rule with the validator since #718.
2099_BOUNDED_FLAGS = (
2100 ("--rounds", "rounds", "rounds"),
2101 ("--max-rounds", "max_rounds", "max_rounds"),
2102 ("--total-timeout", "total_timeout", "total_timeout"),
2103 ("--phase-timeout", "phase_timeout", "phase_timeout"),
2104 ("--retries", "retries", "retries"),
2105 ("--max-diff-bytes", "max_diff_bytes", "diff.max_bytes"),
2106 # The fail-closed guards (docs audit 2026-09-29): they used to be clamped to
2107 # >= 0, so `--min-vendors -5` read as `0` — the documented opt-out — and the
2108 # cross-vendor guard was off with exit 0. `--no-min-vendors` writes 0, which
2109 # passes.
2110 ("--min-vendors", "min_vendors", "ci.min_vendors"),
2111 ("--min-reviews", "min_reviews", "ci.min_reviews"),
2112)
2115def _override_bound_error(args) -> str | None:
2116 """The first CLI override breaking a bound ``validate_config`` enforces.
2118 ``None`` is "the flag was not passed" and leaves the config value — which
2119 the validator has already checked — standing; it is never a value to range
2120 check, so an unset flag can never be mistaken for a zero.
2121 """
2122 for flag, dest, setting in _BOUNDED_FLAGS:
2123 value = getattr(args, dest, None)
2124 if value is None:
2125 continue
2126 message = bound_error(setting, value, where=flag)
2127 if message:
2128 return message
2129 return None
2132def _maybe_add_local_fallback(config, args, log):
2133 """Append a local agent when nothing else can run, offline (issue: zero-config).
2135 Only fires in the safe "fresh user" case: no explicit `--config`, no
2136 `./jury.toml`, not `--mock`, none of the configured agents are available,
2137 and a local OpenAI-compatible server is reachable with at least one model.
2138 Mutates ``config`` in place and points the chair at the local agent.
2140 Returns the seat it added, or ``None``. The built-in seats stay in the config,
2141 so the report still lists them as not found, but none of them can review: the
2142 panel this run can form is the local seat alone, and the caller scopes the
2143 default cross-vendor guard to that (#863). The decision is
2144 :func:`scaffold.zero_config_local_seat`, which ``jury --doctor`` also asks.
2145 """
2146 from .adapters import list_local_models, make_adapter
2147 from .scaffold import zero_config_local_seat
2149 seat = zero_config_local_seat(
2150 args.config,
2151 args.mock,
2152 Path("jury.toml").exists(),
2153 lambda: any(make_adapter(s).available() for s in config.enabled_agents),
2154 list_local_models,
2155 )
2156 if seat is None:
2157 return None
2158 model = seat.model
2159 config.agents.append(seat)
2160 config.chair = "local"
2161 named = getattr(args, "min_vendors", None)
2162 if named is None:
2163 guard = "so the default cross-vendor guard (min_vendors) does not apply to it"
2164 elif named <= 0:
2165 guard = "with the cross-vendor guard off (--no-min-vendors)"
2166 else:
2167 guard = f"held to the --min-vendors {named} you named"
2168 log(
2169 f"no agent CLIs found; using local model '{model}' (offline, $0) as a single-vendor panel, {guard}"
2170 )
2171 # An agy-only machine lands here too: say why its one CLI sat out.
2172 from .config import agy_opt_in_hint
2174 hint = agy_opt_in_hint((s.adapter_key for s in config.agents), shutil.which)
2175 if hint:
2176 log(f"note: {hint}")
2177 return seat
2180def _force_utf8_output() -> None:
2181 """Ensure stdout/stderr can emit the report's Unicode (emoji, arrows).
2183 On Windows the console defaults to a legacy code page (e.g. cp1252) that
2184 can't encode the report's `🏛️`/`⇄` characters, so `print(report)` raises
2185 `UnicodeEncodeError`. Reconfigure the real streams to UTF-8 when possible;
2186 `reconfigure` is absent on replaced streams (tests' StringIO, some pipes),
2187 so this is a best-effort no-op there.
2188 """
2189 for stream in (sys.stdout, sys.stderr):
2190 reconfigure = getattr(stream, "reconfigure", None)
2191 if reconfigure is not None:
2192 with contextlib.suppress(ValueError, OSError):
2193 reconfigure(encoding="utf-8")
2196_OVERVIEW = """\
2197🏛️ ai-jury — a cross-vendor multi-agent review jury.
2199It runs several coding-agent CLIs (Claude, Codex, Antigravity) plus an optional
2200local model over the same diff, PR, or issue; they cross-examine and verify each
2201other, and a chair (or a panel vote) synthesizes one verdict.
2203Common commands:
2204 jury init --wizard guided setup — writes a jury.toml (skippable)
2205 jury --pr 123 review a pull request
2206 jury --issue 42 review an issue for completeness
2207 git diff | jury --diff-file - review the current branch's diff
2208 jury examples more example commands
2209 jury guide a short end-to-end walkthrough
2210 jury --help every option
2212Docs: https://github.com/berkayturanci/ai-jury"""
2214_EXAMPLES = """\
2215ai-jury — example commands
2217Setup
2218 jury init --wizard guided setup (writes jury.toml)
2219 jury init --preset thorough non-interactive preset
2220 jury config show print the effective, resolved config
2221 jury --doctor check which agents/CLIs are available
2223Review
2224 jury --pr 123 review a pull request
2225 jury --issue 42 review an issue for completeness
2226 git diff | jury --diff-file - review the current branch's diff
2227 jury --diff-file changes.patch review a saved patch
2228 jury --pr 123 --verbose full play-by-play (rounds + transcript)
2230Decide & gate
2231 jury --pr 123 --decision vote verdict by panel vote (not a single chair)
2232 jury --pr 123 --ci exit non-zero on a blocking finding (CI gate)
2234Post results back to GitHub
2235 jury --pr 123 --post-summary post one rollup comment
2236 jury --pr 123 --post-inline post line-level review comments
2237 jury --issue 42 --post-summary post the triage verdict on the issue
2239Run `jury guide` for a walkthrough, or `jury --help` for every option."""
2241_GUIDE = """\
2242ai-jury — a short walkthrough
22441. Install the agent CLIs you have (any subset works): Claude Code, Codex,
2245 Antigravity. Optionally run a local model via Ollama for a free panelist.
2246 Check what's available:
2247 jury --doctor
22492. Create a config (picks reviewers, rounds, chair/vote, verify):
2250 jury init --wizard
2251 Every question is skippable — Enter keeps the built-in default.
22533. Run your first review:
2254 jury --pr 123 # a pull request
2255 jury --issue 42 # an issue's completeness
2256 git diff | jury --diff-file - # the current branch
2258 The panel reviews independently, cross-examines (debate), the chair verifies
2259 candidate findings to cut false positives, then synthesizes one verdict.
22614. Post the verdict back to GitHub (optional):
2262 jury --pr 123 --post-summary # one rollup comment
2263 jury --pr 123 --post-inline # line-level comments
22655. Gate CI on blocking findings (optional):
2266 jury --pr 123 --ci # non-zero exit on critical/major
2268Reviewers run sandboxed/read-only over attacker-controlled diffs by default.
2269See `jury examples` for more, or `jury --help` for every option.
2270Docs: https://github.com/berkayturanci/ai-jury"""
2273def main(argv: list[str] | None = None) -> int:
2274 _force_utf8_output()
2275 raw = list(sys.argv[1:] if argv is None else argv)
2277 # First-impression UX (#265): a newcomer running bare `jury` in a terminal
2278 # gets a friendly overview and exits 0 — not the argparse error. The strict
2279 # "provide one of --pr/--issue/--diff-file" error + non-zero exit is kept for
2280 # non-interactive use (piped/CI), so scripts that forget an input still fail.
2281 # `sys.stdin` can be None when stdin is detached (e.g. a background process),
2282 # so guard before calling isatty().
2283 if not raw and sys.stdin is not None and sys.stdin.isatty():
2284 print(_OVERVIEW)
2285 return 0
2287 # Plain-language command overview / walkthrough (#265), argv-intercepts like
2288 # the other subcommands so the main flag surface stays flat. Each has its own
2289 # small parser, so `--help` describes the subcommand the epilogue promises it
2290 # does (it printed the main help), and trailing junk (`jury examples foo`)
2291 # is an error rather than silently ignored.
2292 if raw[:1] in (["examples"], ["guide"]):
2293 argparse.ArgumentParser(
2294 prog=f"jury {raw[0]}",
2295 description=(
2296 "Print example commands." if raw[0] == "examples" else "Print a short walkthrough."
2297 ),
2298 ).parse_args(raw[1:])
2299 print(_EXAMPLES if raw[0] == "examples" else _GUIDE)
2300 return 0
2301 # Documented `jury cache clear` UX (issue #33): handled before argparse so
2302 # the rest of the CLI keeps its flat flag surface (no subcommands). Parsed
2303 # by its own parser BEFORE anything is deleted: `jury cache clear --help`
2304 # used to clear the cache, because every argument but `--cache-dir` was
2305 # ignored — the one command here that destroys data took a request for help
2306 # as a request to run.
2307 if raw[:2] == ["cache", "clear"]:
2308 from .cache import Cache
2310 sub = argparse.ArgumentParser(
2311 prog="jury cache clear",
2312 description="Delete every local review-cache entry and rotate the cache key.",
2313 )
2314 sub.add_argument(
2315 "--cache-dir", default=None, help="cache directory (default: the user cache dir)"
2316 )
2317 ns = sub.parse_args(raw[2:])
2318 removed = Cache(ns.cache_dir).clear()
2319 print(f"Cleared {removed} cache entr{'y' if removed == 1 else 'ies'}.")
2320 return 0
2322 # Comment-command mode (issue #11): `jury comment --text "/jury review"`
2323 # parses an allowlisted PR-comment command and dispatches a safe jury run.
2324 # Handled before the main parser so the comment text is never confused with
2325 # the jury's own flags, and never reaches a shell.
2326 if raw[:1] == ["comment"]:
2327 return _run_comment_command(raw[1:])
2329 # Config scaffolding (issue #107): `jury init` writes a jury.toml from
2330 # detected agents / flags / interactive prompts. Intercepted before the main
2331 # parser so it keeps its own small flag surface.
2332 if raw[:1] == ["init"]:
2333 return _run_init(raw[1:])
2335 # Config introspection: `jury config show` prints the EFFECTIVE resolved
2336 # config + its source so you can see exactly what will run; `config path`
2337 # prints just the source.
2338 if raw[:1] == ["config"]:
2339 return _run_config(raw[1:])
2341 # Apply verified suggested patches (issue #521): `jury apply` applies
2342 # suggested patches directly to the working directory.
2343 if raw[:1] == ["apply"]:
2344 return _run_apply(raw[1:])
2346 # Single-agent role dispatch (issue #661): `jury run-agent` runs ONE agent
2347 # for one role and prints a JSON result with attribution — the integration
2348 # point for an orchestrator. Intercepted like the other subcommands.
2349 if raw[:1] == ["run-agent"]:
2350 return _run_run_agent(raw[1:])
2352 # Theater replay (issue #449): `jury replay <outcome.json>` re-drives the
2353 # deliberation scene from a saved outcome — no agents, no network.
2354 # Intercepted before the main parser like the other subcommands.
2355 if raw[:1] == ["replay"]:
2356 return _run_replay(raw[1:])
2358 args = build_parser().parse_args(argv)
2360 # Checked before any short-circuiting branch below, so `--json` is rejected
2361 # wherever it is misplaced rather than only on the paths that reach the end.
2362 if args.json and not args.doctor:
2363 print(
2364 "error: --json applies to --doctor; use --format json for the review report.",
2365 file=sys.stderr,
2366 )
2367 return 2
2369 # Same vocabulary check `validate_config` applies to `[jury.ci] fail_on`,
2370 # reported with the same message (issue #718). Checked here, beside the
2371 # other flag guards, so a misspelled gate is refused before an agent is
2372 # paid for — and refused even without `--ci`, where the flag is inert:
2373 # otherwise the typo surfaces only on the run it was supposed to gate.
2374 if args.fail_on:
2375 message = fail_on_error(args.fail_on.split(","), "--fail-on")
2376 if message:
2377 print(f"error: {message}", file=sys.stderr)
2378 return 2
2380 if args.clear_cache:
2381 from .cache import Cache
2383 removed = Cache(args.cache_dir).clear()
2384 print(f"Cleared {removed} cache entr{'y' if removed == 1 else 'ies'}.")
2385 return 0
2387 if args.doctor:
2388 # `--doctor` resolves the cross-vendor threshold from `--min-vendors` too
2389 # (#863), so every bounded flag given here holds as it does on a run —
2390 # `--min-reviews` included, not only the one the prediction reads.
2391 message = _override_bound_error(args)
2392 if message:
2393 print(f"error: {message}", file=sys.stderr)
2394 return 2
2395 # Model discovery costs a probe per agent and only the JSON export
2396 # renders it; the human report must not pay for a field it never prints.
2397 diagnostics = doctor_module.build_diagnostics(
2398 args.config, probe_models=args.json, min_vendors=args.min_vendors
2399 )
2400 if args.json:
2401 # Exactly ONE JSON document on stdout, and nothing else: the export
2402 # is meant to be piped straight into `jq` / an orchestrator, so any
2403 # human chatter (including the --write confirmation below) goes to
2404 # stderr.
2405 print(json.dumps(doctor_module.doctor_report_dict(diagnostics), indent=2))
2406 else:
2407 print(doctor_module.render_report(diagnostics))
2408 if args.write:
2409 try:
2410 Path(args.write).write_text(
2411 json.dumps(diagnostics, indent=2) + "\n", encoding="utf-8"
2412 )
2413 except OSError as exc:
2414 print(f"error: {redact(str(exc))[0]}", file=sys.stderr)
2415 return 2
2416 stream = sys.stderr if args.json else sys.stdout
2417 print(f"\nWrote diagnostics to {args.write}", file=stream)
2418 return 0
2420 if args.config_validate:
2421 source = args.config or "jury.toml (or built-in defaults)"
2422 try:
2423 data = load_raw_config(args.config)
2424 warnings = validate_config(data, strict=args.strict_config)
2425 except (ConfigError, FileNotFoundError) as exc:
2426 print(redact(f"Config invalid ({source}): {exc}")[0], file=sys.stderr)
2427 return 2
2428 except OSError as exc:
2429 print(_config_read_error(exc, args.config), file=sys.stderr)
2430 return 2
2431 if warnings:
2432 print(f"Config valid with warnings ({source}):")
2433 for w in warnings:
2434 print(f" - {w}")
2435 else:
2436 print(f"Config valid ({source}).")
2437 return 0
2439 try:
2440 config = load_config(args.config, validate=True, strict=args.strict_config)
2441 except (ConfigError, FileNotFoundError) as exc:
2442 # A missing --config path raises FileNotFoundError; the other load sites already catch
2443 # both, so a bad path prints `Config invalid: …` instead of a traceback (#831).
2444 print(f"Config invalid: {redact(str(exc))[0]}", file=sys.stderr)
2445 return 2
2446 except OSError as exc:
2447 # It exists but cannot be read: an unreadable ./jury.toml, or --config naming
2448 # a directory. Neither is a FileNotFoundError, so both were a traceback (#893).
2449 print(_config_read_error(exc, args.config), file=sys.stderr)
2450 return 2
2451 # An auto-discovered ./jury.toml that runs local commands must be trusted before those
2452 # commands run — the checkout may be one this operator did not write (#831).
2453 try:
2454 configtrust.enforce(args.config, config, mock=args.mock)
2455 except configtrust.ConfigTrustError as exc:
2456 print(f"error: {exc}", file=sys.stderr)
2457 return 2
2458 # One value, one field, one answer, whichever surface it was written on
2459 # (issue #748). The overrides below are assigned straight onto the config
2460 # AFTER `validate_config` has run, so until now nothing range-checked them:
2461 # `--rounds 0` was accepted, ran a full Round 1 and exited 0 while `rounds =
2462 # 0` in `jury.toml` was a hard error, on a flag `docs/parameters.md`
2463 # documents as `≥ 1`. An operator who wrote `--rounds 0` meaning "no
2464 # debate" got a complete review round, a report, and nothing anywhere
2465 # saying the number they passed had been ignored.
2466 #
2467 # Refused here — before the diff is read and long before an agent is paid
2468 # for — with exit 2, like every other bad-input exit on this path.
2469 override_error = _override_bound_error(args)
2470 if override_error:
2471 print(f"error: {override_error}", file=sys.stderr)
2472 return 2
2473 if args.rounds is not None:
2474 config.rounds = args.rounds
2475 # A fixed --rounds is a hard override: it disables adaptive early-stop so
2476 # the run is reproducible fixed-N (issue #40), unless --early-stop is also
2477 # passed explicitly (handled below).
2478 config.early_stop = False
2479 if args.max_rounds is not None:
2480 config.max_rounds = args.max_rounds
2481 if args.early_stop is not None:
2482 config.early_stop = args.early_stop
2483 if args.total_timeout is not None:
2484 config.total_timeout = args.total_timeout
2485 if args.phase_timeout is not None:
2486 config.phase_timeout = args.phase_timeout
2487 if args.retries is not None:
2488 # Assigned as written: the `max(0, …)` clamp that used to stand here was
2489 # the second rule for one value (#748). It silently turned `--retries
2490 # -1` into 0 while `retries = -1` in `jury.toml` was a hard error, and
2491 # is unreachable now that the same bound refuses the flag above.
2492 config.retries = args.retries
2493 if args.seed is not None:
2494 config.seed = args.seed
2495 if args.chair:
2496 # Refused, not silently replaced (docs audit 2026-09-29): an unknown
2497 # `--chair` used to exit 0 with the first usable agent chairing, so the
2498 # operator's choice of synthesizer was ignored without a word. Checked
2499 # against the enabled seats, before anything is fetched or spent.
2500 enabled = [a.name for a in config.enabled_agents]
2501 if args.chair != "rotate" and args.chair not in enabled:
2502 print(
2503 f"error: --chair {args.chair!r} is not an enabled agent (enabled: "
2504 f"{', '.join(enabled) or 'none'}); name one of them, or 'rotate'.",
2505 file=sys.stderr,
2506 )
2507 return 2
2508 config.chair = args.chair
2509 # Applied to the config (rather than resolved late like --min-vendors)
2510 # because the pre-run half of this gate lives in the orchestrator: it has to
2511 # be able to refuse a bench that is too small BEFORE the panel is paid for.
2512 if args.min_reviews is not None:
2513 # Assigned as written: the bound above refuses a negative (docs audit
2514 # 2026-09-29), where a `max(0, …)` clamp here used to turn it into "off".
2515 config.ci.min_reviews = args.min_reviews
2516 # --effort is a whole-panel override: it wins over every [[agent]] effort so
2517 # one flag raises (or lowers) the depth of the entire run.
2518 if args.effort is not None:
2519 for agent in config.agents:
2520 agent.effort = args.effort
2521 # Warn ONCE per run per distinct message, before any agent is invoked, for
2522 # every panelist whose vendor cannot act on the requested effort. Real
2523 # adapters are passed only outside --mock, so an offline demo never spawns a
2524 # vendor CLI to check a model listing.
2525 for warning in effort_warnings(
2526 config.enabled_agents, adapter_factory=None if args.mock else make_adapter
2527 ):
2528 print(f"warning: {warning}", file=sys.stderr)
2529 if args.verify is not None:
2530 config.verify = args.verify
2531 if args.context_mode is not None:
2532 config.context.mode = args.context_mode
2533 if args.redact is not None:
2534 config.context.redact_secrets = args.redact
2535 if args.max_diff_bytes is not None:
2536 config.diff.max_bytes = args.max_diff_bytes
2537 if args.chunk is not None:
2538 config.diff.chunk = args.chunk
2539 if args.exclude:
2540 config.diff.exclude = list(config.diff.exclude) + list(args.exclude)
2541 if args.include:
2542 config.diff.include = list(config.diff.include) + list(args.include)
2544 try:
2545 policy = load_policy(args.policy)
2546 except PolicyError as exc:
2547 print(f"error: {redact(str(exc))[0]}", file=sys.stderr)
2548 return 2
2550 # Issue mode (issue #221) reviews prose, not a diff, so the PR/diff-only
2551 # concepts below have no meaning. Reject them up front with a clear message
2552 # rather than silently ignoring them.
2553 # Exactly one source (issue #367). Listed rather than pairwise so adding a
2554 # source cannot quietly skip the check.
2555 _sources = [
2556 ("--pr", args.pr),
2557 ("--issue", args.issue),
2558 ("--diff-file", args.diff_file),
2559 ("--commit", getattr(args, "commit", None)),
2560 ("--commits", getattr(args, "commits", None)),
2561 ]
2562 _given = [flag for flag, value in _sources if value]
2563 if len(_given) > 1:
2564 raise SystemExit(f"error: choose one input source, got {', '.join(_given)}")
2565 if args.issue:
2566 for flag, on in (
2567 ("--post-inline", args.post_inline),
2568 ("--post-progress", args.post_progress),
2569 ("--label", args.label),
2570 ("--incremental", args.incremental),
2571 # The issue path posts one summary and never reads the mode (docs
2572 # audit 2026-09-29): accepted, `phased` was silently ignored.
2573 ("--post-mode", args.post_mode is not None),
2574 ):
2575 if on:
2576 raise SystemExit(
2577 f"error: {flag} is not supported with --issue (it is a PR/diff concept)"
2578 )
2580 # A posting flag with nowhere to post is a usage error, so it is refused here with
2581 # the rest of the argument checks. It used to be refused after the review, when
2582 # every seat had already run and been paid for (#866). `--post-summary` also posts
2583 # to an issue; the others are PR-only and `--issue` rejected them above.
2584 for flag, on, has_target in (
2585 ("--post-summary", args.post_summary, args.pr or args.issue),
2586 ("--post-inline", args.post_inline, args.pr),
2587 ("--label", args.label, args.pr),
2588 ):
2589 if on and not has_target:
2590 raise SystemExit(f"error: {flag} requires --pr")
2591 # `--post-mode` shapes the summary comment, so without one it shapes nothing
2592 # (docs audit 2026-09-29): `--post-mode phased` alone exited 0 and posted
2593 # nothing, which docs/parameters.md has always said it requires.
2594 if args.post_mode is not None and not args.post_summary:
2595 raise SystemExit("error: --post-mode requires --post-summary (or --post)")
2597 # Live progress on the PR (issue #125): a single sticky comment updated at
2598 # each round/chunk milestone. Opt-in and requires --pr.
2599 progress = None
2600 if args.post_progress:
2601 if not args.pr:
2602 raise SystemExit("error: --post-progress requires --pr")
2603 from .github import ProgressReporter
2605 progress = ProgressReporter(args.pr, args.repo)
2607 def log(msg: str) -> None:
2608 if not args.quiet:
2609 print(f"[jury] {msg}", file=sys.stderr)
2610 if progress is not None and _is_progress_milestone(msg):
2611 progress.update(msg)
2613 # Smart offline fallback: with NO config file and NO usable agent CLI, but a
2614 # local model server reachable, add a local agent so `jury` just works
2615 # offline out of the box (issue: easier zero-config). Never overrides an
2616 # explicit config or a working CLI panel.
2617 local_fallback = _maybe_add_local_fallback(config, args, log)
2619 try:
2620 diff, context = _read_diff(args)
2621 except RuntimeError as exc:
2622 # `--pr` shells out to `gh`; a missing or failing `gh` raises RuntimeError, which was
2623 # otherwise uncaught here and printed a Python traceback (#831).
2624 print(f"error: {redact(str(exc))[0]}", file=sys.stderr)
2625 return 2
2627 # Incremental review (issue #9): when --incremental and a prior jury
2628 # marker exists, narrow the diff to the range since the last reviewed SHA;
2629 # otherwise fall back safely to the full diff. The reviewed head SHA is also
2630 # recorded on the posted summary so a later run can go incremental.
2631 review_scope = None
2632 head_sha = ""
2633 if args.incremental:
2634 if not args.pr:
2635 raise SystemExit("error: --incremental requires --pr")
2636 from . import incremental as inc
2637 from .github import compare_diff, pr_comment_bodies, pr_head_sha
2639 head_sha = pr_head_sha(args.pr, args.repo)
2640 prev_sha = inc.parse_reviewed_sha(pr_comment_bodies(args.pr, args.repo))
2641 mode, reason = inc.decide_review(prev_sha, head_sha)
2642 if mode == inc.MODE_INCREMENTAL:
2643 inc_diff = compare_diff(prev_sha, head_sha, args.repo)
2644 if inc_diff.strip():
2645 diff = inc_diff
2646 else:
2647 mode, reason = inc.MODE_FULL, "incremental range unavailable — full review"
2648 review_scope = inc.scope_note(mode, reason)
2649 log(reason)
2651 if not diff.strip():
2652 raise SystemExit("error: empty diff — nothing to review")
2654 # Risk-aware auto-depth (issue #120): scale rounds/verify to the diff when
2655 # enabled. Explicit --rounds/--verify/--early-stop always win; the panel is
2656 # never trimmed. Off unless --auto or [jury] auto_depth.
2657 if args.auto if args.auto is not None else config.auto_depth:
2658 from .diffprofile import depth_for, describe, profile_diff
2660 prof = profile_diff(diff)
2661 rounds, verify, early_stop = depth_for(prof.risk)
2662 if args.rounds is None:
2663 config.rounds = rounds
2664 if args.early_stop is None:
2665 config.early_stop = early_stop
2666 if args.verify is None:
2667 config.verify = verify
2668 log(describe(prof))
2670 if getattr(args, "tiered", False):
2671 config.routing = "tiered"
2672 if config.routing == "tiered" and getattr(args, "min_vendors", None) is not None:
2673 # The routed panel keeps at least the vendor floor the gate will apply
2674 # (#714); a `--min-vendors` override has to reach the plan, not only the
2675 # gate that runs after the panel has been paid for.
2676 config.ci.min_vendors = int(args.min_vendors)
2677 # ``--hints`` / ``--no-hints`` override ``[jury] hints`` in BOTH directions;
2678 # the sentinel (None) means "not passed", so the config value stands (#715).
2679 hints_override = getattr(args, "hints", None)
2680 if hints_override is not None:
2681 config.hints = hints_override
2683 # The pre-pass block is carried SEPARATELY from the user context (#715).
2684 # Appending it to ``context`` here made it invisible under the default
2685 # context mode: ``run_jury`` clears ``context`` when the mode is "diff-only",
2686 # so the linters ran, this line was logged, and the panel saw nothing. The
2687 # orchestrator joins the block into the Round 1 prompt after that filter.
2688 #
2689 # The linters see the CHANGED files and nothing else (#737). The paths come
2690 # from the same plan ``review_diff`` builds below — ``plan_for`` is pure, so
2691 # the ``kept`` list here is the one the panel is shown — which means a file
2692 # dropped by ``[jury.diff] include/exclude`` never contributes a hint about a
2693 # diff the reviewers cannot read. ``--issue`` reviews prose and has no
2694 # changed paths at all, so it gets no block. When the change touches nothing
2695 # Ruff or ESLint handles, ``collect_static_hints`` returns "" rather than
2696 # falling back to the working tree.
2697 hints_block = ""
2698 if config.hints:
2699 from .hints import collect_static_hints
2700 from .orchestrator import plan_for
2702 changed_paths = [] if args.issue else [p for p in plan_for(config, diff).kept_paths if p]
2703 sh = collect_static_hints(changed_paths)
2704 if sh:
2705 hints_block = sh
2706 log(f"injected static analysis hints for {len(changed_paths)} changed file(s)")
2708 # Optional local result cache (issue #33): a hit skips the run entirely; a
2709 # miss runs the jury and stores the outcome. The key covers the diff,
2710 # effective config, prompt version, package version, context policy, the
2711 # context text that policy admits (#738), the static-analysis block the
2712 # linters produced (#745), and seed. ``context`` is passed straight through from
2713 # ``_read_diff`` and ``hints_block`` is the string resolved just above — both
2714 # are the ones ``run_jury`` gets, so the key is a function of what the panel
2715 # is shown and not of a second reading of the config.
2716 cache = None
2717 cache_k = None
2718 outcome = None
2719 if args.cache:
2720 from .cache import Cache, cache_key
2722 cache = Cache(args.cache_dir)
2723 cache_k = cache_key(
2724 config,
2725 diff,
2726 context=context,
2727 hints=hints_block,
2728 mock=args.mock,
2729 policy=policy,
2730 mode=("issue" if args.issue else "code"),
2731 )
2732 outcome = cache.load(cache_k)
2733 if outcome is not None:
2734 log(f"cache hit ({cache_k[:12]}…) — reusing stored outcome")
2735 else:
2736 log(f"cache miss ({cache_k[:12]}…) — running jury")
2738 # Live play-by-play (issue #210, #229): stream each step as it happens. Prints
2739 # a titled block to stdout the moment a phase result lands. Posting each step to
2740 # the PR/issue is OPT-IN — it requires BOTH a target (--pr or --issue) AND
2741 # --post (a bare target only selects the source, never auto-posts), so `--live`
2742 # alone just streams locally. Posting is best-effort: a GitHub hiccup is logged
2743 # and never aborts the run.
2744 live_target = args.pr or args.issue
2745 # Theater defaults can come from jury.toml (issue #364); the CLI flags
2746 # (--theater / --no-theater, --theater-style) override per run. Sentinels
2747 # (None) distinguish "not passed" from an explicit choice.
2748 theater_on = args.theater if args.theater is not None else config.theater
2749 theater_style = args.theater_style or config.theater_style
2750 live_posts = bool((args.live or theater_on) and args.post_summary and live_target)
2751 live_post = post_issue_comment if args.issue else post_pr_comment
2752 # Opt-in animated "courtroom" scene (--theater): an interactive TTY view of
2753 # the REAL run (each model seated, speaking per phase, gavel/vote finale). It
2754 # needs a wide TTY and an actual run (a cache hit has nothing to replay), so
2755 # it falls back to the plain --live step stream otherwise. The structured
2756 # outcome / report / CI gate are untouched — this is a side channel.
2757 court = None
2758 if theater_on and outcome is None and not args.quiet:
2759 from . import theater as _theater
2761 if _theater.supports_scene(sys.stdout): 2761 ↛ 2791line 2761 didn't jump to line 2791 because the condition on line 2761 was always true
2762 # Display-only chair label for the scene title. The run resolves the
2763 # REAL chair internally (resolve_chair needs the usable/reviewer sets
2764 # and run RNG, which don't exist yet here), so use a best-effort name.
2765 chair_name = (
2766 config.chair
2767 if config.chair and config.chair != "rotate"
2768 else (config.agents[0].name if config.agents else "chair")
2769 )
2770 case = (
2771 f"PR #{args.pr}"
2772 if args.pr
2773 else f"issue #{args.issue}"
2774 if args.issue
2775 else f"commit {args.commit}"
2776 if getattr(args, "commit", None)
2777 else f"range {args.commits}"
2778 if getattr(args, "commits", None)
2779 else "local diff"
2780 )
2781 court = _theater.Courtroom(
2782 [(a.name, a.vendor) for a in config.agents],
2783 chair_name,
2784 case=case,
2785 mode=("issue" if args.issue else "code"),
2786 decision=(args.decision or config.decision),
2787 style=theater_style,
2788 )
2789 court.open()
2791 on_event = None
2792 if args.live or theater_on:
2794 def on_event(kind, result, round_no=None):
2795 if court is not None:
2796 court.step(kind, result, round_no)
2797 else:
2798 # plain step stream (--live, or --theater fallback off a TTY)
2799 title, body = render_live_step(kind, result, round_no)
2800 print(f"## {title}\n\n{body}\n", flush=True)
2801 if live_posts:
2802 try:
2803 title, body = render_live_step(kind, result, round_no)
2804 live_post(live_target, f"## {title}\n\n{body}", args.repo)
2805 except Exception as exc: # noqa: BLE001 - best-effort, never crash
2806 log(f"live: failed to post step to #{live_target}: {redact(str(exc))[0]}")
2808 # We stream live only when actually running the jury; a cache hit has nothing
2809 # to replay, so the consolidated report is still printed in that case.
2810 live_streamed = bool(args.live or theater_on) and outcome is None
2812 if outcome is None:
2813 try:
2814 if args.issue:
2815 # Issue prose bypasses large-diff planning (filter/size/chunk is
2816 # meaningless for an issue body); run the jury directly with the
2817 # issue-quality rubric.
2818 outcome = run_jury(
2819 config,
2820 diff,
2821 context=context,
2822 hints=hints_block,
2823 mock=args.mock,
2824 strict=args.strict,
2825 policy=policy,
2826 log=log,
2827 on_event=on_event,
2828 mode="issue",
2829 )
2830 else:
2831 outcome, _plan = review_diff(
2832 config,
2833 diff,
2834 context=context,
2835 hints=hints_block,
2836 mock=args.mock,
2837 strict=args.strict,
2838 policy=policy,
2839 log=log,
2840 on_event=on_event,
2841 )
2842 except KeyboardInterrupt:
2843 # Graceful cancellation (issue #30): a jury run can be long, so
2844 # Ctrl-C should exit cleanly with the conventional 130 rather than
2845 # dumping a traceback. Work already completed is not partially
2846 # rendered here because the orchestrator returns atomically; we just
2847 # report the cancellation.
2848 print("\n[jury] cancelled (interrupted) — no report produced", file=sys.stderr)
2849 return 130
2850 except RuntimeError as exc:
2851 # Large-diff "too large / nothing to review" (issue #31) and "no
2852 # usable agents" are actionable user errors, not crashes.
2853 print(f"error: {redact(str(exc))[0]}", file=sys.stderr)
2854 return 2
2855 if cache is not None and cache_k is not None:
2856 cache.store(cache_k, outcome)
2857 log(f"cached outcome ({cache_k[:12]}…)")
2859 # Final-verdict mode (issue #220): a panel vote (tally the reviewers) vs the
2860 # chair's synthesis. Rendering-only — the outcome is identical; the severity-
2861 # based CI gate below is unaffected. Effective = CLI flag else config.
2862 decision = args.decision or config.decision
2863 vote = None
2864 if decision == "vote":
2865 from .voting import is_abstention, tally_votes
2867 # A reviewer that abstained (empty reply or a refusal) is excluded from
2868 # the tally — a non-answer must not count as a "clear" vote (issue #251).
2869 voters = [
2870 r.agent for r in outcome.reviews if r.ok and not is_abstention(getattr(r, "output", ""))
2871 ]
2872 vote = tally_votes(
2873 outcome.groups,
2874 voters,
2875 mode=("issue" if args.issue else "code"),
2876 )
2878 # Close the courtroom scene (after the vote is tallied, so the panel-vote
2879 # finale can show the ballots/verdict).
2880 if court is not None:
2881 if vote is not None:
2882 court.set_vote(vote)
2883 court.close()
2885 # Ballot vocabulary follows the review mode, exactly as the vote tally does:
2886 # an issue review votes on completeness (READY/UNCLEAR/NEEDS_INFO), not on a
2887 # diff's correctness. Resolved BEFORE the metadata, which now derives the
2888 # ballots to count them (#700, round 2) and must derive the same ones the
2889 # report renders.
2890 ballot_mode = "issue" if args.issue else "code"
2892 # Whether this run's panel is the zero-config fallback's one local seat (#863):
2893 # recorded in the metadata and stated in the report, not only on stderr, which
2894 # `--quiet` silences.
2895 zero_config = local_fallback is not None
2896 metadata = build_run_metadata(
2897 outcome,
2898 config,
2899 decision=decision,
2900 vote=vote,
2901 mode=ballot_mode,
2902 zero_config_fallback=zero_config,
2903 )
2905 # The attribution footer (issue #911): on unless `--no-attribution` or
2906 # `[jury.output] attribution = false`. Appended below, once the markdown
2907 # report is complete; `transcript_mode` picks the transcript's wording.
2908 attribution_on = args.attribution if args.attribution is not None else config.output.attribution
2909 transcript_mode = False
2911 if args.format == "json":
2912 from .formats import to_json
2914 report = to_json(
2915 outcome,
2916 config,
2917 decision=decision,
2918 vote=vote,
2919 mode=ballot_mode,
2920 zero_config_fallback=zero_config,
2921 )
2922 elif args.format == "sarif":
2923 from .formats import to_sarif
2925 report = to_sarif(outcome, config)
2926 elif args.format == "keel-reviews":
2927 from .formats import to_keel_reviews
2929 report = to_keel_reviews(outcome, config, vote=vote, mode=ballot_mode)
2930 else:
2931 # Output mode (issue: full transcript). --verbose => summary + transcript;
2932 # --transcript (or [jury] transcript, unless --no-transcript) => the
2933 # chronological play-by-play; otherwise the consensus-first summary.
2934 # Rendering-only — the orchestration/outcome is identical either way.
2935 transcript_default = args.transcript if args.transcript is not None else config.transcript
2936 transcript_mode = bool(args.verbose or transcript_default)
2937 if transcript_mode:
2938 report = render_transcript(
2939 outcome.reviews,
2940 outcome.debate,
2941 outcome.synthesis,
2942 chair=outcome.chair,
2943 findings=outcome.findings,
2944 warnings=outcome.warnings,
2945 groups=outcome.groups,
2946 verify=outcome.verify,
2947 context_mode=outcome.context_mode,
2948 redact_secrets=outcome.redact_secrets,
2949 redaction_count=outcome.redaction_count,
2950 metadata=metadata,
2951 review_scope=review_scope,
2952 lead_with_summary=bool(args.verbose),
2953 vote=vote,
2954 footer=False,
2955 )
2956 else:
2957 report = render(
2958 outcome.reviews,
2959 outcome.debate,
2960 outcome.synthesis,
2961 chair=outcome.chair,
2962 findings=outcome.findings,
2963 warnings=outcome.warnings,
2964 groups=outcome.groups,
2965 verify=outcome.verify,
2966 context_mode=outcome.context_mode,
2967 redact_secrets=outcome.redact_secrets,
2968 redaction_count=outcome.redaction_count,
2969 metadata=metadata,
2970 review_scope=review_scope,
2971 vote=vote,
2972 footer=False,
2973 )
2975 if args.metadata_json:
2976 with Path(args.metadata_json).open("w", encoding="utf-8") as fh:
2977 fh.write(json.dumps(metadata, indent=2) + "\n")
2978 log(f"metadata written to {redact(args.metadata_json)[0]}")
2980 ci_exit = 0
2981 # A run whose panel collapsed is a different thing wearing the same output
2982 # (#625/#682). `--strict` fails when a configured CLI is *missing*; this
2983 # fails when one was present, probed fine, and returned nothing — which is
2984 # how a three-vendor panel silently becomes one. It FAILS CLOSED by default
2985 # (`[jury.ci] min_vendors`, shipped as 2) and is scoped to runs that claimed
2986 # cross-vendor consensus, so a single-vendor install is untouched; exit 3, so
2987 # it is distinguishable from a findings failure.
2988 #
2989 # The zero-config local fallback (#863) seats one local reviewer beside the
2990 # built-in seats it found missing — it fires only when none of them can run —
2991 # so the panel it forms is that seat alone and never claimed cross-vendor
2992 # consensus. Counting the missing seats as configured vendors failed the
2993 # documented `jury --diff-file -` offline path with exit 3 on every run. A
2994 # threshold named on the command line is still enforced as asked.
2995 required, explicit = resolve_min_vendors(getattr(args, "min_vendors", None), config)
2996 collapsed = collapse_reason(
2997 outcome.reviews,
2998 required,
2999 None if explicit else claimed_vendors(config.enabled_agents, local_fallback),
3000 )
3001 if collapsed:
3002 log(collapsed)
3003 # A runner with claude + agy and no codex lands here since agy left the
3004 # default panel: name the second vendor it already has, and why it sat out.
3005 from .config import agy_opt_in_hint
3007 hint = agy_opt_in_hint((s.adapter_key for s in config.agents), shutil.which)
3008 if hint:
3009 log(f"note: {hint}; or install another vendor's CLI, or lower --min-vendors")
3010 ci_exit = 3
3011 # No seat returned anything (#849). One local seat pointed at a model its
3012 # server does not have failed every call with `HTTP 404`, and the run still
3013 # exited 0 with a report of nothing. The collapse guard above is scoped to
3014 # runs that claimed cross-vendor consensus and `min_reviews` is off by default,
3015 # so neither saw it. This is narrower than both on purpose: not "too few
3016 # reviews" — a seat that answered "looks good" is an abstention, and a clean
3017 # single-seat run must stay green — but "every seat failed to return a result
3018 # at all". A run that reviewed nothing is not a pass.
3019 ran = metadata.get("panel") or {}
3020 configured_seats = int(ran.get("configured") or 0)
3021 if configured_seats and int(ran.get("failed") or 0) == configured_seats:
3022 log(
3023 f"no reviewer returned a result: all {configured_seats} seat(s) failed, so "
3024 f"nothing was reviewed. A run with no review is not a pass — see each "
3025 f"seat's error above, and `jury --doctor` (e.g. a local model not pulled)."
3026 )
3027 ci_exit = 3
3028 # The other half of the #699 gate. The orchestrator refuses a bench that is
3029 # too small before spending it; this catches the two cases a pre-flight
3030 # cannot predict — an agent that was present, ran, and returned nothing, and
3031 # one that answered with prose naming nothing a reader could check. Both are
3032 # recorded as abstentions and neither is a review, so the number the consumer
3033 # will accept can fall below its minimum while every seat "answered" (#700,
3034 # round 2: `--min-reviews 3` used to be satisfied by three "Looks good to me"
3035 # replies). Exit 3, the same family as a collapsed panel: the panel, not the
3036 # findings, is what fell short. Evaluated on cached outcomes too, which is
3037 # why `min_reviews` is kept out of the cache key.
3038 panel_meta = metadata.get("panel") or {}
3039 short_panel = panel.shortfall(
3040 panel_meta.get("reviews_supplied") or 0,
3041 config.ci.min_reviews,
3042 stage="after the panel ran",
3043 # Every cause the metadata publishes, passed by the one mapping that
3044 # names them: a call site listing two of the four printed the other two
3045 # under a cause their own ballots contradicted (#700, round 5).
3046 **{key: panel_meta.get(key) or 0 for key in panel.PANEL_METADATA_KEYS.values()},
3047 )
3048 if short_panel:
3049 log(short_panel)
3050 ci_exit = 3
3051 if args.ci:
3052 fail_on = config.ci.fail_on
3053 if args.fail_on:
3054 fail_on = [s.strip().lower() for s in args.fail_on.split(",") if s.strip()]
3055 gate_exit, ci_reason = evaluate_ci(outcome.groups, fail_on, config.ci.ignore_unverified)
3056 # A collapsed panel outranks the severity gate: `evaluate_ci` reports on
3057 # findings the panel did or did not raise, and a panel that never formed
3058 # is not evidence either way. Before #682 this assignment overwrote the
3059 # collapse exit, so `--ci --min-vendors 2` reported a clean pass on a
3060 # single-vendor run — the exact failure the flag was added for.
3061 ci_exit = ci_exit or gate_exit
3062 # Only the markdown report carries the human-readable CI gate section;
3063 # json/sarif documents stay machine-clean. The exit code is unchanged.
3064 if args.format == "markdown":
3065 report += f"\n\n## CI gate\n\n{ci_reason}\n"
3067 # Suggested patches (issue #10): opt-in and kept separate from the default
3068 # report. Written to a file with --patches-out, else appended after the
3069 # markdown report under its own heading. The default flow stays read-only.
3070 if args.suggest_patches:
3071 from .patches import render_patch_suggestions
3073 patches_section = render_patch_suggestions(outcome.groups)
3074 if not patches_section:
3075 log("no verified findings with a suggested fix — no patches emitted")
3076 elif args.patches_out:
3077 Path(args.patches_out).write_text(patches_section, encoding="utf-8")
3078 log(f"suggested patches written to {redact(args.patches_out)[0]}")
3079 elif args.format == "markdown":
3080 report += "\n\n" + patches_section.rstrip()
3081 else:
3082 log("--suggest-patches needs markdown output or --patches-out; skipped")
3084 # The attribution footer (issue #911) closes the markdown report: the tool and
3085 # the seats that returned a review. The renderers were asked to leave it off so
3086 # it can go here, AFTER the CI gate and patch sections appended above, and the
3087 # report ends with exactly one footer. JSON/SARIF/keel-reviews are machine
3088 # documents and never get it. Posting appends the hidden SHA marker after it.
3089 if attribution_on and args.format == "markdown":
3090 report += "\n" + render_footer(outcome.reviews, transcript=transcript_mode)
3092 # Turn the live progress comment into the final verdict (issue #125).
3093 if progress is not None:
3094 progress.finish(report)
3095 log(f"progress comment finalized on PR #{args.pr}")
3097 if args.output:
3098 # By the time this write runs the panel has been invoked, the debate is
3099 # over and the verdict is rendered: the run is paid for. An unwritable
3100 # path used to come out of `open()` as a raw `FileNotFoundError`
3101 # traceback (issue #749) — an implementation detail where a sentence
3102 # naming the path belongs, and one that took the report with it, because
3103 # `-o` is exactly what suppresses the stdout copy below.
3104 #
3105 # So it is reported the way the `--doctor --write` failure a few hundred
3106 # lines up is (named path, redacted reason, exit 2), and the report falls
3107 # back to STDOUT rather than to another file. A fallback file would put
3108 # the run somewhere the operator never named — a second surprise on top
3109 # of the first, able to clobber, and no more likely to succeed when the
3110 # cause is a full or read-only filesystem — while stdout is where this
3111 # exact document would have gone had `-o` not been passed at all. It is
3112 # redirectable and greppable, so the run survives; exit 2 keeps a caller
3113 # from reading the delivery as a success.
3114 try:
3115 with Path(args.output).open("w", encoding="utf-8") as fh:
3116 fh.write(report + "\n")
3117 except OSError as exc:
3118 print(
3119 f"error: could not write the report to "
3120 f"'{redact(args.output)[0]}': {redact(str(exc))[0]}",
3121 file=sys.stderr,
3122 )
3123 print(
3124 "error: the review is complete and is printed on stdout instead; "
3125 "redirect it to keep the report.",
3126 file=sys.stderr,
3127 )
3128 print(report)
3129 return 2
3130 log(f"report written to {redact(args.output)[0]}")
3131 elif not (live_streamed and args.format == "markdown"):
3132 # In --live markdown mode the step stream WAS the stdout output; don't also
3133 # dump the consolidated report (it would duplicate everything just shown).
3134 # For json/sarif the stream is human-readable markdown, so the requested
3135 # machine-readable document must still go to stdout.
3136 print(report)
3138 # The read side of `gh` has been guarded since #836; the write side was not (#844).
3139 # By the time control reaches here the panel has run and the verdict is already on
3140 # stdout, so a `gh` failure — missing CLI, bad token, no write permission, a deleted
3141 # PR, a timeout, a refused spawn — must not replace a finished review with a traceback
3142 # and exit 1. The two kinds are deliberately different:
3143 #
3144 # * `--post-summary` is a contract. Asking for the verdict to be posted and not
3145 # posting it is a failure of the run, so it exits 2 like every other user error.
3146 # * `--post-inline` and `--label` are additions to a review that has already been
3147 # delivered. They report the failure and leave `ci_exit` — the gate's own answer
3148 # about the code — intact, because losing it would turn "the code is fine, GitHub
3149 # hiccuped" into "the gate failed".
3150 def _post(action: str, call, *, contractual: bool) -> bool:
3151 """Run *call*, reporting a `gh` failure instead of letting it escape.
3153 Returns **whether the post landed**, so no caller can log a success that
3154 did not happen. An earlier version returned `2 if contractual else None`,
3155 which made failure and success indistinguishable for the non-contractual
3156 callers — both were `None` — and `--post-inline` printed the error and the
3157 "posted inline comments" line one after the other.
3158 """
3159 try:
3160 call()
3161 except RuntimeError as exc:
3162 # `contractual` is the severity: failing a promise is an error, failing
3163 # an addition to a verdict that did land is a warning. Printing both as
3164 # `error:` while exiting 0 told the reader the opposite of the exit code.
3165 prefix = "error" if contractual else "warning"
3166 print(f"{prefix}: could not {action}: {redact(str(exc))[0]}", file=sys.stderr)
3167 return False
3168 return True
3170 if args.post_summary:
3171 if args.issue:
3172 # Plain issues use `gh issue comment`; phased/SHA-marker posting is
3173 # PR-only, so the issue path posts the single rendered report.
3174 if not _post(
3175 f"post the verdict to issue #{args.issue}",
3176 lambda: post_issue_comment(args.issue, report, args.repo),
3177 contractual=True,
3178 ):
3179 return 2
3180 log(f"posted verdict to issue #{args.issue}")
3181 return ci_exit
3182 # Record the reviewed head SHA as a hidden marker so a later
3183 # --incremental run can review only the new range (issue #9).
3184 from .github import pr_head_sha
3185 from .incremental import reviewed_sha_marker
3187 # A missing marker costs a later `--incremental` run its narrowing; it is not
3188 # worth refusing to post the verdict over, so this one degrades rather than exits.
3189 # `pr_head_sha` is best-effort: it catches its own `gh` failure and returns
3190 # "". An earlier version of this guard wrapped it in `try/except RuntimeError`,
3191 # which is unreachable — the warning it promised could never print, and the
3192 # test only passed because it mocked `pr_head_sha` itself rather than the `gh`
3193 # call underneath. The empty string is the failure signal, so that is what is
3194 # checked.
3195 if head_sha:
3196 marker_sha = head_sha
3197 else:
3198 marker_sha = pr_head_sha(args.pr, args.repo)
3199 if not marker_sha:
3200 print(
3201 "warning: could not read the PR head sha, so this review will not "
3202 "carry an incremental marker",
3203 file=sys.stderr,
3204 )
3205 marker = f"\n\n{reviewed_sha_marker(marker_sha)}" if marker_sha else ""
3207 if args.post_mode == "phased":
3208 # Post the flow as separate, readable comments (issue #127):
3209 # Round 1 → debate → decision. The SHA marker rides the last one.
3210 from .report import render_sections
3212 sections = render_sections(
3213 outcome.reviews,
3214 outcome.debate,
3215 outcome.synthesis,
3216 chair=outcome.chair,
3217 findings=outcome.findings,
3218 warnings=outcome.warnings,
3219 groups=outcome.groups,
3220 verify=outcome.verify,
3221 vote=vote,
3222 )
3223 # The sections are markdown whatever `--format` says, so the footer
3224 # follows the opt-out alone here.
3225 phased_footer = f"\n\n{render_footer(outcome.reviews)}" if attribution_on else ""
3226 for i, (title, body) in enumerate(sections):
3227 # The last comment carries the footer (render_sections has none),
3228 # then the SHA marker, which must stay last.
3229 tail = f"{phased_footer}{marker}" if i == len(sections) - 1 else ""
3230 if not _post(
3231 f"post phased comment {i + 1} of {len(sections)} to PR #{args.pr}",
3232 lambda t=title, b=body, x=tail: post_pr_comment(
3233 args.pr, f"## {t}\n\n{b}{x}", args.repo
3234 ),
3235 contractual=True,
3236 ):
3237 return 2
3238 log(f"posted {len(sections)} phased comments to PR #{args.pr}")
3239 else:
3240 if not _post(
3241 f"post the verdict to PR #{args.pr}",
3242 lambda: post_pr_comment(args.pr, f"{report}{marker}", args.repo),
3243 contractual=True,
3244 ):
3245 return 2
3246 log(f"posted verdict to PR #{args.pr}")
3248 if args.post_inline and _post(
3249 f"post inline comments to PR #{args.pr}",
3250 lambda: post_inline_comments(
3251 args.pr, outcome.findings, repo=args.repo, dry_run=args.dry_run
3252 ),
3253 contractual=False,
3254 ):
3255 log(f"posted inline comments to PR #{args.pr}")
3257 # Optional GitHub labels (issue #7): OFF by default. Only applied when
3258 # --label is passed AND a --pr target exists; never automatic.
3259 if args.label:
3260 labels = label_strings(classify(outcome))
3261 if _post(
3262 f"apply labels to PR #{args.pr}",
3263 lambda: apply_labels(args.pr, labels, args.repo),
3264 contractual=False,
3265 ):
3266 log(f"applied labels to PR #{args.pr}: {', '.join(labels)}")
3268 return ci_exit
3271if __name__ == "__main__":
3272 raise SystemExit(main())