Coverage for src/ai_jury/cli.py: 99%

1270 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-30 06:29 +0000

1"""Command-line entry point: ``jury``. 

2 

3Examples: 

4 jury --pr 123 # review a GitHub PR 

5 jury --pr 123 --post # ...and post the verdict as a comment 

6 jury --diff-file changes.diff # review a local diff file 

7 jury --diff-file - # read a diff from stdin 

8 jury --mock # offline pipeline demo (no live CLIs) 

9 jury --doctor # local readiness diagnostics 

10 jury --config-validate # validate jury.toml and exit 

11""" 

12 

13from __future__ import annotations 

14 

15import argparse 

16import contextlib 

17import io 

18import json 

19import os 

20import shutil 

21import sys 

22from importlib import resources 

23from pathlib import Path 

24 

25from . import __version__, configtrust, panel 

26from . import doctor as doctor_module 

27from .adapters import EFFORT_LEVELS, effort_warnings, make_adapter 

28from .ci import evaluate_ci, fail_on_error 

29from .classification import classify, label_strings 

30from .config import ( 

31 ConfigError, 

32 bound_error, 

33 load_config, 

34 load_raw_config, 

35 validate_config, 

36) 

37from .github import ( 

38 apply_labels, 

39 issue_body, 

40 post_inline_comments, 

41 post_issue_comment, 

42 post_pr_comment, 

43 pr_context, 

44 pr_diff, 

45) 

46from .metadata import ( 

47 build_run_metadata, 

48 claimed_vendors, 

49 collapse_reason, 

50 resolve_min_vendors, 

51) 

52from .orchestrator import review_diff, run_jury 

53from .policy import PolicyError, load_policy 

54from .redaction import redact 

55from .report import render, render_footer, render_live_step, render_transcript 

56 

57# Hard ceiling on raw diff ingestion. The per-run ``diff.max_bytes`` budget is 

58# only applied *after* the full diff is read and split, so an unbounded 

59# ``stdin``/``--diff-file`` read could OOM the process before that cap engages 

60# (security audit 2026-06-13). This ceiling sits far above any realistic review 

61# budget; it exists solely to bound memory against a hostile/huge input. 

62_MAX_DIFF_INGEST_BYTES = 64 * 1024 * 1024 # 64 MiB 

63 

64 

65def _read_capped(fh, source: str) -> str: 

66 """Read from ``fh``, refusing inputs above the ingest ceiling. 

67 

68 The cap is enforced on **bytes**, not characters: a text read of N chars can 

69 hold up to 4N bytes for multi-byte UTF-8, so a char ceiling would admit 

70 several times the intended memory (security audit 2026-06-13, red-team). 

71 Callers pass a binary stream for real input (``sys.stdin.buffer`` / a file 

72 opened ``"rb"``); a text stream is also accepted (its read is measured by its 

73 UTF-8 byte length) so test doubles and unusual streams still work. 

74 """ 

75 data = fh.read(_MAX_DIFF_INGEST_BYTES + 1) 

76 if isinstance(data, str): 

77 if len(data.encode("utf-8", "replace")) > _MAX_DIFF_INGEST_BYTES: 

78 raise SystemExit( 

79 f"error: {source} exceeds the {_MAX_DIFF_INGEST_BYTES}-byte ingest limit" 

80 ) 

81 return data 

82 if len(data) > _MAX_DIFF_INGEST_BYTES: 

83 raise SystemExit(f"error: {source} exceeds the {_MAX_DIFF_INGEST_BYTES}-byte ingest limit") 

84 return data.decode("utf-8", errors="replace") 

85 

86 

87def _checked_revision(value: str, flag: str) -> str: 

88 """Reject a revision that cannot safely reach ``git``'s argv (issue #367). 

89 

90 ``run`` uses argv, never a shell, so quoting is not the risk — a value starting 

91 with ``-`` is: git would read it as an option rather than a revision. Refused 

92 rather than escaped, and ``--`` is passed at the call site as a second guard. 

93 Empty is refused too, since it would silently widen the diff. 

94 """ 

95 revision = (value or "").strip() 

96 if not revision: 

97 raise SystemExit(f"error: {flag} needs a revision") 

98 if revision.startswith("-"): 

99 raise SystemExit( 

100 f"error: {flag} revision {revision!r} may not start with '-' " 

101 "(git would read it as an option)" 

102 ) 

103 return revision 

104 

105 

106def _git_diff(argv: list[str], label: str) -> str: 

107 """Run a read-only git command and return its stdout, or exit with its error.""" 

108 import subprocess # local: keeps the module importable where git is absent 

109 

110 try: 

111 proc = subprocess.run(argv, capture_output=True, text=True, timeout=120) 

112 except (OSError, subprocess.SubprocessError) as exc: 

113 raise SystemExit(f"error: could not run git for {label}: {redact(str(exc))[0]}") from None 

114 if proc.returncode != 0: 

115 detail = (proc.stderr or "").strip().splitlines() 

116 raise SystemExit( 

117 f"error: git could not resolve {label}" 

118 + (f": {redact(detail[0])[0]}" if detail else "") 

119 ) 

120 if not proc.stdout.strip(): 

121 raise SystemExit(f"error: {label} produced an empty diff — nothing to review") 

122 # Same ingest ceiling every other source honours; _read_capped wants a handle. 

123 return _read_capped(io.StringIO(proc.stdout), label) 

124 

125 

126def _bundled_sample_diff() -> str: 

127 """The offline-demo diff shipped inside the package (issue #21). 

128 

129 Read from ``ai_jury/data/sample.diff`` via ``importlib.resources`` so it 

130 resolves the same way from a wheel, a zipapp, or a source checkout — no 

131 ``examples/`` directory and no repo needed. Kept byte-identical to 

132 ``examples/sample.diff`` by a test. 

133 """ 

134 return resources.files("ai_jury").joinpath("data/sample.diff").read_text(encoding="utf-8") 

135 

136 

137def _read_diff(args) -> tuple[str, str]: 

138 """Return (diff, context).""" 

139 if getattr(args, "commit", None): 

140 rev = _checked_revision(args.commit, "--commit") 

141 # `git show` of a merge commit prints no diff by default; -m picks the 

142 # first-parent view so a merge is reviewable rather than silently empty. 

143 return _git_diff( 

144 ["git", "show", "--format=", "--patch", "-m", "--first-parent", rev, "--"], 

145 f"commit {rev}", 

146 ), "" 

147 if getattr(args, "commits", None): 

148 rev = _checked_revision(args.commits, "--commits") 

149 return _git_diff(["git", "diff", rev, "--"], f"range {rev}"), "" 

150 if args.pr: 

151 return pr_diff(args.pr, args.repo), pr_context(args.pr, args.repo) 

152 if args.issue: 

153 # Issue mode (issue #221): the issue's rendered text takes the diff slot; 

154 # there is no separate context block (title/labels are folded into it). 

155 return issue_body(args.issue, args.repo), "" 

156 if args.diff_file: 

157 if args.diff_file == "-": 

158 # Prefer the byte stream so the cap is exact; fall back to the text 

159 # stream (e.g. a StringIO test double) which lacks ``.buffer``. 

160 return _read_capped(getattr(sys.stdin, "buffer", sys.stdin), "stdin"), "" 

161 try: 

162 with Path(args.diff_file).open("rb") as fh: 

163 return _read_capped(fh, args.diff_file), "" 

164 except (OSError, UnicodeDecodeError) as exc: 

165 raise SystemExit( 

166 f"error reading diff file '{args.diff_file}': {redact(str(exc))[0]}" 

167 ) from None 

168 if getattr(args, "mock", False): 

169 # Offline demo (issue #21): `--mock` with no diff source reviews a diff 

170 # bundled in the package, so `jury --mock` / `jury --mock --theater` 

171 # deliver the deliberation the docs promise on a fresh install — no 

172 # checkout, no PR, no network. A real (non-`--mock`) run still requires 

173 # an explicit source. Announced on stderr so the payments.py sample is 

174 # never mistaken for the caller's own change. 

175 print( 

176 "no diff source given; reviewing the bundled offline-demo diff. " 

177 "Pass --diff-file/--pr/--commit to review your own change.", 

178 file=sys.stderr, 

179 ) 

180 return _bundled_sample_diff(), "" 

181 raise SystemExit( 

182 "error: provide one of --pr, --issue, --diff-file, --commit, --commits " 

183 "(or --diff-file - for stdin)" 

184 ) 

185 

186 

187def build_parser() -> argparse.ArgumentParser: 

188 p = argparse.ArgumentParser( 

189 prog="jury", 

190 description="Cross-vendor multi-agent PR review jury.", 

191 # The subcommands are argv-intercepts (handled before this parser runs), 

192 # so argparse cannot list them on its own — and until #661 nothing in 

193 # `--help` mentioned that they exist at all. 

194 epilog=( 

195 "Subcommands (each takes its own --help): init, config, comment, " 

196 "apply, replay, run-agent, cache clear, examples, guide." 

197 ), 

198 ) 

199 src = p.add_argument_group("input") 

200 src.add_argument("--pr", help="GitHub PR number/URL to review (uses `gh`)") 

201 src.add_argument( 

202 "--issue", 

203 help="GitHub issue number/URL to review for completeness/clarity (uses " 

204 "`gh`); runs the full jury with an issue-quality rubric", 

205 ) 

206 src.add_argument("--repo", help="owner/name for --pr/--issue (defaults to current repo)") 

207 src.add_argument("--diff-file", help="path to a diff file, or '-' for stdin") 

208 src.add_argument("--commit", help="review the diff one commit introduces (needs a git repo)") 

209 src.add_argument( 

210 "--commits", 

211 help="review a commit range, e.g. origin/main..HEAD or HEAD~5..HEAD (needs a git repo)", 

212 ) 

213 

214 p.add_argument("--config", help="path to jury.toml (default: ./jury.toml or built-in)") 

215 p.add_argument( 

216 "--policy", 

217 type=Path, 

218 default=None, 

219 help="path to an optional repository review policy file (default: " 

220 "auto-discover .jury/policy.toml or jury-policy.toml); " 

221 "missing policy files are allowed", 

222 ) 

223 p.add_argument( 

224 "--context-mode", 

225 choices=["diff-only", "expanded"], 

226 default=None, 

227 help="context policy: diff-only sends only the diff; expanded includes PR context", 

228 ) 

229 p.add_argument( 

230 "--redact", 

231 dest="redact", 

232 action="store_true", 

233 default=None, 

234 help="redact secrets from prompt text before sending (default: from config)", 

235 ) 

236 p.add_argument( 

237 "--no-redact", 

238 dest="redact", 

239 action="store_false", 

240 help="do not redact secrets before sending", 

241 ) 

242 p.add_argument( 

243 "--rounds", 

244 type=int, 

245 help="override number of rounds (1=review, 2=+debate); a fixed value " 

246 "disables early-stop for reproducible benchmarking", 

247 ) 

248 p.add_argument( 

249 "--max-rounds", 

250 type=int, 

251 help="ceiling on adaptive rounds when early-stop is on", 

252 ) 

253 p.add_argument( 

254 "--early-stop", 

255 dest="early_stop", 

256 action="store_true", 

257 default=None, 

258 help="stop after round 1 when reviewers agree; debate only on disagreement", 

259 ) 

260 p.add_argument( 

261 "--no-early-stop", 

262 dest="early_stop", 

263 action="store_false", 

264 help="disable adaptive early-stop (honour a fixed number of rounds)", 

265 ) 

266 p.add_argument( 

267 "--auto", 

268 dest="auto", 

269 action="store_true", 

270 default=None, 

271 help="risk-aware auto-depth: scale rounds/verify to the diff", 

272 ) 

273 p.add_argument( 

274 "--no-auto", 

275 dest="auto", 

276 action="store_false", 

277 help="disable auto-depth (use configured/fixed rounds)", 

278 ) 

279 p.add_argument( 

280 "--total-timeout", 

281 type=int, 

282 help="overall wall-clock budget (seconds) for the whole run", 

283 ) 

284 p.add_argument( 

285 "--phase-timeout", 

286 type=int, 

287 help="per-phase wall-clock budget (seconds)", 

288 ) 

289 p.add_argument( 

290 "--retries", 

291 type=int, 

292 help="extra attempts for transient (timeout/rate-limit/spawn) failures", 

293 ) 

294 p.add_argument( 

295 "--max-diff-bytes", 

296 type=int, 

297 help="size budget for the (filtered) diff before chunking/too-large", 

298 ) 

299 p.add_argument( 

300 "--chunk", 

301 dest="chunk", 

302 action="store_true", 

303 default=None, 

304 help="chunk an over-budget diff by file instead of failing", 

305 ) 

306 p.add_argument( 

307 "--no-chunk", 

308 dest="chunk", 

309 action="store_false", 

310 help="disable diff chunking (fail clearly when over budget)", 

311 ) 

312 p.add_argument( 

313 "--exclude", 

314 action="append", 

315 metavar="GLOB", 

316 default=None, 

317 help="exclude files matching this path glob (repeatable)", 

318 ) 

319 p.add_argument( 

320 "--include", 

321 action="append", 

322 metavar="GLOB", 

323 default=None, 

324 help="only review files matching this path glob (repeatable)", 

325 ) 

326 p.add_argument( 

327 "--seed", 

328 type=int, 

329 help="run seed for reproducible orchestration; mock runs with the same seed " 

330 "produce byte-identical reports (overrides [jury] seed)", 

331 ) 

332 p.add_argument("--chair", help="override the synthesizing chair agent") 

333 p.add_argument( 

334 "--mock", action="store_true", help="offline demo: use deterministic mock agents" 

335 ) 

336 p.add_argument( 

337 "--strict", action="store_true", help="fail if any configured agent CLI is missing" 

338 ) 

339 p.add_argument( 

340 "--min-vendors", 

341 type=int, 

342 default=None, 

343 metavar="N", 

344 help=( 

345 "fail (exit 3) unless at least N distinct vendors contributed a " 

346 "review; 0 disables. Default from config ([jury.ci] min_vendors, " 

347 "shipped as 2). --strict checks availability at startup, this " 

348 "checks participation at the end" 

349 ), 

350 ) 

351 p.add_argument( 

352 "--no-min-vendors", 

353 dest="min_vendors", 

354 action="store_const", 

355 const=0, 

356 help=( 

357 "opt out of the cross-vendor guard: accept a run whose panel " 

358 "collapsed to a single vendor (same as --min-vendors 0)" 

359 ), 

360 ) 

361 p.add_argument( 

362 "--min-reviews", 

363 type=int, 

364 default=None, 

365 metavar="N", 

366 help=( 

367 "require at least N reviews for a downstream consumer (a panel ballot " 

368 "that names what it read and votes; neither the chair's synthesis " 

369 "record nor an abstaining ballot is one); 0 disables. Checked before " 

370 "the panel runs and again on the result (exit 3). " 

371 "Default from config ([jury.ci] min_reviews, shipped as 0)" 

372 ), 

373 ) 

374 p.add_argument( 

375 "--verify", 

376 dest="verify", 

377 action="store_true", 

378 default=None, 

379 help="run the verification round (default: from config)", 

380 ) 

381 p.add_argument( 

382 "--no-verify", 

383 dest="verify", 

384 action="store_false", 

385 help="skip the verification round", 

386 ) 

387 p.add_argument( 

388 "--doctor", 

389 action="store_true", 

390 help="print a local readiness diagnostics report and exit (no telemetry is collected or sent)", 

391 ) 

392 p.add_argument( 

393 "--json", 

394 action="store_true", 

395 help=( 

396 "with --doctor, print the machine-readable provider export " 

397 "(schema ai-jury.doctor.v1) as the only thing on stdout" 

398 ), 

399 ) 

400 p.add_argument( 

401 "--write", 

402 help="with --doctor, also write the diagnostics as JSON to this path (secrets redacted)", 

403 ) 

404 p.add_argument( 

405 "--effort", 

406 choices=list(EFFORT_LEVELS), 

407 default=None, 

408 help=( 

409 "reasoning effort for every agent this run (overrides [[agent]] effort); " 

410 "ignored with a warning for agents whose vendor has no effort control" 

411 ), 

412 ) 

413 p.add_argument("-o", "--output", help="write the report to a file instead of stdout") 

414 p.add_argument( 

415 "--metadata-json", 

416 metavar="PATH", 

417 help="write machine-readable run metadata (durations, status, rounds) as JSON", 

418 ) 

419 p.add_argument( 

420 "--format", 

421 choices=["markdown", "json", "sarif", "keel-reviews"], 

422 default="markdown", 

423 help="output format for stdout/--output (default: markdown); " 

424 "'keel-reviews' emits one review record per panelist plus the chair", 

425 ) 

426 p.add_argument( 

427 "--decision", 

428 choices=["chair", "vote"], 

429 default=None, 

430 help="final verdict: 'chair' synthesis (default) or panel 'vote' (tally " 

431 "the reviewers); overrides [jury] decision", 

432 ) 

433 p.add_argument( 

434 "--transcript", 

435 dest="transcript", 

436 action="store_true", 

437 default=None, 

438 help="render the full play-by-play transcript (each agent's review, the " 

439 "debate, and the chair's reasoning) instead of the summary report", 

440 ) 

441 p.add_argument( 

442 "--no-transcript", 

443 dest="transcript", 

444 action="store_false", 

445 help="force the summary report even if [jury] transcript is set", 

446 ) 

447 p.add_argument( 

448 "--verbose", 

449 dest="verbose", 

450 action="store_true", 

451 help="summary report followed by the full transcript, in one document", 

452 ) 

453 p.add_argument( 

454 "--live", 

455 dest="live", 

456 action="store_true", 

457 help="stream each step (review, debate, verdict) to stdout as it happens; " 

458 "add --pr --post to also post each step as its own PR comment", 

459 ) 

460 p.add_argument( 

461 "--theater", 

462 dest="theater", 

463 action="store_true", 

464 default=None, 

465 help="animated deliberation view of the live run (each model seated " 

466 "around a table, speaking per phase, panel-vote/chair finale); needs an " 

467 "interactive terminal, else falls back to --live. Can be defaulted on in " 

468 "jury.toml ([jury] theater = true)", 

469 ) 

470 p.add_argument( 

471 "--no-theater", 

472 dest="theater", 

473 action="store_false", 

474 help="disable the theater scene even if jury.toml enables it", 

475 ) 

476 p.add_argument( 

477 "--theater-style", 

478 dest="theater_style", 

479 choices=("flat", "pixel"), 

480 default=None, 

481 help="--theater scene style: 'flat' (ANSI line scene, default) or " 

482 "'pixel' (pixel-art room; needs a truecolor+unicode terminal). Defaults " 

483 "from jury.toml ([jury] theater_style)", 

484 ) 

485 p.add_argument( 

486 "--post-summary", 

487 "--post", 

488 dest="post_summary", 

489 action="store_true", 

490 help="post the report as a single summary comment on --pr", 

491 ) 

492 p.add_argument( 

493 "--no-attribution", 

494 dest="attribution", 

495 action="store_false", 

496 default=None, 

497 help="leave the ai-jury footer naming the seats that reviewed off the " 

498 "markdown report and posted comments (default from jury.toml " 

499 "[jury.output] attribution, on)", 

500 ) 

501 p.add_argument( 

502 "--post-inline", 

503 dest="post_inline", 

504 action="store_true", 

505 help="post inline review comments for located findings on --pr", 

506 ) 

507 p.add_argument( 

508 "--post-progress", 

509 dest="post_progress", 

510 action="store_true", 

511 help="keep a live, sticky status comment on --pr updated per round/chunk", 

512 ) 

513 p.add_argument( 

514 "--post-mode", 

515 choices=["single", "phased"], 

516 # None is "not passed" (read as 'single'), so a `--post-mode` with no 

517 # summary to shape can be told apart from the default and refused. 

518 default=None, 

519 help="with --post-summary: 'single' (one comment) or 'phased' (separate " 

520 "Round 1 / debate / decision comments)", 

521 ) 

522 p.add_argument( 

523 "--dry-run", 

524 dest="dry_run", 

525 action="store_true", 

526 help="with --post-inline, print what would be posted without calling GitHub", 

527 ) 

528 p.add_argument( 

529 "--label", 

530 dest="label", 

531 action="store_true", 

532 help="apply classification labels (review effort / risk / security) to " 

533 "--pr (off by default; never applied automatically)", 

534 ) 

535 p.add_argument( 

536 "--ci", 

537 action="store_true", 

538 help="CI mode: exit non-zero when blocking findings remain", 

539 ) 

540 p.add_argument( 

541 "--fail-on", 

542 help="comma-separated severities that fail CI (overrides config)", 

543 ) 

544 p.add_argument( 

545 "--cache", 

546 action="store_true", 

547 help="use the local result cache: reuse a cached outcome for an unchanged " 

548 "diff+config, else run and store it (off by default)", 

549 ) 

550 p.add_argument( 

551 "--clear-cache", 

552 action="store_true", 

553 help="delete all local cache entries and exit (also: `jury cache clear`)", 

554 ) 

555 p.add_argument( 

556 "--cache-dir", 

557 help="override the cache directory (default: $JURY_CACHE_DIR or ~/.cache/ai-jury)", 

558 ) 

559 p.add_argument( 

560 "--suggest-patches", 

561 dest="suggest_patches", 

562 action="store_true", 

563 help="emit a separate, opt-in suggested-patches section for VERIFIED " 

564 "findings (read-only; never applied automatically)", 

565 ) 

566 p.add_argument( 

567 "--patches-out", 

568 metavar="PATH", 

569 help="with --suggest-patches, write the patches to this file instead of " 

570 "appending them after the report", 

571 ) 

572 p.add_argument( 

573 "--incremental", 

574 action="store_true", 

575 help="review only the diff since the last jury run on --pr when a prior " 

576 "marker exists, else fall back to a full review", 

577 ) 

578 p.add_argument("-q", "--quiet", action="store_true", help="suppress progress logs on stderr") 

579 p.add_argument( 

580 "--config-validate", 

581 action="store_true", 

582 help="validate the resolved config and exit (0 valid, 2 invalid)", 

583 ) 

584 p.add_argument( 

585 "--strict-config", 

586 action="store_true", 

587 help="treat configuration warnings as errors", 

588 ) 

589 p.add_argument( 

590 "--tiered", 

591 action="store_true", 

592 help="opt-in risk-aware tiered model routing with frontier anchor (issue #524)", 

593 ) 

594 p.add_argument( 

595 "--hints", 

596 dest="hints", 

597 action="store_true", 

598 default=None, 

599 help="run local static analysis pre-pass (Ruff/ESLint) to inject hints (issue #523)", 

600 ) 

601 p.add_argument( 

602 "--no-hints", 

603 dest="hints", 

604 action="store_false", 

605 help="skip the static analysis pre-pass even if jury.toml enables it", 

606 ) 

607 p.add_argument("--version", action="version", version=f"%(prog)s {__version__}") 

608 return p 

609 

610 

611def _run_apply(rest: list[str]) -> int: 

612 """Handle `jury apply` (issue #521): apply verified patch suggestions.""" 

613 from .patches import ( 

614 apply_patch_suggestion, 

615 parse_patch_suggestions, 

616 preview_patch_suggestion, 

617 ) 

618 

619 sub = argparse.ArgumentParser( 

620 prog="jury apply", description="Apply verified suggested patches to the repository." 

621 ) 

622 sub.add_argument( 

623 "index", 

624 nargs="?", 

625 default=None, 

626 help="1-indexed patch suggestion number to apply, or 'all'. Required — " 

627 "applying every suggestion is the wrong default for a command that " 

628 "writes to the working tree", 

629 ) 

630 sub.add_argument( 

631 "--report", 

632 "-r", 

633 help="Path to a markdown report file or patch file (defaults to stdin)", 

634 ) 

635 sub.add_argument( 

636 "--dry-run", 

637 action="store_true", 

638 help="Print the paths each suggestion would touch and write nothing", 

639 ) 

640 sub.add_argument( 

641 "--yes", 

642 "-y", 

643 action="store_true", 

644 help="Skip the confirmation prompt; required when stdin is not a terminal", 

645 ) 

646 ns = sub.parse_args(rest) 

647 

648 content = "" 

649 if ns.report: 

650 p = Path(ns.report) 

651 try: 

652 if not p.is_file(): 

653 print(f"Error: report file not found: {ns.report}", file=sys.stderr) 

654 return 2 

655 content = p.read_text(encoding="utf-8") 

656 except (OSError, UnicodeDecodeError) as exc: 

657 print( 

658 f"Error reading report file '{ns.report}': {redact(str(exc))[0]}", file=sys.stderr 

659 ) 

660 return 2 

661 elif sys.stdin is not None and not sys.stdin.isatty(): 

662 content = sys.stdin.read() 

663 else: 

664 print( 

665 "Error: provide a report file via --report <file> or pipe a report via stdin", 

666 file=sys.stderr, 

667 ) 

668 return 2 

669 

670 suggestions = parse_patch_suggestions(content) 

671 if not suggestions: 

672 print("No verified patch suggestions found in the provided report.", file=sys.stderr) 

673 return 1 

674 

675 if ns.index is None: 

676 print( 

677 f"Error: choose what to apply — an index from 1 to {len(suggestions)}, or 'all'.\n" 

678 " `jury apply --dry-run all` shows what each one would touch.", 

679 file=sys.stderr, 

680 ) 

681 return 2 

682 

683 if ns.index.lower() == "all": 

684 selected = list(enumerate(suggestions, 1)) 

685 else: 

686 try: 

687 target_idx = int(ns.index) - 1 

688 except ValueError: 

689 print(f"Error: invalid index '{ns.index}'", file=sys.stderr) 

690 return 2 

691 if not (0 <= target_idx < len(suggestions)): 

692 print( 

693 f"Error: patch index {ns.index} out of range (found {len(suggestions)} suggestions)", 

694 file=sys.stderr, 

695 ) 

696 return 2 

697 selected = [(target_idx + 1, suggestions[target_idx])] 

698 

699 # Preview before writing, always. The report is derived from a diff that may be 

700 # attacker-influenced and its suggestions were written by an LLM, so the 

701 # operator gets git's own answer about what would change *before* anything 

702 # changes. The old output was printed after the write had happened (#605). 

703 print("These suggestions would touch:", file=sys.stderr) 

704 for number, suggestion in selected: 

705 paths, refusal = preview_patch_suggestion(suggestion) 

706 listed = ", ".join(paths) if paths else "(nothing git could read)" 

707 print(f" [{number}] {suggestion.file}: {listed}", file=sys.stderr) 

708 if refusal is not None: 

709 print(f" would be refused: {refusal}", file=sys.stderr) 

710 

711 if ns.dry_run: 

712 print("Dry run: nothing was written.", file=sys.stderr) 

713 return 0 

714 

715 if not ns.yes: 

716 # Piping a report in is exactly the unattended case, and stdin may already 

717 # be consumed by the report itself — so silence is refused rather than read 

718 # as consent. 

719 if sys.stdin is None or not sys.stdin.isatty(): 

720 print( 

721 "Error: refusing to write without confirmation. Re-run with --yes to " 

722 "confirm, or --dry-run to preview.", 

723 file=sys.stderr, 

724 ) 

725 return 2 

726 answer = input(f"Apply {len(selected)} suggestion(s)? [y/N] ").strip().lower() 

727 if answer not in ("y", "yes"): 

728 print("Aborted; nothing was written.", file=sys.stderr) 

729 return 1 

730 

731 success_count = 0 

732 total = len(selected) 

733 for number, suggestion in selected: 

734 ok, msg = apply_patch_suggestion(suggestion) 

735 if ok: 

736 success_count += 1 

737 print(f"✓ [{number}/{total}] {msg}") 

738 else: 

739 print(f"✗ [{number}/{total}] {msg}", file=sys.stderr) 

740 return 0 if success_count > 0 else 1 

741 

742 

743def _run_comment_command(rest: list[str]) -> int: 

744 """Handle ``jury comment`` (issue #11): parse an allowlisted PR-comment 

745 command and either print the resolved jury args or dispatch the run. 

746 

747 Returns 2 on a rejected/invalid command (so a workflow can ignore it), else 

748 the dispatched run's exit code (or 0 with --print-args). 

749 """ 

750 import shlex 

751 

752 from .commands import CommandError, parse_comment 

753 

754 sub = argparse.ArgumentParser(prog="jury comment", add_help=True) 

755 sub.add_argument("--text", required=True, help="the PR comment body to parse") 

756 sub.add_argument("--pr", help="PR number/URL to review and post back to") 

757 sub.add_argument("--repo", help="owner/name (defaults to current repo)") 

758 sub.add_argument( 

759 "--print-args", 

760 dest="print_args", 

761 action="store_true", 

762 help="print the resolved jury args instead of running", 

763 ) 

764 sub.add_argument( 

765 "--no-post", 

766 dest="no_post", 

767 action="store_true", 

768 help="do not post the result back as a summary comment", 

769 ) 

770 ns = sub.parse_args(rest) 

771 

772 try: 

773 parsed = parse_comment(ns.text) 

774 except CommandError as exc: 

775 print(f"comment command rejected: {redact(str(exc))[0]}", file=sys.stderr) 

776 return 2 

777 

778 inner = parsed.to_cli_args() 

779 if ns.pr: 

780 inner += ["--pr", ns.pr] 

781 if not ns.no_post: 

782 inner += ["--post-summary"] 

783 if ns.repo: 

784 inner += ["--repo", ns.repo] 

785 

786 if ns.print_args: 

787 print(" ".join(shlex.quote(a) for a in inner)) 

788 return 0 

789 return main(inner) 

790 

791 

792_AGENT_BLURB = { 

793 "claude": "Claude Code (Anthropic)", 

794 "codex": "Codex CLI (OpenAI)", 

795 "agy": "Antigravity (Google) — opt-in only, cannot be confined", 

796 "qwen": "local / open-weight via Ollama (free, offline)", 

797 "claude-api": "hosted Anthropic API (ANTHROPIC_API_KEY, no CLI needed)", 

798 "codex-api": "hosted OpenAI API (OPENAI_API_KEY, no CLI needed)", 

799 "gemini-api": "hosted Google Gemini API (GEMINI_API_KEY, no CLI needed)", 

800 "openrouter": "hosted OpenRouter API (OPENROUTER_API_KEY)", 

801 "deepseek": "hosted DeepSeek API (DEEPSEEK_API_KEY)", 

802 "groq": "hosted Groq API (GROQ_API_KEY)", 

803 "aider": "generic CLI coding agent (Aider)", 

804} 

805 

806 

807def _default_init_agents(available: dict) -> list[str]: 

808 """The agents an interactive `jury init` pre-fills: detected, never opt-in-only. 

809 

810 agy stays listed and can be typed in, but pressing Enter never seats it. 

811 """ 

812 from .scaffold import KNOWN_AGENTS, implicit_choices 

813 

814 detected = implicit_choices(n for n in KNOWN_AGENTS if available.get(n)) 

815 return detected or implicit_choices(KNOWN_AGENTS) 

816 

817 

818def _init_available() -> dict: 

819 """Map each known agent name to whether it is reachable right now.""" 

820 from .config import AgentSpec 

821 from .scaffold import KNOWN_AGENTS, agent_templates 

822 

823 templates = agent_templates() 

824 out = {} 

825 for name in KNOWN_AGENTS: 

826 try: 

827 out[name] = make_adapter(AgentSpec(**templates[name])).available() 

828 except Exception: # noqa: BLE001 - detection is best-effort 

829 out[name] = False 

830 return out 

831 

832 

833def _stdin_is_terminal() -> bool: 

834 """Whether ``jury init`` can prompt: stdin exists and is a terminal. 

835 

836 ``sys.stdin`` is ``None`` when the process has no stdin at all (a detached 

837 or daemonised launch), and ``None.isatty()`` is an ``AttributeError`` (#897). 

838 """ 

839 return sys.stdin is not None and sys.stdin.isatty() 

840 

841 

842def _init_interactive(available: dict, input_fn=input, local_endpoint=None, models_fn=None) -> dict: 

843 """Prompt for jury settings; returns kwargs for scaffold.build_config. 

844 

845 ``input_fn`` and ``models_fn`` are injectable for testing (the latter lists 

846 local models). Defaults are pre-filled from the detected agents/models so 

847 pressing Enter accepts a sensible config. 

848 """ 

849 from .scaffold import KNOWN_AGENTS 

850 

851 if models_fn is None: 

852 from .adapters import list_local_models as models_fn 

853 

854 print("Configure a review jury (jury.toml).\n", file=sys.stderr) 

855 for name in KNOWN_AGENTS: 

856 mark = "available" if available.get(name) else "not found" 

857 print(f" - {name}: {_AGENT_BLURB[name]} [{mark}]", file=sys.stderr) 

858 default_agents = _default_init_agents(available) 

859 raw_agents = input_fn(f"\nAgents to include [default: {','.join(default_agents)}]: ").strip() 

860 agents = [a.strip() for a in raw_agents.split(",") if a.strip()] or default_agents 

861 

862 rounds_raw = input_fn("Rounds — 1=review, 2=+debate [2]: ").strip() 

863 rounds = int(rounds_raw) if rounds_raw.isdigit() else 2 

864 

865 chair_default = agents[0] if agents else "claude" 

866 chair = input_fn(f"Chair agent [{chair_default}]: ").strip() or chair_default 

867 

868 verify = (input_fn("Run verification round? [Y/n]: ").strip().lower() or "y") != "n" 

869 

870 local_model = None 

871 has_local = any(a in agents for a in ("qwen", "local")) 

872 if has_local: 

873 from .scaffold import pick_default_model 

874 

875 models = models_fn(local_endpoint or "http://localhost:11434/v1") 

876 if models: 

877 default = pick_default_model(models) 

878 print("\nLocal models available on the server:", file=sys.stderr) 

879 for i, m in enumerate(models, 1): 

880 star = " (default)" if m == default else "" 

881 print(f" {i}. {m}{star}", file=sys.stderr) 

882 raw = input_fn(f"Pick a local model [number or name, default: {default}]: ").strip() 

883 if raw.isdigit() and 1 <= int(raw) <= len(models): 

884 local_model = models[int(raw) - 1] 

885 elif raw: 

886 local_model = raw 

887 else: 

888 local_model = default 

889 else: 

890 print( 

891 "\n(could not reach the local server to list models; using the default)", 

892 file=sys.stderr, 

893 ) 

894 local_model = input_fn("Local model name [qwen2.5-coder:7b]: ").strip() or None 

895 

896 # Reasoning effort (issue #662). Skippable: Enter leaves it unset, so 

897 # nothing is written and every agent keeps its vendor default. Asked last so 

898 # the established question order is unchanged. 

899 effort_raw = input_fn("Reasoning effort — low/medium/high [skip]: ").strip().lower() 

900 effort = effort_raw if effort_raw in EFFORT_LEVELS else None 

901 

902 return { 

903 "agents": agents, 

904 "rounds": rounds, 

905 "chair": chair, 

906 "verify": verify, 

907 "local_model": local_model, 

908 "effort": effort, 

909 } 

910 

911 

912def _init_wizard(available: dict, input_fn=input, local_endpoint=None, models_fn=None) -> dict: 

913 """Guided, numbered-option setup for ``jury init --wizard`` (issue #231). 

914 

915 Mirrors :func:`_init_interactive`'s injectable params for offline testing. 

916 Every question is SKIPPABLE: pressing Enter leaves the setting unset, so it 

917 falls back to the built-in default and is NOT written to ``jury.toml`` (which 

918 keeps the generated file minimal). Returns kwargs for ``scaffold.build_config`` 

919 containing only the values the user explicitly chose. 

920 """ 

921 from .scaffold import KNOWN_AGENTS 

922 

923 if models_fn is None: 

924 from .adapters import list_local_models as models_fn 

925 

926 def ask(prompt: str) -> str: 

927 return input_fn(prompt).strip() 

928 

929 def choose(prompt: str, options: list[str], default_idx: int) -> int | None: 

930 """Print numbered options and read a 1-based pick. Enter -> None (skip).""" 

931 print(prompt, file=sys.stderr) 

932 for i, label in enumerate(options, 1): 

933 star = " (default)" if i - 1 == default_idx else "" 

934 print(f" {i}. {label}{star}", file=sys.stderr) 

935 raw = ask("Pick a number [Enter to keep default]: ") 

936 if not raw: 

937 return None 

938 if raw.isdigit() and 1 <= int(raw) <= len(options): 

939 return int(raw) - 1 

940 return None 

941 

942 print( 

943 "jury init --wizard — guided setup (writes jury.toml).\n" 

944 "Every question is optional: press Enter to keep the default and skip it;\n" 

945 "skipped settings are left at their built-in defaults (not written).\n", 

946 file=sys.stderr, 

947 ) 

948 

949 # Reviewers (always written — like plain init). 

950 for name in KNOWN_AGENTS: 

951 mark = "available" if available.get(name) else "not found" 

952 print(f" - {name}: {_AGENT_BLURB[name]} [{mark}]", file=sys.stderr) 

953 default_agents = _default_init_agents(available) 

954 raw_agents = ask(f"\nReviewers to include [default: {','.join(default_agents)}]: ") 

955 agents = [a.strip() for a in raw_agents.split(",") if a.strip()] or default_agents 

956 

957 kwargs: dict = {"agents": agents} 

958 

959 # Depth -> rounds / early_stop / auto_depth. 

960 depth = choose( 

961 "\nDepth:", 

962 [ 

963 "1 round (review only)", 

964 "2 rounds + debate", 

965 "adaptive (early-stop)", 

966 "auto-depth (scale to the diff)", 

967 ], 

968 default_idx=1, 

969 ) 

970 if depth == 0: 

971 kwargs["rounds"] = 1 

972 elif depth == 1: 

973 kwargs["rounds"] = 2 

974 elif depth == 2: 

975 kwargs["rounds"] = 2 

976 kwargs["early_stop"] = True 

977 elif depth == 3: 

978 kwargs["auto_depth"] = True 

979 

980 # Decision: chair (default) or panel vote. Only written on a non-default. 

981 decision = choose("\nDecision:", ["chair synthesis", "panel vote"], default_idx=0) 

982 if decision == 1: 

983 kwargs["decision"] = "vote" 

984 

985 # Verification (always written — like plain init). 

986 verify_raw = ask("\nRun verification round? [Y/n]: ").lower() 

987 if verify_raw: 

988 kwargs["verify"] = verify_raw != "n" 

989 

990 # Context: diff-only (default) or expanded; redact secrets Y/n. 

991 ctx = choose( 

992 "\nContext sent to reviewers:", 

993 ["diff-only", "expanded (include PR context)"], 

994 default_idx=0, 

995 ) 

996 if ctx == 1: 

997 kwargs["context_mode"] = "expanded" 

998 redact_raw = ask("Redact secrets before sending? [Y/n]: ").lower() 

999 if redact_raw == "n": 

1000 kwargs["redact_secrets"] = False 

1001 

1002 # CI gate fail-on. Only write [jury.ci] on a non-default pick. 

1003 gate = choose( 

1004 "\nCI gate — fail on which severities?", 

1005 ["critical,major", "critical only", "skip (never fail CI)"], 

1006 default_idx=0, 

1007 ) 

1008 if gate == 1: 

1009 kwargs["ci_fail_on"] = ["critical"] 

1010 elif gate == 2: 

1011 kwargs["ci_fail_on"] = [] 

1012 

1013 # Chair (always written — like plain init; default = first reviewer). 

1014 chair_default = agents[0] if agents else "claude" 

1015 chair = ask(f"\nChair agent [{chair_default}]: ") or chair_default 

1016 kwargs["chair"] = chair 

1017 

1018 # Local model pick when a local reviewer is chosen (reuse init's logic). 

1019 if any(a in agents for a in ("qwen", "local")): 

1020 from .scaffold import pick_default_model 

1021 

1022 models = models_fn(local_endpoint or "http://localhost:11434/v1") 

1023 if models: 

1024 default = pick_default_model(models) 

1025 print("\nLocal models available on the server:", file=sys.stderr) 

1026 for i, m in enumerate(models, 1): 

1027 star = " (default)" if m == default else "" 

1028 print(f" {i}. {m}{star}", file=sys.stderr) 

1029 raw = ask(f"Pick a local model [number or name, default: {default}]: ") 

1030 if raw.isdigit() and 1 <= int(raw) <= len(models): 

1031 kwargs["local_model"] = models[int(raw) - 1] 

1032 elif raw: 

1033 kwargs["local_model"] = raw 

1034 else: 

1035 kwargs["local_model"] = default 

1036 else: 

1037 print( 

1038 "\n(could not reach the local server to list models; using the default)", 

1039 file=sys.stderr, 

1040 ) 

1041 typed = ask("Local model name [qwen2.5-coder:7b]: ") 

1042 if typed: 

1043 kwargs["local_model"] = typed 

1044 

1045 return kwargs 

1046 

1047 

1048def _run_init(rest: list[str]) -> int: 

1049 """Handle ``jury init`` (issue #107): scaffold a jury.toml.""" 

1050 from .config import AGY_OPT_IN_NOTE, ConfigError, validate_config 

1051 from .scaffold import ( 

1052 KNOWN_AGENTS, 

1053 OPT_IN_AGENTS, 

1054 PRESETS, 

1055 agent_templates, 

1056 agents_needing_remote_opt_in, 

1057 build_config, 

1058 implicit_choices, 

1059 render_toml, 

1060 seat_local_agents, 

1061 ) 

1062 

1063 sub = argparse.ArgumentParser(prog="jury init") 

1064 sub.add_argument( 

1065 "--preset", 

1066 choices=sorted(PRESETS), 

1067 help="setup preset: offline (local-only), fast (1 round), balanced " 

1068 "(debate + early-stop), thorough (all agents + debate + verify)", 

1069 ) 

1070 sub.add_argument( 

1071 "--agents", 

1072 help="comma-separated: claude,codex,qwen (agy only by name: it cannot be confined)", 

1073 ) 

1074 sub.add_argument("--rounds", type=int, default=None) 

1075 sub.add_argument("--chair") 

1076 sub.add_argument("--verify", dest="verify", action="store_true", default=None) 

1077 sub.add_argument("--no-verify", dest="verify", action="store_false") 

1078 sub.add_argument("--local-model", help="model id for a local agent (qwen)") 

1079 sub.add_argument("--local-endpoint", help="OpenAI-compatible base URL for a local agent") 

1080 sub.add_argument("-o", "--output", default="jury.toml") 

1081 sub.add_argument("--force", action="store_true", help="overwrite an existing file") 

1082 sub.add_argument("--interactive", action="store_true", help="force interactive prompts") 

1083 sub.add_argument( 

1084 "--wizard", 

1085 action="store_true", 

1086 help="guided, numbered-option setup; every question is skippable (Enter " 

1087 "keeps the built-in default) and only chosen keys are written", 

1088 ) 

1089 sub.add_argument( 

1090 "--list-agents", action="store_true", help="list known agents + availability and exit" 

1091 ) 

1092 sub.add_argument( 

1093 "--list-models", action="store_true", help="list local models on the server and exit" 

1094 ) 

1095 ns = sub.parse_args(rest) 

1096 

1097 # The wizard reads every answer from a terminal. With none — CI, a pipe, an 

1098 # agent's shell — its first prompt died as an `EOFError` traceback (#865), so it 

1099 # is refused here, before the availability probes spend anything. The listing 

1100 # flags answer before the wizard would ever ask, so they are left to do so. 

1101 listing = ns.list_models or ns.list_agents 

1102 if ns.wizard and not listing and not _stdin_is_terminal(): 

1103 print( 

1104 "error: the wizard needs a terminal; use `jury init --preset <name>`", 

1105 file=sys.stderr, 

1106 ) 

1107 return 2 

1108 # `--interactive` is the same promise and died the same way (#897): an 

1109 # `EOFError` on its first prompt from a pipe, and with no stdin at all an 

1110 # `AttributeError` from the TTY test below. Refused where it would prompt — 

1111 # `--agents`/`--preset` answer the questions, so it never asks then. 

1112 if ns.interactive and not (listing or ns.agents or ns.preset) and not _stdin_is_terminal(): 

1113 print( 

1114 "error: --interactive needs a terminal; use `jury init --preset <name>` " 

1115 "or `jury init --agents <list>`", 

1116 file=sys.stderr, 

1117 ) 

1118 return 2 

1119 

1120 from .adapters import list_local_models, local_model_listing 

1121 from .redaction import redact_url_userinfo 

1122 

1123 endpoint = ns.local_endpoint or "http://localhost:11434/v1" 

1124 # Strip any userinfo credentials before echoing the endpoint to stdout/CI 

1125 # logs (issue #316/L-7, completed in v1.5.0/L-1: structural strip catches 

1126 # short and colon-less userinfo the regex missed), mirroring doctor.py. 

1127 endpoint_disp = redact_url_userinfo(endpoint) 

1128 

1129 if ns.list_models: 

1130 models = list_local_models(endpoint) 

1131 if not models: 

1132 print(f"No local models found (is a server reachable at {endpoint_disp}?).") 

1133 return 0 

1134 print(f"Local models at {endpoint_disp}:") 

1135 for m in models: 

1136 print(f" - {m}") 

1137 return 0 

1138 

1139 available = _init_available() 

1140 

1141 if ns.list_agents: 

1142 for name in KNOWN_AGENTS: 

1143 mark = "available" if available.get(name) else "not found" 

1144 print(f"{name:8} {_AGENT_BLURB[name]:45} [{mark}]") 

1145 # Show discovered local models so the user sees what they can pick. 

1146 models = list_local_models(endpoint) 

1147 if models: 

1148 print(f"\nlocal models at {endpoint_disp}: {', '.join(models)}") 

1149 return 0 

1150 

1151 preset = PRESETS.get(ns.preset, {}) 

1152 templates = agent_templates() 

1153 # Seats `jury init` writes commented out, under a hint (#864). Only plain init 

1154 # fills it; the prompts and the wizard ask the operator instead. 

1155 commented: list[dict] = [] 

1156 

1157 def _detected_agents(): 

1158 # Never an opt-in-only agent (agy): "detected" is a default, not a choice. 

1159 return implicit_choices(n for n in KNOWN_AGENTS if available.get(n)) 

1160 

1161 def _selectable_agents(): 

1162 """Every known agent whose template scaffolds to a *valid* config here. 

1163 

1164 Three hosted templates point at real vendor hosts, and `config` refuses 

1165 a non-loopback endpoint unless `JURY_ALLOW_REMOTE_ENDPOINT` is set. A 

1166 preset that silently includes them writes a config `jury init` then 

1167 rejects — so `--preset thorough` failed outright rather than producing 

1168 something usable. They stay in `--list-agents` and remain selectable by 

1169 name; they are only excluded from "all" until the opt-in is present. 

1170 """ 

1171 if os.environ.get("JURY_ALLOW_REMOTE_ENDPOINT"): 1171 ↛ 1172line 1171 didn't jump to line 1172 because the condition on line 1171 was never true

1172 return implicit_choices(KNOWN_AGENTS) 

1173 needs_opt_in = set(agents_needing_remote_opt_in()) 

1174 return implicit_choices(n for n in KNOWN_AGENTS if n not in needs_opt_in) 

1175 

1176 def _resolve_preset_agents(spec): 

1177 if spec == "all": 

1178 return _selectable_agents() 

1179 if spec == "detected": 

1180 return _detected_agents() or _selectable_agents() 

1181 return list(spec) 

1182 

1183 # rounds / verify / early_stop: explicit flag > preset > built-in default. 

1184 rounds = ns.rounds if ns.rounds is not None else preset.get("rounds", 2) 

1185 verify = ns.verify if ns.verify is not None else preset.get("verify", True) 

1186 early_stop = preset.get("early_stop") 

1187 

1188 # Guided wizard (issue #231): opt-in via --wizard. A numbered-option flow 

1189 # where every question is skippable; only explicitly-chosen settings are 

1190 # written, so the file stays minimal. Explicit, so it skips the TTY test the 

1191 # prompts below use; a missing terminal was refused above instead (#865). 

1192 if ns.wizard: 

1193 kwargs = _init_wizard(available, local_endpoint=ns.local_endpoint) 

1194 kwargs["local_endpoint"] = ns.local_endpoint 

1195 if ns.local_model: 

1196 kwargs["local_model"] = ns.local_model 

1197 # Interactive only when neither --agents nor --preset was given and we're on a 

1198 # TTY (or --interactive, refused above without one). Presets/flags are 

1199 # non-interactive by design. With no stdin at all, plain init detects (#897). 

1200 elif not ns.agents and not ns.preset and (ns.interactive or _stdin_is_terminal()): 

1201 kwargs = _init_interactive(available, local_endpoint=ns.local_endpoint) 

1202 kwargs["local_endpoint"] = ns.local_endpoint 

1203 if ns.local_model: 

1204 kwargs["local_model"] = ns.local_model 

1205 else: 

1206 if ns.agents: 

1207 agents = [a.strip() for a in ns.agents.split(",") if a.strip()] 

1208 elif ns.preset: 

1209 agents = _resolve_preset_agents(preset["agents"]) 

1210 else: 

1211 agents = _detected_agents() 

1212 if not agents: 

1213 print( 

1214 "error: no agents detected and none specified; pass --agents " 

1215 "or --preset (e.g. --preset offline), or run interactively.", 

1216 file=sys.stderr, 

1217 ) 

1218 if available.get("agy"): 

1219 print(f"note: {AGY_OPT_IN_NOTE}.", file=sys.stderr) 

1220 return 2 

1221 # A local seat names a model its server lists, as the prompts and the wizard 

1222 # already did (#864); plain init wrote the template's model and then warned 

1223 # about it. Asked only when a local seat is chosen and no model was named. 

1224 # `local_model_listing`, not `list_local_models`: only a server that answered 

1225 # with no model gets its seat commented out, and a failed listing (None) 

1226 # leaves the seat as it was — no evidence is not evidence of a fault (#849). 

1227 local_model = ns.local_model 

1228 if not local_model and any(templates.get(a, {}).get("vendor") == "local" for a in agents): 

1229 agents, local_model, left_out = seat_local_agents(agents, local_model_listing(endpoint)) 

1230 if left_out and ns.chair in left_out: 

1231 # The operator named this chair. Writing it over a commented seat warns 

1232 # on every run, and picking another chair overrides them, so neither. 

1233 print( 

1234 f"error: the chair {ns.chair} is the local seat, but the server at " 

1235 f"{endpoint_disp} lists no model; pass --local-model <model> or choose " 

1236 "another --chair", 

1237 file=sys.stderr, 

1238 ) 

1239 return 2 

1240 if left_out: 

1241 commented = build_config(left_out, local_endpoint=ns.local_endpoint)["agent"] 

1242 print( 

1243 f"note: the local server at {endpoint_disp} lists no model, so " 

1244 f"{', '.join(repr(a) for a in left_out)} is written commented out; " 

1245 "pull one and uncomment it, or pass --local-model.", 

1246 file=sys.stderr, 

1247 ) 

1248 kwargs = { 

1249 "agents": agents, 

1250 "rounds": rounds, 

1251 "chair": ns.chair, 

1252 "verify": verify, 

1253 "early_stop": early_stop, 

1254 "local_model": local_model, 

1255 "local_endpoint": ns.local_endpoint, 

1256 } 

1257 

1258 try: 

1259 config = build_config(**kwargs) 

1260 except ValueError as exc: 

1261 print(f"error: {redact(str(exc))[0]}", file=sys.stderr) 

1262 return 2 

1263 

1264 # The scaffolded config must itself be valid (fail loudly if a template drifts). 

1265 try: 

1266 validate_config(config) 

1267 except ConfigError as exc: 

1268 print(f"error: generated config is invalid: {redact(str(exc))[0]}", file=sys.stderr) 

1269 return 2 

1270 

1271 out_path = Path(ns.output) 

1272 if out_path.exists() and not ns.force: 

1273 print( 

1274 f"error: {out_path} already exists; pass --force to overwrite.", 

1275 file=sys.stderr, 

1276 ) 

1277 return 2 

1278 

1279 out_path.write_text(render_toml(config, commented_agents=commented), encoding="utf-8") 

1280 # This config is the operator's own deliberate act, so record it as trusted: the 

1281 # ordinary `jury init` → `jury` path then never has to ask (#831). 

1282 configtrust.record_trust(out_path, configtrust.content_digest(out_path.read_bytes())) 

1283 chosen = ", ".join(a["name"] for a in config["agent"]) 

1284 print(f"Wrote {out_path} — panel: {chosen} · rounds: {config['jury']['rounds']}") 

1285 # Seated by name only, so the operator chose it — say what they chose. 

1286 if any(a["name"] in OPT_IN_AGENTS for a in config["agent"]): 

1287 print( 

1288 "warning: agy cannot be confined (it reads, writes and reaches the network " 

1289 "even with --sandbox); do not use it on untrusted diffs. A review run warns " 

1290 "about this seat, and `--strict` fails on it.", 

1291 file=sys.stderr, 

1292 ) 

1293 # A local seat whose server does not list its model is still a valid config, 

1294 # so it is written — but said now, not discovered by the first review as an 

1295 # `HTTP 404` (#849). `--list-models` already knew; init never asked. The 

1296 # doctor's own diagnosis is reused so the two cannot disagree about a seat. 

1297 from types import SimpleNamespace 

1298 

1299 from .doctor import _local_model_gap 

1300 

1301 for agent in config["agent"]: 

1302 fields = ("name", "vendor", "adapter", "endpoint", "model") 

1303 gap = _local_model_gap(SimpleNamespace(**{k: agent.get(k) for k in fields})) 

1304 if gap: 

1305 print(f"warning: agent '{agent['name']}': {gap[1]}", file=sys.stderr) 

1306 print(f"Next: jury --config-validate --config {out_path}") 

1307 print("Then: git diff main... | jury --diff-file -") 

1308 return 0 

1309 

1310 

1311def _config_read_error(exc: OSError, config_arg) -> str: 

1312 """The one line a config that exists but cannot be read is reported with (#893). 

1313 

1314 Every config load site catches ``ConfigError`` and ``FileNotFoundError``, but an 

1315 unreadable ``./jury.toml`` raises ``PermissionError`` and ``--config <directory>`` 

1316 raises ``IsADirectoryError`` (``PermissionError`` on Windows), and both escaped as a 

1317 traceback. The path comes from the exception, so an auto-discovered ``jury.toml`` is 

1318 named as well as an explicit one; ``strerror`` is the reason without the errno and 

1319 path ``str(exc)`` would repeat. 

1320 """ 

1321 path = exc.filename or config_arg or "jury.toml" 

1322 reason = exc.strerror or str(exc) 

1323 return redact(f"error: cannot read config {path}: {reason}")[0] 

1324 

1325 

1326def _config_source(config_arg) -> str: 

1327 """Human-readable source of the config the jury would load.""" 

1328 if config_arg: 

1329 return str(config_arg) 

1330 return "jury.toml" if Path("jury.toml").exists() else "(built-in defaults)" 

1331 

1332 

1333def _render_effective_config(cfg) -> str: 

1334 """Render the EFFECTIVE resolved config as a readable summary (config show).""" 

1335 on = lambda b: "on" if b else "off" # noqa: E731 

1336 lines = [] 

1337 lines.append( 

1338 f"[jury] rounds={cfg.rounds} chair={cfg.chair} verify={on(cfg.verify)} " 

1339 f"parallel={on(cfg.parallel)} timeout={cfg.timeout}s" 

1340 ) 

1341 adaptive = f"early_stop={on(cfg.early_stop)} max_rounds={cfg.effective_max_rounds}" 

1342 budget = ( 

1343 f"total_timeout={cfg.total_timeout or '—'} " 

1344 f"phase_timeout={cfg.phase_timeout or '—'} retries={cfg.retries}" 

1345 ) 

1346 lines.append( 

1347 f" {adaptive} · {budget} · seed={cfg.seed if cfg.seed is not None else '—'}" 

1348 ) 

1349 lines.append( 

1350 f"[jury.ci] fail_on={cfg.ci.fail_on} ignore_unverified={on(cfg.ci.ignore_unverified)}" 

1351 ) 

1352 lines.append( 

1353 f"[jury.context] mode={cfg.context.mode} redact_secrets={on(cfg.context.redact_secrets)}" 

1354 ) 

1355 d = cfg.diff 

1356 lines.append( 

1357 f"[jury.diff] max_bytes={d.max_bytes} chunk={on(d.chunk)} " 

1358 f"exclude_generated={on(d.exclude_generated)} " 

1359 f"exclude={d.exclude or '[]'} include={d.include or '[]'}" 

1360 ) 

1361 lines.append("agents:") 

1362 for a in cfg.agents: 

1363 flag = "" if a.enabled else " (disabled)" 

1364 target = a.endpoint if a.adapter_key == "local" else (a.command or "—") 

1365 model = f" model={a.model}" if a.model else "" 

1366 # The adapter is shown only when it is not the vendor's own (#705), so 

1367 # this line keeps its shape for every configuration that predates the key. 

1368 via = f" via {a.adapter_key}" if a.adapter_key != a.vendor else "" 

1369 lines.append(f" - {a.name} ({a.vendor}{via}) → {target}{model}{flag}") 

1370 return "\n".join(lines) 

1371 

1372 

1373def _run_config(rest: list[str]) -> int: 

1374 """Handle ``jury config show|path``.""" 

1375 from .config import ConfigError, load_config 

1376 

1377 sub = argparse.ArgumentParser(prog="jury config") 

1378 sub.add_argument("action", choices=["show", "path"]) 

1379 sub.add_argument("--config", help="path to jury.toml (default: ./jury.toml or built-in)") 

1380 ns = sub.parse_args(rest) 

1381 

1382 source = _config_source(ns.config) 

1383 if ns.action == "path": 

1384 print(source) 

1385 return 0 

1386 

1387 try: 

1388 cfg = load_config(ns.config, validate=True) 

1389 except (ConfigError, FileNotFoundError) as exc: 

1390 print(f"error: {redact(str(exc))[0]}", file=sys.stderr) 

1391 return 2 

1392 except OSError as exc: 

1393 print(_config_read_error(exc, ns.config), file=sys.stderr) 

1394 return 2 

1395 print(f"source: {source}") 

1396 print(_render_effective_config(cfg)) 

1397 return 0 

1398 

1399 

1400# Cap on a prompt file. An orchestrator's prompt is a few KB of instructions 

1401# plus, at most, a diff; this only bounds memory against a pathological file. 

1402_MAX_PROMPT_BYTES = 8 * 1024 * 1024 

1403 

1404 

1405def _read_prompt(path: str) -> tuple[str, str | None]: 

1406 """Read a prompt file (or stdin for ``-``). Returns ``(text, error)``.""" 

1407 if path == "-": 

1408 stream = sys.stdin.buffer if sys.stdin is not None else None 

1409 if stream is None: 

1410 return "", "cannot read the prompt from stdin: stdin is not available" 

1411 raw = stream.read(_MAX_PROMPT_BYTES + 1) 

1412 else: 

1413 target = Path(path) 

1414 if not target.is_file(): 

1415 return "", f"prompt file not found: {path}" 

1416 try: 

1417 raw = target.read_bytes()[: _MAX_PROMPT_BYTES + 1] 

1418 except OSError as exc: 

1419 return "", f"could not read prompt file '{path}': {redact(str(exc))[0]}" 

1420 if len(raw) > _MAX_PROMPT_BYTES: 

1421 return "", f"prompt exceeds the {_MAX_PROMPT_BYTES}-byte limit" 

1422 text = raw.decode("utf-8", errors="replace") 

1423 if not text.strip(): 

1424 return "", "empty prompt — nothing to send to the agent" 

1425 return text, None 

1426 

1427 

1428def _run_agent_parser() -> argparse.ArgumentParser: 

1429 """The ``jury run-agent`` flag surface (built separately so it can be tested).""" 

1430 from . import runagent 

1431 

1432 sub = argparse.ArgumentParser( 

1433 prog="jury run-agent", 

1434 description="Run ONE configured agent for one role and print a JSON " 

1435 "result with attribution. The integration point for an orchestrator " 

1436 "(keel's `keel delegate run`) or a CI script that needs a single " 

1437 "agent rather than the whole panel.", 

1438 ) 

1439 sub.add_argument( 

1440 "--agent", 

1441 help="agent to run: a [[agent]] name from jury.toml, or a built-in " 

1442 f"vendor ({', '.join(runagent.BUILTIN_AGENTS)}); append ':<model>' to " 

1443 "override the model (e.g. codex:gpt-5.2)", 

1444 ) 

1445 sub.add_argument( 

1446 "--role", 

1447 help="what the agent is being asked to do: " 

1448 f"{'|'.join(runagent.ROLES)}. " 

1449 f"{'/'.join(runagent.READ_ONLY_ROLES)} always run read-only; " 

1450 f"{'/'.join(sorted(runagent.WRITE_ROLES))} need --allow-write", 

1451 ) 

1452 sub.add_argument("--prompt-file", help="path to the prompt to send, or '-' for stdin") 

1453 sub.add_argument( 

1454 "--cwd", 

1455 help="directory a write role (implement/fix) runs in (default: the current one); " 

1456 "read-only roles on claude/codex/agy start in an empty temporary directory", 

1457 ) 

1458 sub.add_argument( 

1459 "--timeout", 

1460 type=int, 

1461 help="seconds before the AGENT is killed (default: the agent's configured " 

1462 "timeout). This bounds the agent, never the wait — see --wait-timeout", 

1463 ) 

1464 sub.add_argument( 

1465 "--effort", 

1466 choices=list(EFFORT_LEVELS), 

1467 help="reasoning effort for vendors that support one (see `jury --doctor --json`)", 

1468 ) 

1469 sub.add_argument( 

1470 "--allow-write", 

1471 action="store_true", 

1472 help="grant the vendor's write/tool mode; required by the implement and " 

1473 "fix roles, ignored by the read-only ones", 

1474 ) 

1475 sub.add_argument( 

1476 "--format", 

1477 choices=["json", "text"], 

1478 default="json", 

1479 help="json: the full result document (default); text: only the agent's text", 

1480 ) 

1481 sub.add_argument( 

1482 "--detach", 

1483 action="store_true", 

1484 help="start the run in the background and print its run id immediately", 

1485 ) 

1486 sub.add_argument( 

1487 "--run-id", 

1488 help="id for a detached run: letters, digits, '.', '_' or '-', max 64 " 

1489 "characters (default: a random one). Only meaningful with --detach", 

1490 ) 

1491 sub.add_argument("--wait", metavar="RUN_ID", help="block until a detached run finishes") 

1492 sub.add_argument( 

1493 "--wait-timeout", 

1494 type=int, 

1495 metavar="SECONDS", 

1496 help="seconds --wait will block before giving up (default: the run's own " 

1497 f"timeout + {runagent.WAIT_GRACE_S}s, else {runagent.DEFAULT_WAIT_TIMEOUT_S}s)", 

1498 ) 

1499 sub.add_argument("--status", action="store_true", help="list detached runs and exit") 

1500 sub.add_argument("--config", help="path to jury.toml (default: ./jury.toml or built-in)") 

1501 sub.add_argument( 

1502 "--cache-dir", 

1503 help="where detached-run state lives (default: $JURY_CACHE_DIR or ~/.cache/ai-jury)", 

1504 ) 

1505 sub.add_argument( 

1506 "--mock", action="store_true", help="run the offline mock adapter instead of a real agent" 

1507 ) 

1508 sub.add_argument( 

1509 "--strict", 

1510 action="store_true", 

1511 help="refuse (exit 2) a read-only role whose seat draws a least-privilege " 

1512 "warning, as a panel run with --strict does", 

1513 ) 

1514 # The detached child re-enters this same command; the flag only tells it to 

1515 # record its result in the run's state file. Hidden: it is not a surface 

1516 # anyone should call directly. 

1517 sub.add_argument("--_child", dest="child", action="store_true", help=argparse.SUPPRESS) 

1518 return sub 

1519 

1520 

1521def _child_argv(ns, run_id: str, python: str) -> list[str]: 

1522 """The argv of the background child a ``--detach`` run spawns (pure).""" 

1523 argv = [ 

1524 python, 

1525 "-m", 

1526 "ai_jury", 

1527 "run-agent", 

1528 "--agent", 

1529 ns.agent, 

1530 "--role", 

1531 ns.role, 

1532 "--prompt-file", 

1533 str(Path(ns.prompt_file).resolve()), 

1534 "--run-id", 

1535 run_id, 

1536 "--_child", 

1537 ] 

1538 if ns.cwd: 

1539 argv += ["--cwd", str(Path(ns.cwd).resolve())] 

1540 if ns.timeout is not None: 

1541 argv += ["--timeout", str(ns.timeout)] 

1542 if ns.effort: 

1543 argv += ["--effort", ns.effort] 

1544 if ns.allow_write: 

1545 argv.append("--allow-write") 

1546 if ns.config: 

1547 argv += ["--config", str(Path(ns.config).resolve())] 

1548 if ns.cache_dir: 

1549 argv += ["--cache-dir", str(Path(ns.cache_dir).resolve())] 

1550 if ns.mock: 

1551 argv.append("--mock") 

1552 if ns.strict: 

1553 argv.append("--strict") 

1554 return argv 

1555 

1556 

1557def _run_run_agent(rest: list[str], spawn=None, sleep=None, clock=None) -> int: 

1558 """Handle ``jury run-agent`` (issue #661): one agent, one role, one JSON result. 

1559 

1560 Exit codes: ``0`` the agent ran and produced output, ``1`` it ran and 

1561 failed (the JSON says why — same fail-soft vocabulary as a panel run), 

1562 ``2`` the request itself was refused (bad role, missing write grant, 

1563 unknown agent, unreadable prompt). 

1564 """ 

1565 import dataclasses 

1566 import time 

1567 

1568 from . import runagent 

1569 from .adapters import effort_args, make_adapter 

1570 

1571 ns = _run_agent_parser().parse_args(rest) 

1572 

1573 # Validate every id the caller supplies BEFORE it can reach a path. The 

1574 # library refuses an unsafe one on its own (runagent.run_path), so this is 

1575 # about the error the operator sees, not about whether the write is safe. 

1576 for flag, value in (("--run-id", ns.run_id), ("--wait", ns.wait)): 

1577 if value is not None: 

1578 problem = runagent.check_run_id(value) 

1579 if problem: 

1580 print(f"error: {flag}: {problem}", file=sys.stderr) 

1581 return 2 

1582 if ns.run_id and not (ns.detach or ns.child): 

1583 # It named nothing and recorded nothing — silently ignoring it would let 

1584 # a script believe it had a handle it could later --wait on. 

1585 print( 

1586 "error: --run-id only applies to a detached run; add --detach, or " 

1587 "drop --run-id to run in the foreground", 

1588 file=sys.stderr, 

1589 ) 

1590 return 2 

1591 

1592 if ns.status: 

1593 print(json.dumps({"runs": runagent.list_runs(ns.cache_dir)}, indent=2)) 

1594 return 0 

1595 

1596 if ns.wait: 

1597 if ns.wait_timeout is not None and ns.wait_timeout <= 0: 

1598 print("error: --wait-timeout must be a positive number of seconds", file=sys.stderr) 

1599 return 2 

1600 # `--detach` reserves the id by writing the state file BEFORE it returns, 

1601 # so a run with no state file was never started — a typo, or the wrong 

1602 # --cache-dir. Say so now instead of blocking until a deadline expires. 

1603 known = runagent.read_state(ns.wait, ns.cache_dir) 

1604 if known is None: 

1605 print(f"error: no such run '{ns.wait}'", file=sys.stderr) 

1606 return 2 

1607 # The deadline comes from what the run was actually given (its own 

1608 # timeout plus head-room to write its state file), so a long agent run 

1609 # is not cut short and a dead one is not waited on forever. 

1610 deadline = ns.wait_timeout 

1611 if deadline is None: 

1612 deadline = runagent.default_wait_timeout(known) 

1613 # When liveness cannot be determined, a run that has already died can 

1614 # only be noticed when the deadline expires. Say so once, so a long 

1615 # silence reads as a known limitation rather than a hung command. The 

1616 # opening is cause-neutral and the reason names which case it is — 

1617 # "this platform cannot probe" is false on POSIX for a run that simply 

1618 # has not been claimed yet. The deadline always applies regardless, so 

1619 # the wait is bounded either way. 

1620 unknown = runagent.liveness_unknown_reason(known) 

1621 if unknown: 

1622 print( 

1623 f"warning: cannot tell whether run '{ns.wait}' is still alive — " 

1624 f"{unknown}; waiting up to {deadline:g}s for it to report", 

1625 file=sys.stderr, 

1626 ) 

1627 state, timed_out = runagent.wait_for_run( 

1628 ns.wait, 

1629 cache_dir=ns.cache_dir, 

1630 timeout=deadline, 

1631 sleep=sleep or time.sleep, 

1632 clock=clock or time.monotonic, 

1633 ) 

1634 if timed_out: 

1635 print( 

1636 f"error: run '{ns.wait}' did not finish within {deadline:g}s", 

1637 file=sys.stderr, 

1638 ) 

1639 return 2 

1640 if state is None: 

1641 # The file vanished while we waited (a `jury cache clear`, a manual 

1642 # rm). Distinguished from "never existed", which returned above. 

1643 print(f"error: run '{ns.wait}' disappeared while waiting", file=sys.stderr) 

1644 return 2 

1645 if state.get("status") == runagent.STATUS_LOST: 

1646 # Returned as soon as the pid was seen gone, not at the deadline: 

1647 # there is no result coming, and blocking for one is just delay. 

1648 print( 

1649 f"error: run '{ns.wait}' was lost — its process is gone and it " 

1650 f"never recorded a result (see {runagent.output_path(ns.wait, ns.cache_dir)})", 

1651 file=sys.stderr, 

1652 ) 

1653 print(json.dumps(state, indent=2)) 

1654 return 0 if state.get("ok") else 1 

1655 

1656 missing = [ 

1657 flag 

1658 for flag, value in ( 

1659 ("--agent", ns.agent), 

1660 ("--role", ns.role), 

1661 ("--prompt-file", ns.prompt_file), 

1662 ) 

1663 if not value 

1664 ] 

1665 if missing: 

1666 print( 

1667 f"error: {', '.join(missing)} required (or use --wait <run-id> / --status)", 

1668 file=sys.stderr, 

1669 ) 

1670 return 2 

1671 

1672 # Role policy first: an implement run without --allow-write must be refused 

1673 # before a prompt is read, a config is loaded, or a child is spawned. 

1674 policy = runagent.role_policy(ns.role, ns.allow_write) 

1675 if policy.refusal: 

1676 print(f"error: {policy.refusal}", file=sys.stderr) 

1677 return 2 

1678 if policy.warning: 

1679 print(f"warning: {policy.warning}", file=sys.stderr) 

1680 

1681 if ns.cwd and not Path(ns.cwd).is_dir(): 

1682 print(f"error: --cwd is not a directory: {ns.cwd}", file=sys.stderr) 

1683 return 2 

1684 if ns.cwd and not policy.write: 

1685 # Said, not silently dropped: a read-only role on a native CLI starts in an 

1686 # empty directory, so a checkout's project settings cannot reach it. 

1687 print( 

1688 f"note: --cwd applies to write roles; the read-only role '{policy.role}' " 

1689 f"starts in an empty temporary directory (claude/codex/agy).", 

1690 file=sys.stderr, 

1691 ) 

1692 

1693 # Validated as the panel validates it (#903): an `endpoint` on a CLI seat, a 

1694 # `command` path rule, an unknown adapter — each refused here as it is by a 

1695 # review, `--config-validate`, `jury config` and `--doctor`, rather than 

1696 # reaching the spawner as a config the rest of the tool rejects. 

1697 try: 

1698 config = load_config(ns.config, validate=True) 

1699 except (ConfigError, FileNotFoundError) as exc: 

1700 print(redact(f"error: {exc}")[0], file=sys.stderr) 

1701 return 2 

1702 except OSError as exc: 

1703 print(_config_read_error(exc, ns.config), file=sys.stderr) 

1704 return 2 

1705 

1706 # run-agent runs a config-defined command too, and a discovered config can even shadow 

1707 # a built-in name (`--agent claude` with `command = "sh"`), so it needs the same trust 

1708 # gate as a review (#831). 

1709 try: 

1710 configtrust.enforce(ns.config, config, mock=ns.mock) 

1711 except configtrust.ConfigTrustError as exc: 

1712 print(f"error: {exc}", file=sys.stderr) 

1713 return 2 

1714 

1715 spec, error = runagent.resolve_agent(config, ns.agent) 

1716 if error is not None: 

1717 print(f"error: {error}", file=sys.stderr) 

1718 return 2 

1719 if ns.timeout is not None: 

1720 if ns.timeout <= 0: 

1721 print("error: --timeout must be a positive number of seconds", file=sys.stderr) 

1722 return 2 

1723 spec = dataclasses.replace(spec, timeout=ns.timeout) 

1724 if ns.effort: 

1725 spec = dataclasses.replace(spec, effort=ns.effort) 

1726 warning = effort_args(spec.adapter_key, ns.effort, spec.model).warning 

1727 if warning: 

1728 print(f"warning: {warning}", file=sys.stderr) 

1729 

1730 # The panel's least-privilege audit, for this one seat (#903). A read-only 

1731 # role is spawned with the argv a panel review uses, so it is audited the 

1732 # same way, and `--strict` refuses it the same way. A write role is not: it 

1733 # asked for write access with `--allow-write`, and the audit describes the 

1734 # read-only argv, which is not the one that role runs. Before `--detach`, so a 

1735 # refused seat never starts a background run. 

1736 if not policy.write: 

1737 from .privilege import audit_agent 

1738 

1739 privilege_warnings = audit_agent(spec) 

1740 for w in privilege_warnings: 

1741 print(f"warning: least-privilege: {w}", file=sys.stderr) 

1742 if ns.strict and privilege_warnings: 

1743 print( 

1744 "error: least-privilege check failed (--strict): " + "; ".join(privilege_warnings), 

1745 file=sys.stderr, 

1746 ) 

1747 return 2 

1748 

1749 if ns.detach: 

1750 return _detach_run_agent(ns, spec, policy, spawn=spawn) 

1751 

1752 prompt, error = _read_prompt(ns.prompt_file) 

1753 if error is not None: 

1754 print(f"error: {error}", file=sys.stderr) 

1755 return 2 

1756 

1757 # The adapter name is also checked where it is used (#708). Unreachable today: 

1758 # validation above refuses an unknown adapter (#903), and a built-in agent 

1759 # names none. Kept so a future path that skips validation fails with the 

1760 # config error, not a traceback. 

1761 try: 

1762 adapter = make_adapter(spec, mock=ns.mock) 

1763 except ConfigError as exc: # pragma: no cover - see above 

1764 print(redact(f"error: {exc}")[0], file=sys.stderr) 

1765 return 2 

1766 print( 

1767 f"run-agent: {spec.name} ({spec.vendor}) role={policy.role} " 

1768 f"{'write' if policy.write else 'read-only'}", 

1769 file=sys.stderr, 

1770 ) 

1771 started = time.time() 

1772 record = _child_recorder(ns, spec, policy, started) if ns.child and ns.run_id else None 

1773 if record is not None: 

1774 # Claim the run with our OWN pid before doing any work. Every write to 

1775 # this state file now comes from this process, in order, so nothing can 

1776 # overwrite the terminal document the way the parent's post-spawn write 

1777 # used to. Until this lands the run has no pid, which reads as 

1778 # "liveness unknown" — reported as running, never as lost. 

1779 record(runagent.initial_state(ns.run_id, spec, policy.role, started), terminal=False) 

1780 try: 

1781 if ns.cwd: 

1782 with contextlib.chdir(ns.cwd): 

1783 result = adapter.run(prompt, phase=policy.role, role_policy=policy) 

1784 else: 

1785 result = adapter.run(prompt, phase=policy.role, role_policy=policy) 

1786 # The second (and last) place a seat is invoked, so it records the id it 

1787 # sent exactly as the orchestrator's does (#709). 

1788 result.model = adapter.resolved_model() 

1789 document = runagent.result_dict(spec, policy.role, result) 

1790 except BaseException as exc: 

1791 # A detached child that dies without writing leaves its run at 

1792 # "running" forever, and a `--wait` on it can only time out. Adapters 

1793 # are fail-soft, so reaching here means something unexpected — a 

1794 # KeyboardInterrupt, a bug — and the run must still end in a terminal 

1795 # state that says so. Re-raised: this records the death, it does not 

1796 # swallow it. 

1797 if record is not None: 1797 ↛ 1799line 1797 didn't jump to line 1799 because the condition on line 1797 was always true

1798 record(_crash_document(spec, policy.role, started, exc)) 

1799 raise 

1800 

1801 if record is not None: 

1802 record(document) 

1803 if ns.format == "text": 

1804 print(document["text"]) 

1805 else: 

1806 print(json.dumps(document, indent=2)) 

1807 return 0 if document["ok"] else 1 

1808 

1809 

1810def _crash_document(spec, role: str, started: float, exc: BaseException) -> dict: 

1811 """A result document for a child that died before producing one (#661).""" 

1812 import time 

1813 

1814 from . import runagent 

1815 from .adapters import ERR_UNKNOWN, AgentResult 

1816 

1817 reason = f"{type(exc).__name__}: {redact(str(exc))[0]}" 

1818 return runagent.result_dict( 

1819 spec, 

1820 role, 

1821 AgentResult( 

1822 spec.name, 

1823 spec.vendor, 

1824 False, 

1825 "", 

1826 max(0.0, time.time() - started), 

1827 f"run-agent exited before the agent finished: {reason}", 

1828 error_code=ERR_UNKNOWN, 

1829 ), 

1830 ) 

1831 

1832 

1833def _child_recorder(ns, spec, policy, started: float): 

1834 """A callable that writes a detached child's state file (issue #661). 

1835 

1836 Called twice: once at the top with ``terminal=False`` to claim the run with 

1837 this process's pid, and once at the end — from the success path or the 

1838 crash handler — with the result document. Both writes come from the child, 

1839 which is what keeps them ordered: the parent writes only before the spawn, 

1840 so no stale copy can land on top of the finished document. 

1841 

1842 Fail-soft: a child that cannot write its state must still print its 

1843 document to the log, since the log is then the only record. 

1844 """ 

1845 from . import runagent 

1846 

1847 del spec, policy 

1848 

1849 def record(document: dict, terminal: bool = True) -> None: 

1850 state = dict(document) 

1851 state.update( 

1852 { 

1853 "run_id": ns.run_id, 

1854 "status": runagent.STATUS_DONE if terminal else runagent.STATUS_RUNNING, 

1855 # `started_at` travels with the pid so a reader can see how old 

1856 # the claim is — see runagent.pid_alive on why a pid alone is 

1857 # not proof of identity. 

1858 "started_at": started, 

1859 "pid": os.getpid(), 

1860 } 

1861 ) 

1862 try: 

1863 runagent.write_state(ns.run_id, state, ns.cache_dir) 

1864 except (OSError, ValueError) as exc: 

1865 print(f"warning: could not record run state: {redact(str(exc))[0]}", file=sys.stderr) 

1866 

1867 return record 

1868 

1869 

1870def _detach_run_agent(ns, spec, policy, spawn=None) -> int: 

1871 """Start a ``run-agent`` call in the background and print its run id.""" 

1872 import time 

1873 

1874 from . import runagent 

1875 

1876 if ns.prompt_file == "-": 

1877 print( 

1878 "error: --detach needs a prompt FILE; the background run has no stdin to read from", 

1879 file=sys.stderr, 

1880 ) 

1881 return 2 

1882 if not Path(ns.prompt_file).is_file(): 

1883 print(f"error: prompt file not found: {ns.prompt_file}", file=sys.stderr) 

1884 return 2 

1885 

1886 # An operator-supplied id was already validated at parse time, and a 

1887 # generated one is valid by construction — `runagent.run_path` refuses an 

1888 # unsafe id regardless, so there is nothing left to re-check here. 

1889 run_id = ns.run_id or runagent.new_run_id() 

1890 if runagent.state_path(run_id, ns.cache_dir).exists(): 

1891 print( 

1892 f"error: run '{run_id}' already exists; choose another --run-id", 

1893 file=sys.stderr, 

1894 ) 

1895 return 2 

1896 

1897 state = runagent.initial_state(run_id, spec, policy.role, time.time()) 

1898 try: 

1899 # Reserve the id BEFORE spawning, so a duplicate is refused rather than 

1900 # discovered by two children racing for the same state file. 

1901 runagent.write_state(run_id, state, ns.cache_dir) 

1902 out_path = runagent.output_path(run_id, ns.cache_dir) 

1903 argv = _child_argv(ns, run_id, sys.executable) 

1904 launch = spawn or _spawn_detached 

1905 pid = launch(argv, out_path) 

1906 except (OSError, ValueError) as exc: 

1907 print(f"error: could not start the detached run: {redact(str(exc))[0]}", file=sys.stderr) 

1908 return 2 

1909 

1910 # The parent writes the state file EXACTLY ONCE, before the spawn. It used 

1911 # to write a second time afterwards to record the pid, from the stale dict 

1912 # it still held — and a child that finished during launch had its terminal 

1913 # document overwritten with `status: running`, losing the answer entirely. 

1914 # The child records its own pid instead (see `_child_running_state`), so 

1915 # every write after this one comes from the child, in order. 

1916 print( 

1917 json.dumps( 

1918 { 

1919 # The pid is the parent's own knowledge, reported for the 

1920 # operator's benefit; it is deliberately NOT written to the 

1921 # state file from here. 

1922 **runagent.run_summary(state), 

1923 "pid": pid if isinstance(pid, int) else None, 

1924 "state_file": str(runagent.state_path(run_id, ns.cache_dir)), 

1925 "output_file": str(runagent.output_path(run_id, ns.cache_dir)), 

1926 }, 

1927 indent=2, 

1928 ) 

1929 ) 

1930 return 0 

1931 

1932 

1933#: Deliberately-unreaped background children (see :func:`_spawn_detached`). The 

1934#: parent is a short-lived CLI invocation, so this holds at most one entry per 

1935#: `--detach` in a process that is about to exit. 

1936_DETACHED_CHILDREN: list = [] 

1937 

1938 

1939def _spawn_detached(argv: list[str], out_path: Path) -> int: 

1940 """Start the background child, streaming its stdout+stderr to ``out_path``. 

1941 

1942 Its own session, so the child outlives the shell that launched it — the 

1943 whole point of ``--detach`` for an orchestrator that dispatches work and 

1944 comes back for it later. Returns the child's pid, which the caller records 

1945 so ``--status`` can later tell a live run from an abandoned one. 

1946 """ 

1947 import subprocess 

1948 

1949 kwargs: dict = {} 

1950 if hasattr(os, "setsid"): 1950 ↛ 1954line 1950 didn't jump to line 1954 because the condition on line 1950 was always true

1951 kwargs["start_new_session"] = True 

1952 # 0600 like the state file: the log holds the agent's full output, which is 

1953 # derived from the prompt and is no less sensitive than the diff a review sees. 

1954 fd = os.open(out_path, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o600) 

1955 with os.fdopen(fd, "wb") as out: 

1956 proc = subprocess.Popen( # noqa: S603 - argv is built here, never shell-parsed 

1957 argv, 

1958 stdin=subprocess.DEVNULL, 

1959 stdout=out, 

1960 stderr=subprocess.STDOUT, 

1961 **kwargs, 

1962 ) 

1963 # Not waiting for this child is the whole point, but a dropped Popen warns 

1964 # ("subprocess N is still running") when it is collected, and the parent 

1965 # exits moments later so there is nothing to reap. Hold the reference 

1966 # instead of lying about the return code or reaching into private state. 

1967 _DETACHED_CHILDREN.append(proc) 

1968 return proc.pid 

1969 

1970 

1971def _run_replay(rest: list[str]) -> int: 

1972 """Handle ``jury replay <outcome.json>`` (issue #449). 

1973 

1974 Replays a saved run in the deliberation theater — or, off a TTY / without 

1975 ``--theater``, as the same plain step stream ``--live`` prints. Pure 

1976 presentation: no orchestration, no network, no agents. 

1977 """ 

1978 from .replay import ReplayError, load_outcome, replay_events, replay_into 

1979 

1980 sub = argparse.ArgumentParser( 

1981 prog="jury replay", 

1982 description="Replay a saved jury outcome (a result-cache entry or a " 

1983 "serialized outcome dict) in the deliberation theater. No agents run.", 

1984 ) 

1985 sub.add_argument( 

1986 "outcome", 

1987 help="path to a saved outcome JSON (cache entry or outcome dict)", 

1988 ) 

1989 sub.add_argument( 

1990 "--theater", 

1991 action="store_true", 

1992 help="replay in the animated deliberation scene (needs a wide TTY; " 

1993 "falls back to plain transcript lines otherwise)", 

1994 ) 

1995 sub.add_argument( 

1996 "--theater-style", 

1997 choices=["flat", "pixel"], 

1998 default="flat", 

1999 help="--theater scene style: 'flat' (ANSI line scene, default) or " 

2000 "'pixel' (half-block pixel-art room)", 

2001 ) 

2002 sub.add_argument( 

2003 "--decision", 

2004 choices=["chair", "vote"], 

2005 default="chair", 

2006 help="finale mode: 'chair' shows the stored synthesis verdict (default); " 

2007 "'vote' re-tallies the panel ballots for the vote finale", 

2008 ) 

2009 sub.add_argument( 

2010 "--mode", 

2011 choices=["code", "issue"], 

2012 default="code", 

2013 help="vote vocabulary for --decision vote (the serialized outcome does " 

2014 "not record the run mode): 'code' (APPROVE/COMMENT/REQUEST CHANGES, " 

2015 "default) or 'issue' (READY/UNCLEAR/NEEDS-INFO)", 

2016 ) 

2017 ns = sub.parse_args(rest) 

2018 

2019 try: 

2020 outcome = load_outcome(Path(ns.outcome)) 

2021 except ReplayError as exc: 

2022 print(f"error: {redact(str(exc))[0]}", file=sys.stderr) 

2023 return 2 

2024 

2025 # Panel-vote finale (mirrors the live path): re-tally from the stored 

2026 # groups/reviews — deterministic, no agents involved. 

2027 vote = None 

2028 if ns.decision == "vote": 

2029 from .voting import is_abstention, tally_votes 

2030 

2031 voters = [ 

2032 r.agent for r in outcome.reviews if r.ok and not is_abstention(getattr(r, "output", "")) 

2033 ] 

2034 vote = tally_votes(outcome.groups, voters, mode=ns.mode) 

2035 

2036 # Same TTY gate as the live path: the scene needs a wide TTY, otherwise 

2037 # degrade to the plain --live step stream. 

2038 court = None 

2039 if ns.theater: 

2040 from . import theater as _theater 

2041 

2042 if _theater.supports_scene(sys.stdout): 

2043 seats: dict[str, str] = {} 

2044 for r in outcome.reviews: 

2045 seats.setdefault(r.agent, r.vendor) 

2046 court = _theater.Courtroom( 

2047 list(seats.items()), 

2048 outcome.chair or "chair", 

2049 case=Path(ns.outcome).name, 

2050 decision=ns.decision, 

2051 style=ns.theater_style, 

2052 ) 

2053 

2054 if court is not None: 

2055 replay_into(court, outcome, vote=vote) 

2056 else: 

2057 for kind, result, round_no in replay_events(outcome): 

2058 title, body = render_live_step(kind, result, round_no) 

2059 print(f"## {title}\n\n{body}\n", flush=True) 

2060 if vote is not None: 

2061 # The vote finale must survive the transcript fallback too (review 

2062 # finding: --decision vote was computed then silently dropped here). 

2063 print("## Panel vote\n", flush=True) 

2064 for ballot in vote.ballots: 

2065 print(f"- {ballot.reviewer}: {ballot.vote} ({ballot.reason})", flush=True) 

2066 print(f"\nVerdict: {vote.verdict}\n", flush=True) 

2067 return 0 

2068 

2069 

2070_PROGRESS_PREFIXES = ( 

2071 "round ", 

2072 "reviewing chunk", 

2073 "verification", 

2074 "synthesis", 

2075 "diff size", 

2076 "early stop", 

2077 "auto-depth", 

2078) 

2079 

2080 

2081def _is_progress_milestone(msg: str) -> bool: 

2082 """Whether a log line is a coarse milestone worth a sticky-comment update.""" 

2083 return msg.startswith(_PROGRESS_PREFIXES) 

2084 

2085 

2086#: Every CLI flag that writes a `[jury]` setting `validate_config` puts a bound 

2087#: on, as `(flag, argparse dest, setting)` (issue #748). The settings are keys 

2088#: of `config._NUMERIC_BOUNDS`, so the rule and the message are the config 

2089#: path's own; only the `where` differs, naming the flag the operator typed 

2090#: rather than a key they may never have written. 

2091#: 

2092#: The flags NOT here were checked against the same table and belong nowhere 

2093#: else: `--seed` has no bound to break (a malformed `[jury] seed` is read as 

2094#: "no seed" on purpose, not as an error); `--chunk`, `--early-stop`, 

2095#: `--verify`, `--redact`, `--auto` and `--hints` are booleans; `--effort`, 

2096#: `--decision`, `--context-mode` and `--format` are argparse `choices`, which 

2097#: already refuse a value outside the vocabulary; and `--fail-on` has shared its 

2098#: rule with the validator since #718. 

2099_BOUNDED_FLAGS = ( 

2100 ("--rounds", "rounds", "rounds"), 

2101 ("--max-rounds", "max_rounds", "max_rounds"), 

2102 ("--total-timeout", "total_timeout", "total_timeout"), 

2103 ("--phase-timeout", "phase_timeout", "phase_timeout"), 

2104 ("--retries", "retries", "retries"), 

2105 ("--max-diff-bytes", "max_diff_bytes", "diff.max_bytes"), 

2106 # The fail-closed guards (docs audit 2026-09-29): they used to be clamped to 

2107 # >= 0, so `--min-vendors -5` read as `0` — the documented opt-out — and the 

2108 # cross-vendor guard was off with exit 0. `--no-min-vendors` writes 0, which 

2109 # passes. 

2110 ("--min-vendors", "min_vendors", "ci.min_vendors"), 

2111 ("--min-reviews", "min_reviews", "ci.min_reviews"), 

2112) 

2113 

2114 

2115def _override_bound_error(args) -> str | None: 

2116 """The first CLI override breaking a bound ``validate_config`` enforces. 

2117 

2118 ``None`` is "the flag was not passed" and leaves the config value — which 

2119 the validator has already checked — standing; it is never a value to range 

2120 check, so an unset flag can never be mistaken for a zero. 

2121 """ 

2122 for flag, dest, setting in _BOUNDED_FLAGS: 

2123 value = getattr(args, dest, None) 

2124 if value is None: 

2125 continue 

2126 message = bound_error(setting, value, where=flag) 

2127 if message: 

2128 return message 

2129 return None 

2130 

2131 

2132def _maybe_add_local_fallback(config, args, log): 

2133 """Append a local agent when nothing else can run, offline (issue: zero-config). 

2134 

2135 Only fires in the safe "fresh user" case: no explicit `--config`, no 

2136 `./jury.toml`, not `--mock`, none of the configured agents are available, 

2137 and a local OpenAI-compatible server is reachable with at least one model. 

2138 Mutates ``config`` in place and points the chair at the local agent. 

2139 

2140 Returns the seat it added, or ``None``. The built-in seats stay in the config, 

2141 so the report still lists them as not found, but none of them can review: the 

2142 panel this run can form is the local seat alone, and the caller scopes the 

2143 default cross-vendor guard to that (#863). The decision is 

2144 :func:`scaffold.zero_config_local_seat`, which ``jury --doctor`` also asks. 

2145 """ 

2146 from .adapters import list_local_models, make_adapter 

2147 from .scaffold import zero_config_local_seat 

2148 

2149 seat = zero_config_local_seat( 

2150 args.config, 

2151 args.mock, 

2152 Path("jury.toml").exists(), 

2153 lambda: any(make_adapter(s).available() for s in config.enabled_agents), 

2154 list_local_models, 

2155 ) 

2156 if seat is None: 

2157 return None 

2158 model = seat.model 

2159 config.agents.append(seat) 

2160 config.chair = "local" 

2161 named = getattr(args, "min_vendors", None) 

2162 if named is None: 

2163 guard = "so the default cross-vendor guard (min_vendors) does not apply to it" 

2164 elif named <= 0: 

2165 guard = "with the cross-vendor guard off (--no-min-vendors)" 

2166 else: 

2167 guard = f"held to the --min-vendors {named} you named" 

2168 log( 

2169 f"no agent CLIs found; using local model '{model}' (offline, $0) as a single-vendor panel, {guard}" 

2170 ) 

2171 # An agy-only machine lands here too: say why its one CLI sat out. 

2172 from .config import agy_opt_in_hint 

2173 

2174 hint = agy_opt_in_hint((s.adapter_key for s in config.agents), shutil.which) 

2175 if hint: 

2176 log(f"note: {hint}") 

2177 return seat 

2178 

2179 

2180def _force_utf8_output() -> None: 

2181 """Ensure stdout/stderr can emit the report's Unicode (emoji, arrows). 

2182 

2183 On Windows the console defaults to a legacy code page (e.g. cp1252) that 

2184 can't encode the report's `🏛️`/`⇄` characters, so `print(report)` raises 

2185 `UnicodeEncodeError`. Reconfigure the real streams to UTF-8 when possible; 

2186 `reconfigure` is absent on replaced streams (tests' StringIO, some pipes), 

2187 so this is a best-effort no-op there. 

2188 """ 

2189 for stream in (sys.stdout, sys.stderr): 

2190 reconfigure = getattr(stream, "reconfigure", None) 

2191 if reconfigure is not None: 

2192 with contextlib.suppress(ValueError, OSError): 

2193 reconfigure(encoding="utf-8") 

2194 

2195 

2196_OVERVIEW = """\ 

2197🏛️ ai-jury — a cross-vendor multi-agent review jury. 

2198 

2199It runs several coding-agent CLIs (Claude, Codex, Antigravity) plus an optional 

2200local model over the same diff, PR, or issue; they cross-examine and verify each 

2201other, and a chair (or a panel vote) synthesizes one verdict. 

2202 

2203Common commands: 

2204 jury init --wizard guided setup — writes a jury.toml (skippable) 

2205 jury --pr 123 review a pull request 

2206 jury --issue 42 review an issue for completeness 

2207 git diff | jury --diff-file - review the current branch's diff 

2208 jury examples more example commands 

2209 jury guide a short end-to-end walkthrough 

2210 jury --help every option 

2211 

2212Docs: https://github.com/berkayturanci/ai-jury""" 

2213 

2214_EXAMPLES = """\ 

2215ai-jury — example commands 

2216 

2217Setup 

2218 jury init --wizard guided setup (writes jury.toml) 

2219 jury init --preset thorough non-interactive preset 

2220 jury config show print the effective, resolved config 

2221 jury --doctor check which agents/CLIs are available 

2222 

2223Review 

2224 jury --pr 123 review a pull request 

2225 jury --issue 42 review an issue for completeness 

2226 git diff | jury --diff-file - review the current branch's diff 

2227 jury --diff-file changes.patch review a saved patch 

2228 jury --pr 123 --verbose full play-by-play (rounds + transcript) 

2229 

2230Decide & gate 

2231 jury --pr 123 --decision vote verdict by panel vote (not a single chair) 

2232 jury --pr 123 --ci exit non-zero on a blocking finding (CI gate) 

2233 

2234Post results back to GitHub 

2235 jury --pr 123 --post-summary post one rollup comment 

2236 jury --pr 123 --post-inline post line-level review comments 

2237 jury --issue 42 --post-summary post the triage verdict on the issue 

2238 

2239Run `jury guide` for a walkthrough, or `jury --help` for every option.""" 

2240 

2241_GUIDE = """\ 

2242ai-jury — a short walkthrough 

2243 

22441. Install the agent CLIs you have (any subset works): Claude Code, Codex, 

2245 Antigravity. Optionally run a local model via Ollama for a free panelist. 

2246 Check what's available: 

2247 jury --doctor 

2248 

22492. Create a config (picks reviewers, rounds, chair/vote, verify): 

2250 jury init --wizard 

2251 Every question is skippable — Enter keeps the built-in default. 

2252 

22533. Run your first review: 

2254 jury --pr 123 # a pull request 

2255 jury --issue 42 # an issue's completeness 

2256 git diff | jury --diff-file - # the current branch 

2257 

2258 The panel reviews independently, cross-examines (debate), the chair verifies 

2259 candidate findings to cut false positives, then synthesizes one verdict. 

2260 

22614. Post the verdict back to GitHub (optional): 

2262 jury --pr 123 --post-summary # one rollup comment 

2263 jury --pr 123 --post-inline # line-level comments 

2264 

22655. Gate CI on blocking findings (optional): 

2266 jury --pr 123 --ci # non-zero exit on critical/major 

2267 

2268Reviewers run sandboxed/read-only over attacker-controlled diffs by default. 

2269See `jury examples` for more, or `jury --help` for every option. 

2270Docs: https://github.com/berkayturanci/ai-jury""" 

2271 

2272 

2273def main(argv: list[str] | None = None) -> int: 

2274 _force_utf8_output() 

2275 raw = list(sys.argv[1:] if argv is None else argv) 

2276 

2277 # First-impression UX (#265): a newcomer running bare `jury` in a terminal 

2278 # gets a friendly overview and exits 0 — not the argparse error. The strict 

2279 # "provide one of --pr/--issue/--diff-file" error + non-zero exit is kept for 

2280 # non-interactive use (piped/CI), so scripts that forget an input still fail. 

2281 # `sys.stdin` can be None when stdin is detached (e.g. a background process), 

2282 # so guard before calling isatty(). 

2283 if not raw and sys.stdin is not None and sys.stdin.isatty(): 

2284 print(_OVERVIEW) 

2285 return 0 

2286 

2287 # Plain-language command overview / walkthrough (#265), argv-intercepts like 

2288 # the other subcommands so the main flag surface stays flat. Each has its own 

2289 # small parser, so `--help` describes the subcommand the epilogue promises it 

2290 # does (it printed the main help), and trailing junk (`jury examples foo`) 

2291 # is an error rather than silently ignored. 

2292 if raw[:1] in (["examples"], ["guide"]): 

2293 argparse.ArgumentParser( 

2294 prog=f"jury {raw[0]}", 

2295 description=( 

2296 "Print example commands." if raw[0] == "examples" else "Print a short walkthrough." 

2297 ), 

2298 ).parse_args(raw[1:]) 

2299 print(_EXAMPLES if raw[0] == "examples" else _GUIDE) 

2300 return 0 

2301 # Documented `jury cache clear` UX (issue #33): handled before argparse so 

2302 # the rest of the CLI keeps its flat flag surface (no subcommands). Parsed 

2303 # by its own parser BEFORE anything is deleted: `jury cache clear --help` 

2304 # used to clear the cache, because every argument but `--cache-dir` was 

2305 # ignored — the one command here that destroys data took a request for help 

2306 # as a request to run. 

2307 if raw[:2] == ["cache", "clear"]: 

2308 from .cache import Cache 

2309 

2310 sub = argparse.ArgumentParser( 

2311 prog="jury cache clear", 

2312 description="Delete every local review-cache entry and rotate the cache key.", 

2313 ) 

2314 sub.add_argument( 

2315 "--cache-dir", default=None, help="cache directory (default: the user cache dir)" 

2316 ) 

2317 ns = sub.parse_args(raw[2:]) 

2318 removed = Cache(ns.cache_dir).clear() 

2319 print(f"Cleared {removed} cache entr{'y' if removed == 1 else 'ies'}.") 

2320 return 0 

2321 

2322 # Comment-command mode (issue #11): `jury comment --text "/jury review"` 

2323 # parses an allowlisted PR-comment command and dispatches a safe jury run. 

2324 # Handled before the main parser so the comment text is never confused with 

2325 # the jury's own flags, and never reaches a shell. 

2326 if raw[:1] == ["comment"]: 

2327 return _run_comment_command(raw[1:]) 

2328 

2329 # Config scaffolding (issue #107): `jury init` writes a jury.toml from 

2330 # detected agents / flags / interactive prompts. Intercepted before the main 

2331 # parser so it keeps its own small flag surface. 

2332 if raw[:1] == ["init"]: 

2333 return _run_init(raw[1:]) 

2334 

2335 # Config introspection: `jury config show` prints the EFFECTIVE resolved 

2336 # config + its source so you can see exactly what will run; `config path` 

2337 # prints just the source. 

2338 if raw[:1] == ["config"]: 

2339 return _run_config(raw[1:]) 

2340 

2341 # Apply verified suggested patches (issue #521): `jury apply` applies 

2342 # suggested patches directly to the working directory. 

2343 if raw[:1] == ["apply"]: 

2344 return _run_apply(raw[1:]) 

2345 

2346 # Single-agent role dispatch (issue #661): `jury run-agent` runs ONE agent 

2347 # for one role and prints a JSON result with attribution — the integration 

2348 # point for an orchestrator. Intercepted like the other subcommands. 

2349 if raw[:1] == ["run-agent"]: 

2350 return _run_run_agent(raw[1:]) 

2351 

2352 # Theater replay (issue #449): `jury replay <outcome.json>` re-drives the 

2353 # deliberation scene from a saved outcome — no agents, no network. 

2354 # Intercepted before the main parser like the other subcommands. 

2355 if raw[:1] == ["replay"]: 

2356 return _run_replay(raw[1:]) 

2357 

2358 args = build_parser().parse_args(argv) 

2359 

2360 # Checked before any short-circuiting branch below, so `--json` is rejected 

2361 # wherever it is misplaced rather than only on the paths that reach the end. 

2362 if args.json and not args.doctor: 

2363 print( 

2364 "error: --json applies to --doctor; use --format json for the review report.", 

2365 file=sys.stderr, 

2366 ) 

2367 return 2 

2368 

2369 # Same vocabulary check `validate_config` applies to `[jury.ci] fail_on`, 

2370 # reported with the same message (issue #718). Checked here, beside the 

2371 # other flag guards, so a misspelled gate is refused before an agent is 

2372 # paid for — and refused even without `--ci`, where the flag is inert: 

2373 # otherwise the typo surfaces only on the run it was supposed to gate. 

2374 if args.fail_on: 

2375 message = fail_on_error(args.fail_on.split(","), "--fail-on") 

2376 if message: 

2377 print(f"error: {message}", file=sys.stderr) 

2378 return 2 

2379 

2380 if args.clear_cache: 

2381 from .cache import Cache 

2382 

2383 removed = Cache(args.cache_dir).clear() 

2384 print(f"Cleared {removed} cache entr{'y' if removed == 1 else 'ies'}.") 

2385 return 0 

2386 

2387 if args.doctor: 

2388 # `--doctor` resolves the cross-vendor threshold from `--min-vendors` too 

2389 # (#863), so every bounded flag given here holds as it does on a run — 

2390 # `--min-reviews` included, not only the one the prediction reads. 

2391 message = _override_bound_error(args) 

2392 if message: 

2393 print(f"error: {message}", file=sys.stderr) 

2394 return 2 

2395 # Model discovery costs a probe per agent and only the JSON export 

2396 # renders it; the human report must not pay for a field it never prints. 

2397 diagnostics = doctor_module.build_diagnostics( 

2398 args.config, probe_models=args.json, min_vendors=args.min_vendors 

2399 ) 

2400 if args.json: 

2401 # Exactly ONE JSON document on stdout, and nothing else: the export 

2402 # is meant to be piped straight into `jq` / an orchestrator, so any 

2403 # human chatter (including the --write confirmation below) goes to 

2404 # stderr. 

2405 print(json.dumps(doctor_module.doctor_report_dict(diagnostics), indent=2)) 

2406 else: 

2407 print(doctor_module.render_report(diagnostics)) 

2408 if args.write: 

2409 try: 

2410 Path(args.write).write_text( 

2411 json.dumps(diagnostics, indent=2) + "\n", encoding="utf-8" 

2412 ) 

2413 except OSError as exc: 

2414 print(f"error: {redact(str(exc))[0]}", file=sys.stderr) 

2415 return 2 

2416 stream = sys.stderr if args.json else sys.stdout 

2417 print(f"\nWrote diagnostics to {args.write}", file=stream) 

2418 return 0 

2419 

2420 if args.config_validate: 

2421 source = args.config or "jury.toml (or built-in defaults)" 

2422 try: 

2423 data = load_raw_config(args.config) 

2424 warnings = validate_config(data, strict=args.strict_config) 

2425 except (ConfigError, FileNotFoundError) as exc: 

2426 print(redact(f"Config invalid ({source}): {exc}")[0], file=sys.stderr) 

2427 return 2 

2428 except OSError as exc: 

2429 print(_config_read_error(exc, args.config), file=sys.stderr) 

2430 return 2 

2431 if warnings: 

2432 print(f"Config valid with warnings ({source}):") 

2433 for w in warnings: 

2434 print(f" - {w}") 

2435 else: 

2436 print(f"Config valid ({source}).") 

2437 return 0 

2438 

2439 try: 

2440 config = load_config(args.config, validate=True, strict=args.strict_config) 

2441 except (ConfigError, FileNotFoundError) as exc: 

2442 # A missing --config path raises FileNotFoundError; the other load sites already catch 

2443 # both, so a bad path prints `Config invalid: …` instead of a traceback (#831). 

2444 print(f"Config invalid: {redact(str(exc))[0]}", file=sys.stderr) 

2445 return 2 

2446 except OSError as exc: 

2447 # It exists but cannot be read: an unreadable ./jury.toml, or --config naming 

2448 # a directory. Neither is a FileNotFoundError, so both were a traceback (#893). 

2449 print(_config_read_error(exc, args.config), file=sys.stderr) 

2450 return 2 

2451 # An auto-discovered ./jury.toml that runs local commands must be trusted before those 

2452 # commands run — the checkout may be one this operator did not write (#831). 

2453 try: 

2454 configtrust.enforce(args.config, config, mock=args.mock) 

2455 except configtrust.ConfigTrustError as exc: 

2456 print(f"error: {exc}", file=sys.stderr) 

2457 return 2 

2458 # One value, one field, one answer, whichever surface it was written on 

2459 # (issue #748). The overrides below are assigned straight onto the config 

2460 # AFTER `validate_config` has run, so until now nothing range-checked them: 

2461 # `--rounds 0` was accepted, ran a full Round 1 and exited 0 while `rounds = 

2462 # 0` in `jury.toml` was a hard error, on a flag `docs/parameters.md` 

2463 # documents as `≥ 1`. An operator who wrote `--rounds 0` meaning "no 

2464 # debate" got a complete review round, a report, and nothing anywhere 

2465 # saying the number they passed had been ignored. 

2466 # 

2467 # Refused here — before the diff is read and long before an agent is paid 

2468 # for — with exit 2, like every other bad-input exit on this path. 

2469 override_error = _override_bound_error(args) 

2470 if override_error: 

2471 print(f"error: {override_error}", file=sys.stderr) 

2472 return 2 

2473 if args.rounds is not None: 

2474 config.rounds = args.rounds 

2475 # A fixed --rounds is a hard override: it disables adaptive early-stop so 

2476 # the run is reproducible fixed-N (issue #40), unless --early-stop is also 

2477 # passed explicitly (handled below). 

2478 config.early_stop = False 

2479 if args.max_rounds is not None: 

2480 config.max_rounds = args.max_rounds 

2481 if args.early_stop is not None: 

2482 config.early_stop = args.early_stop 

2483 if args.total_timeout is not None: 

2484 config.total_timeout = args.total_timeout 

2485 if args.phase_timeout is not None: 

2486 config.phase_timeout = args.phase_timeout 

2487 if args.retries is not None: 

2488 # Assigned as written: the `max(0, …)` clamp that used to stand here was 

2489 # the second rule for one value (#748). It silently turned `--retries 

2490 # -1` into 0 while `retries = -1` in `jury.toml` was a hard error, and 

2491 # is unreachable now that the same bound refuses the flag above. 

2492 config.retries = args.retries 

2493 if args.seed is not None: 

2494 config.seed = args.seed 

2495 if args.chair: 

2496 # Refused, not silently replaced (docs audit 2026-09-29): an unknown 

2497 # `--chair` used to exit 0 with the first usable agent chairing, so the 

2498 # operator's choice of synthesizer was ignored without a word. Checked 

2499 # against the enabled seats, before anything is fetched or spent. 

2500 enabled = [a.name for a in config.enabled_agents] 

2501 if args.chair != "rotate" and args.chair not in enabled: 

2502 print( 

2503 f"error: --chair {args.chair!r} is not an enabled agent (enabled: " 

2504 f"{', '.join(enabled) or 'none'}); name one of them, or 'rotate'.", 

2505 file=sys.stderr, 

2506 ) 

2507 return 2 

2508 config.chair = args.chair 

2509 # Applied to the config (rather than resolved late like --min-vendors) 

2510 # because the pre-run half of this gate lives in the orchestrator: it has to 

2511 # be able to refuse a bench that is too small BEFORE the panel is paid for. 

2512 if args.min_reviews is not None: 

2513 # Assigned as written: the bound above refuses a negative (docs audit 

2514 # 2026-09-29), where a `max(0, …)` clamp here used to turn it into "off". 

2515 config.ci.min_reviews = args.min_reviews 

2516 # --effort is a whole-panel override: it wins over every [[agent]] effort so 

2517 # one flag raises (or lowers) the depth of the entire run. 

2518 if args.effort is not None: 

2519 for agent in config.agents: 

2520 agent.effort = args.effort 

2521 # Warn ONCE per run per distinct message, before any agent is invoked, for 

2522 # every panelist whose vendor cannot act on the requested effort. Real 

2523 # adapters are passed only outside --mock, so an offline demo never spawns a 

2524 # vendor CLI to check a model listing. 

2525 for warning in effort_warnings( 

2526 config.enabled_agents, adapter_factory=None if args.mock else make_adapter 

2527 ): 

2528 print(f"warning: {warning}", file=sys.stderr) 

2529 if args.verify is not None: 

2530 config.verify = args.verify 

2531 if args.context_mode is not None: 

2532 config.context.mode = args.context_mode 

2533 if args.redact is not None: 

2534 config.context.redact_secrets = args.redact 

2535 if args.max_diff_bytes is not None: 

2536 config.diff.max_bytes = args.max_diff_bytes 

2537 if args.chunk is not None: 

2538 config.diff.chunk = args.chunk 

2539 if args.exclude: 

2540 config.diff.exclude = list(config.diff.exclude) + list(args.exclude) 

2541 if args.include: 

2542 config.diff.include = list(config.diff.include) + list(args.include) 

2543 

2544 try: 

2545 policy = load_policy(args.policy) 

2546 except PolicyError as exc: 

2547 print(f"error: {redact(str(exc))[0]}", file=sys.stderr) 

2548 return 2 

2549 

2550 # Issue mode (issue #221) reviews prose, not a diff, so the PR/diff-only 

2551 # concepts below have no meaning. Reject them up front with a clear message 

2552 # rather than silently ignoring them. 

2553 # Exactly one source (issue #367). Listed rather than pairwise so adding a 

2554 # source cannot quietly skip the check. 

2555 _sources = [ 

2556 ("--pr", args.pr), 

2557 ("--issue", args.issue), 

2558 ("--diff-file", args.diff_file), 

2559 ("--commit", getattr(args, "commit", None)), 

2560 ("--commits", getattr(args, "commits", None)), 

2561 ] 

2562 _given = [flag for flag, value in _sources if value] 

2563 if len(_given) > 1: 

2564 raise SystemExit(f"error: choose one input source, got {', '.join(_given)}") 

2565 if args.issue: 

2566 for flag, on in ( 

2567 ("--post-inline", args.post_inline), 

2568 ("--post-progress", args.post_progress), 

2569 ("--label", args.label), 

2570 ("--incremental", args.incremental), 

2571 # The issue path posts one summary and never reads the mode (docs 

2572 # audit 2026-09-29): accepted, `phased` was silently ignored. 

2573 ("--post-mode", args.post_mode is not None), 

2574 ): 

2575 if on: 

2576 raise SystemExit( 

2577 f"error: {flag} is not supported with --issue (it is a PR/diff concept)" 

2578 ) 

2579 

2580 # A posting flag with nowhere to post is a usage error, so it is refused here with 

2581 # the rest of the argument checks. It used to be refused after the review, when 

2582 # every seat had already run and been paid for (#866). `--post-summary` also posts 

2583 # to an issue; the others are PR-only and `--issue` rejected them above. 

2584 for flag, on, has_target in ( 

2585 ("--post-summary", args.post_summary, args.pr or args.issue), 

2586 ("--post-inline", args.post_inline, args.pr), 

2587 ("--label", args.label, args.pr), 

2588 ): 

2589 if on and not has_target: 

2590 raise SystemExit(f"error: {flag} requires --pr") 

2591 # `--post-mode` shapes the summary comment, so without one it shapes nothing 

2592 # (docs audit 2026-09-29): `--post-mode phased` alone exited 0 and posted 

2593 # nothing, which docs/parameters.md has always said it requires. 

2594 if args.post_mode is not None and not args.post_summary: 

2595 raise SystemExit("error: --post-mode requires --post-summary (or --post)") 

2596 

2597 # Live progress on the PR (issue #125): a single sticky comment updated at 

2598 # each round/chunk milestone. Opt-in and requires --pr. 

2599 progress = None 

2600 if args.post_progress: 

2601 if not args.pr: 

2602 raise SystemExit("error: --post-progress requires --pr") 

2603 from .github import ProgressReporter 

2604 

2605 progress = ProgressReporter(args.pr, args.repo) 

2606 

2607 def log(msg: str) -> None: 

2608 if not args.quiet: 

2609 print(f"[jury] {msg}", file=sys.stderr) 

2610 if progress is not None and _is_progress_milestone(msg): 

2611 progress.update(msg) 

2612 

2613 # Smart offline fallback: with NO config file and NO usable agent CLI, but a 

2614 # local model server reachable, add a local agent so `jury` just works 

2615 # offline out of the box (issue: easier zero-config). Never overrides an 

2616 # explicit config or a working CLI panel. 

2617 local_fallback = _maybe_add_local_fallback(config, args, log) 

2618 

2619 try: 

2620 diff, context = _read_diff(args) 

2621 except RuntimeError as exc: 

2622 # `--pr` shells out to `gh`; a missing or failing `gh` raises RuntimeError, which was 

2623 # otherwise uncaught here and printed a Python traceback (#831). 

2624 print(f"error: {redact(str(exc))[0]}", file=sys.stderr) 

2625 return 2 

2626 

2627 # Incremental review (issue #9): when --incremental and a prior jury 

2628 # marker exists, narrow the diff to the range since the last reviewed SHA; 

2629 # otherwise fall back safely to the full diff. The reviewed head SHA is also 

2630 # recorded on the posted summary so a later run can go incremental. 

2631 review_scope = None 

2632 head_sha = "" 

2633 if args.incremental: 

2634 if not args.pr: 

2635 raise SystemExit("error: --incremental requires --pr") 

2636 from . import incremental as inc 

2637 from .github import compare_diff, pr_comment_bodies, pr_head_sha 

2638 

2639 head_sha = pr_head_sha(args.pr, args.repo) 

2640 prev_sha = inc.parse_reviewed_sha(pr_comment_bodies(args.pr, args.repo)) 

2641 mode, reason = inc.decide_review(prev_sha, head_sha) 

2642 if mode == inc.MODE_INCREMENTAL: 

2643 inc_diff = compare_diff(prev_sha, head_sha, args.repo) 

2644 if inc_diff.strip(): 

2645 diff = inc_diff 

2646 else: 

2647 mode, reason = inc.MODE_FULL, "incremental range unavailable — full review" 

2648 review_scope = inc.scope_note(mode, reason) 

2649 log(reason) 

2650 

2651 if not diff.strip(): 

2652 raise SystemExit("error: empty diff — nothing to review") 

2653 

2654 # Risk-aware auto-depth (issue #120): scale rounds/verify to the diff when 

2655 # enabled. Explicit --rounds/--verify/--early-stop always win; the panel is 

2656 # never trimmed. Off unless --auto or [jury] auto_depth. 

2657 if args.auto if args.auto is not None else config.auto_depth: 

2658 from .diffprofile import depth_for, describe, profile_diff 

2659 

2660 prof = profile_diff(diff) 

2661 rounds, verify, early_stop = depth_for(prof.risk) 

2662 if args.rounds is None: 

2663 config.rounds = rounds 

2664 if args.early_stop is None: 

2665 config.early_stop = early_stop 

2666 if args.verify is None: 

2667 config.verify = verify 

2668 log(describe(prof)) 

2669 

2670 if getattr(args, "tiered", False): 

2671 config.routing = "tiered" 

2672 if config.routing == "tiered" and getattr(args, "min_vendors", None) is not None: 

2673 # The routed panel keeps at least the vendor floor the gate will apply 

2674 # (#714); a `--min-vendors` override has to reach the plan, not only the 

2675 # gate that runs after the panel has been paid for. 

2676 config.ci.min_vendors = int(args.min_vendors) 

2677 # ``--hints`` / ``--no-hints`` override ``[jury] hints`` in BOTH directions; 

2678 # the sentinel (None) means "not passed", so the config value stands (#715). 

2679 hints_override = getattr(args, "hints", None) 

2680 if hints_override is not None: 

2681 config.hints = hints_override 

2682 

2683 # The pre-pass block is carried SEPARATELY from the user context (#715). 

2684 # Appending it to ``context`` here made it invisible under the default 

2685 # context mode: ``run_jury`` clears ``context`` when the mode is "diff-only", 

2686 # so the linters ran, this line was logged, and the panel saw nothing. The 

2687 # orchestrator joins the block into the Round 1 prompt after that filter. 

2688 # 

2689 # The linters see the CHANGED files and nothing else (#737). The paths come 

2690 # from the same plan ``review_diff`` builds below — ``plan_for`` is pure, so 

2691 # the ``kept`` list here is the one the panel is shown — which means a file 

2692 # dropped by ``[jury.diff] include/exclude`` never contributes a hint about a 

2693 # diff the reviewers cannot read. ``--issue`` reviews prose and has no 

2694 # changed paths at all, so it gets no block. When the change touches nothing 

2695 # Ruff or ESLint handles, ``collect_static_hints`` returns "" rather than 

2696 # falling back to the working tree. 

2697 hints_block = "" 

2698 if config.hints: 

2699 from .hints import collect_static_hints 

2700 from .orchestrator import plan_for 

2701 

2702 changed_paths = [] if args.issue else [p for p in plan_for(config, diff).kept_paths if p] 

2703 sh = collect_static_hints(changed_paths) 

2704 if sh: 

2705 hints_block = sh 

2706 log(f"injected static analysis hints for {len(changed_paths)} changed file(s)") 

2707 

2708 # Optional local result cache (issue #33): a hit skips the run entirely; a 

2709 # miss runs the jury and stores the outcome. The key covers the diff, 

2710 # effective config, prompt version, package version, context policy, the 

2711 # context text that policy admits (#738), the static-analysis block the 

2712 # linters produced (#745), and seed. ``context`` is passed straight through from 

2713 # ``_read_diff`` and ``hints_block`` is the string resolved just above — both 

2714 # are the ones ``run_jury`` gets, so the key is a function of what the panel 

2715 # is shown and not of a second reading of the config. 

2716 cache = None 

2717 cache_k = None 

2718 outcome = None 

2719 if args.cache: 

2720 from .cache import Cache, cache_key 

2721 

2722 cache = Cache(args.cache_dir) 

2723 cache_k = cache_key( 

2724 config, 

2725 diff, 

2726 context=context, 

2727 hints=hints_block, 

2728 mock=args.mock, 

2729 policy=policy, 

2730 mode=("issue" if args.issue else "code"), 

2731 ) 

2732 outcome = cache.load(cache_k) 

2733 if outcome is not None: 

2734 log(f"cache hit ({cache_k[:12]}…) — reusing stored outcome") 

2735 else: 

2736 log(f"cache miss ({cache_k[:12]}…) — running jury") 

2737 

2738 # Live play-by-play (issue #210, #229): stream each step as it happens. Prints 

2739 # a titled block to stdout the moment a phase result lands. Posting each step to 

2740 # the PR/issue is OPT-IN — it requires BOTH a target (--pr or --issue) AND 

2741 # --post (a bare target only selects the source, never auto-posts), so `--live` 

2742 # alone just streams locally. Posting is best-effort: a GitHub hiccup is logged 

2743 # and never aborts the run. 

2744 live_target = args.pr or args.issue 

2745 # Theater defaults can come from jury.toml (issue #364); the CLI flags 

2746 # (--theater / --no-theater, --theater-style) override per run. Sentinels 

2747 # (None) distinguish "not passed" from an explicit choice. 

2748 theater_on = args.theater if args.theater is not None else config.theater 

2749 theater_style = args.theater_style or config.theater_style 

2750 live_posts = bool((args.live or theater_on) and args.post_summary and live_target) 

2751 live_post = post_issue_comment if args.issue else post_pr_comment 

2752 # Opt-in animated "courtroom" scene (--theater): an interactive TTY view of 

2753 # the REAL run (each model seated, speaking per phase, gavel/vote finale). It 

2754 # needs a wide TTY and an actual run (a cache hit has nothing to replay), so 

2755 # it falls back to the plain --live step stream otherwise. The structured 

2756 # outcome / report / CI gate are untouched — this is a side channel. 

2757 court = None 

2758 if theater_on and outcome is None and not args.quiet: 

2759 from . import theater as _theater 

2760 

2761 if _theater.supports_scene(sys.stdout): 2761 ↛ 2791line 2761 didn't jump to line 2791 because the condition on line 2761 was always true

2762 # Display-only chair label for the scene title. The run resolves the 

2763 # REAL chair internally (resolve_chair needs the usable/reviewer sets 

2764 # and run RNG, which don't exist yet here), so use a best-effort name. 

2765 chair_name = ( 

2766 config.chair 

2767 if config.chair and config.chair != "rotate" 

2768 else (config.agents[0].name if config.agents else "chair") 

2769 ) 

2770 case = ( 

2771 f"PR #{args.pr}" 

2772 if args.pr 

2773 else f"issue #{args.issue}" 

2774 if args.issue 

2775 else f"commit {args.commit}" 

2776 if getattr(args, "commit", None) 

2777 else f"range {args.commits}" 

2778 if getattr(args, "commits", None) 

2779 else "local diff" 

2780 ) 

2781 court = _theater.Courtroom( 

2782 [(a.name, a.vendor) for a in config.agents], 

2783 chair_name, 

2784 case=case, 

2785 mode=("issue" if args.issue else "code"), 

2786 decision=(args.decision or config.decision), 

2787 style=theater_style, 

2788 ) 

2789 court.open() 

2790 

2791 on_event = None 

2792 if args.live or theater_on: 

2793 

2794 def on_event(kind, result, round_no=None): 

2795 if court is not None: 

2796 court.step(kind, result, round_no) 

2797 else: 

2798 # plain step stream (--live, or --theater fallback off a TTY) 

2799 title, body = render_live_step(kind, result, round_no) 

2800 print(f"## {title}\n\n{body}\n", flush=True) 

2801 if live_posts: 

2802 try: 

2803 title, body = render_live_step(kind, result, round_no) 

2804 live_post(live_target, f"## {title}\n\n{body}", args.repo) 

2805 except Exception as exc: # noqa: BLE001 - best-effort, never crash 

2806 log(f"live: failed to post step to #{live_target}: {redact(str(exc))[0]}") 

2807 

2808 # We stream live only when actually running the jury; a cache hit has nothing 

2809 # to replay, so the consolidated report is still printed in that case. 

2810 live_streamed = bool(args.live or theater_on) and outcome is None 

2811 

2812 if outcome is None: 

2813 try: 

2814 if args.issue: 

2815 # Issue prose bypasses large-diff planning (filter/size/chunk is 

2816 # meaningless for an issue body); run the jury directly with the 

2817 # issue-quality rubric. 

2818 outcome = run_jury( 

2819 config, 

2820 diff, 

2821 context=context, 

2822 hints=hints_block, 

2823 mock=args.mock, 

2824 strict=args.strict, 

2825 policy=policy, 

2826 log=log, 

2827 on_event=on_event, 

2828 mode="issue", 

2829 ) 

2830 else: 

2831 outcome, _plan = review_diff( 

2832 config, 

2833 diff, 

2834 context=context, 

2835 hints=hints_block, 

2836 mock=args.mock, 

2837 strict=args.strict, 

2838 policy=policy, 

2839 log=log, 

2840 on_event=on_event, 

2841 ) 

2842 except KeyboardInterrupt: 

2843 # Graceful cancellation (issue #30): a jury run can be long, so 

2844 # Ctrl-C should exit cleanly with the conventional 130 rather than 

2845 # dumping a traceback. Work already completed is not partially 

2846 # rendered here because the orchestrator returns atomically; we just 

2847 # report the cancellation. 

2848 print("\n[jury] cancelled (interrupted) — no report produced", file=sys.stderr) 

2849 return 130 

2850 except RuntimeError as exc: 

2851 # Large-diff "too large / nothing to review" (issue #31) and "no 

2852 # usable agents" are actionable user errors, not crashes. 

2853 print(f"error: {redact(str(exc))[0]}", file=sys.stderr) 

2854 return 2 

2855 if cache is not None and cache_k is not None: 

2856 cache.store(cache_k, outcome) 

2857 log(f"cached outcome ({cache_k[:12]}…)") 

2858 

2859 # Final-verdict mode (issue #220): a panel vote (tally the reviewers) vs the 

2860 # chair's synthesis. Rendering-only — the outcome is identical; the severity- 

2861 # based CI gate below is unaffected. Effective = CLI flag else config. 

2862 decision = args.decision or config.decision 

2863 vote = None 

2864 if decision == "vote": 

2865 from .voting import is_abstention, tally_votes 

2866 

2867 # A reviewer that abstained (empty reply or a refusal) is excluded from 

2868 # the tally — a non-answer must not count as a "clear" vote (issue #251). 

2869 voters = [ 

2870 r.agent for r in outcome.reviews if r.ok and not is_abstention(getattr(r, "output", "")) 

2871 ] 

2872 vote = tally_votes( 

2873 outcome.groups, 

2874 voters, 

2875 mode=("issue" if args.issue else "code"), 

2876 ) 

2877 

2878 # Close the courtroom scene (after the vote is tallied, so the panel-vote 

2879 # finale can show the ballots/verdict). 

2880 if court is not None: 

2881 if vote is not None: 

2882 court.set_vote(vote) 

2883 court.close() 

2884 

2885 # Ballot vocabulary follows the review mode, exactly as the vote tally does: 

2886 # an issue review votes on completeness (READY/UNCLEAR/NEEDS_INFO), not on a 

2887 # diff's correctness. Resolved BEFORE the metadata, which now derives the 

2888 # ballots to count them (#700, round 2) and must derive the same ones the 

2889 # report renders. 

2890 ballot_mode = "issue" if args.issue else "code" 

2891 

2892 # Whether this run's panel is the zero-config fallback's one local seat (#863): 

2893 # recorded in the metadata and stated in the report, not only on stderr, which 

2894 # `--quiet` silences. 

2895 zero_config = local_fallback is not None 

2896 metadata = build_run_metadata( 

2897 outcome, 

2898 config, 

2899 decision=decision, 

2900 vote=vote, 

2901 mode=ballot_mode, 

2902 zero_config_fallback=zero_config, 

2903 ) 

2904 

2905 # The attribution footer (issue #911): on unless `--no-attribution` or 

2906 # `[jury.output] attribution = false`. Appended below, once the markdown 

2907 # report is complete; `transcript_mode` picks the transcript's wording. 

2908 attribution_on = args.attribution if args.attribution is not None else config.output.attribution 

2909 transcript_mode = False 

2910 

2911 if args.format == "json": 

2912 from .formats import to_json 

2913 

2914 report = to_json( 

2915 outcome, 

2916 config, 

2917 decision=decision, 

2918 vote=vote, 

2919 mode=ballot_mode, 

2920 zero_config_fallback=zero_config, 

2921 ) 

2922 elif args.format == "sarif": 

2923 from .formats import to_sarif 

2924 

2925 report = to_sarif(outcome, config) 

2926 elif args.format == "keel-reviews": 

2927 from .formats import to_keel_reviews 

2928 

2929 report = to_keel_reviews(outcome, config, vote=vote, mode=ballot_mode) 

2930 else: 

2931 # Output mode (issue: full transcript). --verbose => summary + transcript; 

2932 # --transcript (or [jury] transcript, unless --no-transcript) => the 

2933 # chronological play-by-play; otherwise the consensus-first summary. 

2934 # Rendering-only — the orchestration/outcome is identical either way. 

2935 transcript_default = args.transcript if args.transcript is not None else config.transcript 

2936 transcript_mode = bool(args.verbose or transcript_default) 

2937 if transcript_mode: 

2938 report = render_transcript( 

2939 outcome.reviews, 

2940 outcome.debate, 

2941 outcome.synthesis, 

2942 chair=outcome.chair, 

2943 findings=outcome.findings, 

2944 warnings=outcome.warnings, 

2945 groups=outcome.groups, 

2946 verify=outcome.verify, 

2947 context_mode=outcome.context_mode, 

2948 redact_secrets=outcome.redact_secrets, 

2949 redaction_count=outcome.redaction_count, 

2950 metadata=metadata, 

2951 review_scope=review_scope, 

2952 lead_with_summary=bool(args.verbose), 

2953 vote=vote, 

2954 footer=False, 

2955 ) 

2956 else: 

2957 report = render( 

2958 outcome.reviews, 

2959 outcome.debate, 

2960 outcome.synthesis, 

2961 chair=outcome.chair, 

2962 findings=outcome.findings, 

2963 warnings=outcome.warnings, 

2964 groups=outcome.groups, 

2965 verify=outcome.verify, 

2966 context_mode=outcome.context_mode, 

2967 redact_secrets=outcome.redact_secrets, 

2968 redaction_count=outcome.redaction_count, 

2969 metadata=metadata, 

2970 review_scope=review_scope, 

2971 vote=vote, 

2972 footer=False, 

2973 ) 

2974 

2975 if args.metadata_json: 

2976 with Path(args.metadata_json).open("w", encoding="utf-8") as fh: 

2977 fh.write(json.dumps(metadata, indent=2) + "\n") 

2978 log(f"metadata written to {redact(args.metadata_json)[0]}") 

2979 

2980 ci_exit = 0 

2981 # A run whose panel collapsed is a different thing wearing the same output 

2982 # (#625/#682). `--strict` fails when a configured CLI is *missing*; this 

2983 # fails when one was present, probed fine, and returned nothing — which is 

2984 # how a three-vendor panel silently becomes one. It FAILS CLOSED by default 

2985 # (`[jury.ci] min_vendors`, shipped as 2) and is scoped to runs that claimed 

2986 # cross-vendor consensus, so a single-vendor install is untouched; exit 3, so 

2987 # it is distinguishable from a findings failure. 

2988 # 

2989 # The zero-config local fallback (#863) seats one local reviewer beside the 

2990 # built-in seats it found missing — it fires only when none of them can run — 

2991 # so the panel it forms is that seat alone and never claimed cross-vendor 

2992 # consensus. Counting the missing seats as configured vendors failed the 

2993 # documented `jury --diff-file -` offline path with exit 3 on every run. A 

2994 # threshold named on the command line is still enforced as asked. 

2995 required, explicit = resolve_min_vendors(getattr(args, "min_vendors", None), config) 

2996 collapsed = collapse_reason( 

2997 outcome.reviews, 

2998 required, 

2999 None if explicit else claimed_vendors(config.enabled_agents, local_fallback), 

3000 ) 

3001 if collapsed: 

3002 log(collapsed) 

3003 # A runner with claude + agy and no codex lands here since agy left the 

3004 # default panel: name the second vendor it already has, and why it sat out. 

3005 from .config import agy_opt_in_hint 

3006 

3007 hint = agy_opt_in_hint((s.adapter_key for s in config.agents), shutil.which) 

3008 if hint: 

3009 log(f"note: {hint}; or install another vendor's CLI, or lower --min-vendors") 

3010 ci_exit = 3 

3011 # No seat returned anything (#849). One local seat pointed at a model its 

3012 # server does not have failed every call with `HTTP 404`, and the run still 

3013 # exited 0 with a report of nothing. The collapse guard above is scoped to 

3014 # runs that claimed cross-vendor consensus and `min_reviews` is off by default, 

3015 # so neither saw it. This is narrower than both on purpose: not "too few 

3016 # reviews" — a seat that answered "looks good" is an abstention, and a clean 

3017 # single-seat run must stay green — but "every seat failed to return a result 

3018 # at all". A run that reviewed nothing is not a pass. 

3019 ran = metadata.get("panel") or {} 

3020 configured_seats = int(ran.get("configured") or 0) 

3021 if configured_seats and int(ran.get("failed") or 0) == configured_seats: 

3022 log( 

3023 f"no reviewer returned a result: all {configured_seats} seat(s) failed, so " 

3024 f"nothing was reviewed. A run with no review is not a pass — see each " 

3025 f"seat's error above, and `jury --doctor` (e.g. a local model not pulled)." 

3026 ) 

3027 ci_exit = 3 

3028 # The other half of the #699 gate. The orchestrator refuses a bench that is 

3029 # too small before spending it; this catches the two cases a pre-flight 

3030 # cannot predict — an agent that was present, ran, and returned nothing, and 

3031 # one that answered with prose naming nothing a reader could check. Both are 

3032 # recorded as abstentions and neither is a review, so the number the consumer 

3033 # will accept can fall below its minimum while every seat "answered" (#700, 

3034 # round 2: `--min-reviews 3` used to be satisfied by three "Looks good to me" 

3035 # replies). Exit 3, the same family as a collapsed panel: the panel, not the 

3036 # findings, is what fell short. Evaluated on cached outcomes too, which is 

3037 # why `min_reviews` is kept out of the cache key. 

3038 panel_meta = metadata.get("panel") or {} 

3039 short_panel = panel.shortfall( 

3040 panel_meta.get("reviews_supplied") or 0, 

3041 config.ci.min_reviews, 

3042 stage="after the panel ran", 

3043 # Every cause the metadata publishes, passed by the one mapping that 

3044 # names them: a call site listing two of the four printed the other two 

3045 # under a cause their own ballots contradicted (#700, round 5). 

3046 **{key: panel_meta.get(key) or 0 for key in panel.PANEL_METADATA_KEYS.values()}, 

3047 ) 

3048 if short_panel: 

3049 log(short_panel) 

3050 ci_exit = 3 

3051 if args.ci: 

3052 fail_on = config.ci.fail_on 

3053 if args.fail_on: 

3054 fail_on = [s.strip().lower() for s in args.fail_on.split(",") if s.strip()] 

3055 gate_exit, ci_reason = evaluate_ci(outcome.groups, fail_on, config.ci.ignore_unverified) 

3056 # A collapsed panel outranks the severity gate: `evaluate_ci` reports on 

3057 # findings the panel did or did not raise, and a panel that never formed 

3058 # is not evidence either way. Before #682 this assignment overwrote the 

3059 # collapse exit, so `--ci --min-vendors 2` reported a clean pass on a 

3060 # single-vendor run — the exact failure the flag was added for. 

3061 ci_exit = ci_exit or gate_exit 

3062 # Only the markdown report carries the human-readable CI gate section; 

3063 # json/sarif documents stay machine-clean. The exit code is unchanged. 

3064 if args.format == "markdown": 

3065 report += f"\n\n## CI gate\n\n{ci_reason}\n" 

3066 

3067 # Suggested patches (issue #10): opt-in and kept separate from the default 

3068 # report. Written to a file with --patches-out, else appended after the 

3069 # markdown report under its own heading. The default flow stays read-only. 

3070 if args.suggest_patches: 

3071 from .patches import render_patch_suggestions 

3072 

3073 patches_section = render_patch_suggestions(outcome.groups) 

3074 if not patches_section: 

3075 log("no verified findings with a suggested fix — no patches emitted") 

3076 elif args.patches_out: 

3077 Path(args.patches_out).write_text(patches_section, encoding="utf-8") 

3078 log(f"suggested patches written to {redact(args.patches_out)[0]}") 

3079 elif args.format == "markdown": 

3080 report += "\n\n" + patches_section.rstrip() 

3081 else: 

3082 log("--suggest-patches needs markdown output or --patches-out; skipped") 

3083 

3084 # The attribution footer (issue #911) closes the markdown report: the tool and 

3085 # the seats that returned a review. The renderers were asked to leave it off so 

3086 # it can go here, AFTER the CI gate and patch sections appended above, and the 

3087 # report ends with exactly one footer. JSON/SARIF/keel-reviews are machine 

3088 # documents and never get it. Posting appends the hidden SHA marker after it. 

3089 if attribution_on and args.format == "markdown": 

3090 report += "\n" + render_footer(outcome.reviews, transcript=transcript_mode) 

3091 

3092 # Turn the live progress comment into the final verdict (issue #125). 

3093 if progress is not None: 

3094 progress.finish(report) 

3095 log(f"progress comment finalized on PR #{args.pr}") 

3096 

3097 if args.output: 

3098 # By the time this write runs the panel has been invoked, the debate is 

3099 # over and the verdict is rendered: the run is paid for. An unwritable 

3100 # path used to come out of `open()` as a raw `FileNotFoundError` 

3101 # traceback (issue #749) — an implementation detail where a sentence 

3102 # naming the path belongs, and one that took the report with it, because 

3103 # `-o` is exactly what suppresses the stdout copy below. 

3104 # 

3105 # So it is reported the way the `--doctor --write` failure a few hundred 

3106 # lines up is (named path, redacted reason, exit 2), and the report falls 

3107 # back to STDOUT rather than to another file. A fallback file would put 

3108 # the run somewhere the operator never named — a second surprise on top 

3109 # of the first, able to clobber, and no more likely to succeed when the 

3110 # cause is a full or read-only filesystem — while stdout is where this 

3111 # exact document would have gone had `-o` not been passed at all. It is 

3112 # redirectable and greppable, so the run survives; exit 2 keeps a caller 

3113 # from reading the delivery as a success. 

3114 try: 

3115 with Path(args.output).open("w", encoding="utf-8") as fh: 

3116 fh.write(report + "\n") 

3117 except OSError as exc: 

3118 print( 

3119 f"error: could not write the report to " 

3120 f"'{redact(args.output)[0]}': {redact(str(exc))[0]}", 

3121 file=sys.stderr, 

3122 ) 

3123 print( 

3124 "error: the review is complete and is printed on stdout instead; " 

3125 "redirect it to keep the report.", 

3126 file=sys.stderr, 

3127 ) 

3128 print(report) 

3129 return 2 

3130 log(f"report written to {redact(args.output)[0]}") 

3131 elif not (live_streamed and args.format == "markdown"): 

3132 # In --live markdown mode the step stream WAS the stdout output; don't also 

3133 # dump the consolidated report (it would duplicate everything just shown). 

3134 # For json/sarif the stream is human-readable markdown, so the requested 

3135 # machine-readable document must still go to stdout. 

3136 print(report) 

3137 

3138 # The read side of `gh` has been guarded since #836; the write side was not (#844). 

3139 # By the time control reaches here the panel has run and the verdict is already on 

3140 # stdout, so a `gh` failure — missing CLI, bad token, no write permission, a deleted 

3141 # PR, a timeout, a refused spawn — must not replace a finished review with a traceback 

3142 # and exit 1. The two kinds are deliberately different: 

3143 # 

3144 # * `--post-summary` is a contract. Asking for the verdict to be posted and not 

3145 # posting it is a failure of the run, so it exits 2 like every other user error. 

3146 # * `--post-inline` and `--label` are additions to a review that has already been 

3147 # delivered. They report the failure and leave `ci_exit` — the gate's own answer 

3148 # about the code — intact, because losing it would turn "the code is fine, GitHub 

3149 # hiccuped" into "the gate failed". 

3150 def _post(action: str, call, *, contractual: bool) -> bool: 

3151 """Run *call*, reporting a `gh` failure instead of letting it escape. 

3152 

3153 Returns **whether the post landed**, so no caller can log a success that 

3154 did not happen. An earlier version returned `2 if contractual else None`, 

3155 which made failure and success indistinguishable for the non-contractual 

3156 callers — both were `None` — and `--post-inline` printed the error and the 

3157 "posted inline comments" line one after the other. 

3158 """ 

3159 try: 

3160 call() 

3161 except RuntimeError as exc: 

3162 # `contractual` is the severity: failing a promise is an error, failing 

3163 # an addition to a verdict that did land is a warning. Printing both as 

3164 # `error:` while exiting 0 told the reader the opposite of the exit code. 

3165 prefix = "error" if contractual else "warning" 

3166 print(f"{prefix}: could not {action}: {redact(str(exc))[0]}", file=sys.stderr) 

3167 return False 

3168 return True 

3169 

3170 if args.post_summary: 

3171 if args.issue: 

3172 # Plain issues use `gh issue comment`; phased/SHA-marker posting is 

3173 # PR-only, so the issue path posts the single rendered report. 

3174 if not _post( 

3175 f"post the verdict to issue #{args.issue}", 

3176 lambda: post_issue_comment(args.issue, report, args.repo), 

3177 contractual=True, 

3178 ): 

3179 return 2 

3180 log(f"posted verdict to issue #{args.issue}") 

3181 return ci_exit 

3182 # Record the reviewed head SHA as a hidden marker so a later 

3183 # --incremental run can review only the new range (issue #9). 

3184 from .github import pr_head_sha 

3185 from .incremental import reviewed_sha_marker 

3186 

3187 # A missing marker costs a later `--incremental` run its narrowing; it is not 

3188 # worth refusing to post the verdict over, so this one degrades rather than exits. 

3189 # `pr_head_sha` is best-effort: it catches its own `gh` failure and returns 

3190 # "". An earlier version of this guard wrapped it in `try/except RuntimeError`, 

3191 # which is unreachable — the warning it promised could never print, and the 

3192 # test only passed because it mocked `pr_head_sha` itself rather than the `gh` 

3193 # call underneath. The empty string is the failure signal, so that is what is 

3194 # checked. 

3195 if head_sha: 

3196 marker_sha = head_sha 

3197 else: 

3198 marker_sha = pr_head_sha(args.pr, args.repo) 

3199 if not marker_sha: 

3200 print( 

3201 "warning: could not read the PR head sha, so this review will not " 

3202 "carry an incremental marker", 

3203 file=sys.stderr, 

3204 ) 

3205 marker = f"\n\n{reviewed_sha_marker(marker_sha)}" if marker_sha else "" 

3206 

3207 if args.post_mode == "phased": 

3208 # Post the flow as separate, readable comments (issue #127): 

3209 # Round 1 → debate → decision. The SHA marker rides the last one. 

3210 from .report import render_sections 

3211 

3212 sections = render_sections( 

3213 outcome.reviews, 

3214 outcome.debate, 

3215 outcome.synthesis, 

3216 chair=outcome.chair, 

3217 findings=outcome.findings, 

3218 warnings=outcome.warnings, 

3219 groups=outcome.groups, 

3220 verify=outcome.verify, 

3221 vote=vote, 

3222 ) 

3223 # The sections are markdown whatever `--format` says, so the footer 

3224 # follows the opt-out alone here. 

3225 phased_footer = f"\n\n{render_footer(outcome.reviews)}" if attribution_on else "" 

3226 for i, (title, body) in enumerate(sections): 

3227 # The last comment carries the footer (render_sections has none), 

3228 # then the SHA marker, which must stay last. 

3229 tail = f"{phased_footer}{marker}" if i == len(sections) - 1 else "" 

3230 if not _post( 

3231 f"post phased comment {i + 1} of {len(sections)} to PR #{args.pr}", 

3232 lambda t=title, b=body, x=tail: post_pr_comment( 

3233 args.pr, f"## {t}\n\n{b}{x}", args.repo 

3234 ), 

3235 contractual=True, 

3236 ): 

3237 return 2 

3238 log(f"posted {len(sections)} phased comments to PR #{args.pr}") 

3239 else: 

3240 if not _post( 

3241 f"post the verdict to PR #{args.pr}", 

3242 lambda: post_pr_comment(args.pr, f"{report}{marker}", args.repo), 

3243 contractual=True, 

3244 ): 

3245 return 2 

3246 log(f"posted verdict to PR #{args.pr}") 

3247 

3248 if args.post_inline and _post( 

3249 f"post inline comments to PR #{args.pr}", 

3250 lambda: post_inline_comments( 

3251 args.pr, outcome.findings, repo=args.repo, dry_run=args.dry_run 

3252 ), 

3253 contractual=False, 

3254 ): 

3255 log(f"posted inline comments to PR #{args.pr}") 

3256 

3257 # Optional GitHub labels (issue #7): OFF by default. Only applied when 

3258 # --label is passed AND a --pr target exists; never automatic. 

3259 if args.label: 

3260 labels = label_strings(classify(outcome)) 

3261 if _post( 

3262 f"apply labels to PR #{args.pr}", 

3263 lambda: apply_labels(args.pr, labels, args.repo), 

3264 contractual=False, 

3265 ): 

3266 log(f"applied labels to PR #{args.pr}: {', '.join(labels)}") 

3267 

3268 return ci_exit 

3269 

3270 

3271if __name__ == "__main__": 

3272 raise SystemExit(main())