Coverage for src/ai_jury/report.py: 99%
371 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Render the jury run into a single markdown report."""
3from __future__ import annotations
5import re
7from . import classification as _classification
8from . import panel as panel_mod
9from .adapters import AgentResult
10from .findings import SEVERITY_ORDER, Finding, flatten_inline
11from .voting import is_abstention
13#: A fenced ``suggestion`` block opener (``` ```suggestion ``` or a longer run, tildes too).
14#: ``jury apply`` parses ``` ```suggestion ``` blocks out of a report's prose to build a patch,
15#: and a reviewer's raw output is rendered into the transcript verbatim (#831 F3). A space
16#: between the fence and ``suggestion`` breaks the apply parser's exact match while the block
17#: still renders as a labelled code block, so a reviewer cannot smuggle in an applicable patch.
18#:
19#: NOT anchored to the line start: the apply parser (``patches.parse_patch_suggestions``) is
20#: itself unanchored, so it matches ``` ```suggestion ``` after a blockquote (``> ``) or list
21#: (``- ``) marker too, and the defuser has to reach the same spellings (agy round 1).
22_SUGGESTION_FENCE = re.compile(r"(?m)(`{3,}|~{3,})(suggestion)(?=\s*$)")
25def _defuse_patch_syntax(text: str) -> str:
26 """Neutralise a ``suggestion`` code fence in untrusted agent output (#831 F3)."""
27 return _SUGGESTION_FENCE.sub(r"\1 \2", text)
30def _block(title: str, body: str) -> str:
31 return f"### {title}\n\n{_defuse_patch_syntax(body).strip() or '_(no output)_'}\n"
34def _seat_list(reviews) -> str:
35 """The seats that RETURNED a review, deduplicated, in seat order (issue #911).
37 A seat that failed did not sit, so it is not claimed; nor is one that
38 abstained — answered with nothing, or refused (``voting.is_abstention``,
39 docs audit 2026-09-29) — because it returned no review either, and naming it
40 credited it with one. Chunked reviews carry one result per chunk per seat,
41 hence the dedupe: a seat that reviewed any chunk is named. The name comes
42 from the operator's config, not from an agent, but the footer is posted as
43 raw HTML, so it is flattened and escaped anyway.
44 """
45 import html
47 seats: list[str] = []
48 for r in reviews:
49 name = html.escape(flatten_inline(r.agent or ""))
50 if r.ok and not is_abstention(getattr(r, "output", "")) and name and name not in seats:
51 seats.append(name)
52 return ", ".join(seats)
55def render_footer(reviews, *, transcript: bool = False) -> str:
56 """The attribution footer a markdown report ends with (pure, issue #911).
58 One ``<sub>`` line naming the tool and the seats that returned a review, so a
59 reader of a posted verdict can tell what produced it and where to get it.
60 With no seat that returned a review the seat list is left out rather than
61 claiming an empty panel. Returned with its ``---`` rule, ready to append.
62 ``[jury.output] attribution = false`` / ``--no-attribution`` leave it off.
63 """
64 seats = _seat_list(reviews)
65 by = f" · {seats}" if seats else ""
66 if transcript:
67 body = (
68 f"<sub>Generated by [ai-jury](https://github.com/berkayturanci/ai-jury){by}"
69 " — a cross-vendor multi-agent PR review jury.</sub>"
70 )
71 else:
72 body = (
73 f"<sub>🏛️ Synthesized by [ai-jury](https://github.com/berkayturanci/ai-jury){by}"
74 " — Cross-vendor multi-agent code review · "
75 "[⭐ Star on GitHub](https://github.com/berkayturanci/ai-jury) · "
76 "[Add to your repo](https://ai-jury.dev/)</sub>"
77 )
78 return f"---\n\n{body}"
81def _fail_status(r: AgentResult) -> str:
82 """Failed-agent status line with a concise typed-error-code prefix."""
83 prefix = f"[{r.error_code}] " if getattr(r, "error_code", None) else ""
84 # The error snippet quotes the agent CLI's stderr (attacker-influenced) and
85 # is posted to the PR, so flatten it like every other untrusted field so it
86 # can't forge a heading/fence in the comment (audit 2026-06-13 r4).
87 return f"⚠️ {prefix}{flatten_inline(r.error)}"
90def _finding_line(f: Finding) -> str:
91 loc = flatten_inline(f.file) or "?"
92 if f.line is not None:
93 loc = f"{loc}:{f.line}"
94 claim = flatten_inline(f.claim)
95 return f"- [{f.severity}] {loc} — {claim} ({f.confidence}, by {f.reviewer})"
98_BUCKET_LABELS = {
99 "consensus": "Consensus (all reviewers)",
100 "majority": "Majority",
101 "single_reviewer": "Single reviewer",
102 "disputed": "Disputed (needs human decision)",
103 "rejected": "Rejected (unsupported by verifier)",
104}
105_BUCKET_ORDER = ["consensus", "majority", "single_reviewer", "disputed", "rejected"]
107_STATUS_LABELS = {
108 "verified": "verified",
109 "unsupported": "unsupported",
110 "needs_human_decision": "needs human decision",
111}
114def _group_line(g) -> str:
115 f = g.representative
116 loc = flatten_inline(f.file) or "?"
117 if f.line is not None:
118 loc = f"{loc}:{f.line}"
119 reviewers = ", ".join(g.reviewers) if g.reviewers else "(unknown)"
121 # Attacker-influenced fields (claim/evidence/fix) are flattened to one line
122 # so they cannot forge a heading or open a code fence in the posted report
123 # (audit 2026-06-13 r3).
124 parts = [f"- [{g.severity}] {loc} — {flatten_inline(f.claim)} (reviewers: {reviewers})"]
126 # Surface the reviewer's supporting evidence — the "why" behind the claim —
127 # so the verdict is auditable, not just asserted (issue: evidence surfacing).
128 if getattr(f, "evidence", ""):
129 parts.append(f"\n - _evidence:_ {flatten_inline(f.evidence)}")
131 status = getattr(g, "status", "")
132 if status:
133 reasoning = flatten_inline(getattr(g, "status_reasoning", ""))
134 if reasoning:
135 parts.append(
136 f"\n - _verification:_ {_STATUS_LABELS.get(status, status)} — {reasoning}"
137 )
138 else:
139 parts.append(f"\n - _verification:_ {_STATUS_LABELS.get(status, status)}")
141 if f.suggested_fix:
142 parts.append(f"\n - _fix:_ {flatten_inline(f.suggested_fix)}")
144 return "".join(parts) if len(parts) > 1 else parts[0]
147def _metadata_block(metadata: dict) -> list[str]:
148 """Render the deterministic run-metadata section.
150 Intentionally omits non-deterministic fields (e.g. ``generated_at``) so the
151 Markdown report stays stable for snapshot tests. Per-agent durations are
152 deterministic under mock (0s) and scrubbed by the golden test's duration
153 normalizer otherwise. Wall-clock is labelled a cost proxy, not a dollar cost.
154 """
155 lines = ["## Run metadata\n"]
156 lines.append(f"- rounds executed: {metadata['rounds_executed']}")
157 # Adaptive-rounds explanation (issue #40): only shown when the orchestrator
158 # recorded a reason, so a plain fixed-N run stays unchanged.
159 if metadata.get("from_cache"):
160 lines.append("- ♻️ served from local cache (not re-computed)")
161 stop_reason = metadata.get("stop_reason")
162 if stop_reason:
163 # flatten metadata strings too (defense-in-depth, audit r6/L): these are
164 # config/internal-controlled today, but keeping them single-line means a
165 # name/reason can never break the table or forge structure if a future
166 # source carries agent/diff text.
167 lines.append(f"- rounds decision: {flatten_inline(stop_reason)}")
168 routing_meta = metadata.get("routing") or {}
169 if routing_meta.get("mode") == "tiered":
170 # Tiered routing (#714): the cost decision is part of the run record.
171 line = f"- routing: tiered — {flatten_inline(routing_meta.get('reason', ''))}"
172 if routing_meta.get("escalated"):
173 line += f"; escalated: {flatten_inline(routing_meta.get('escalation_reason', ''))}"
174 lines.append(line)
175 lines.append(f"- verify: {'on' if metadata['verify_enabled'] else 'off'}")
176 lines.append(f"- context mode: {metadata['context_mode']}")
177 # Partial-result signals (issue #30): only rendered when relevant so a
178 # complete, unbudgeted run is unaffected.
179 if metadata.get("budget_exhausted"):
180 lines.append("- ⚠️ run budget exhausted: some phases were skipped")
181 skipped = metadata.get("skipped") or []
182 if skipped:
183 # bolt: CPython optimization — list comprehension inside join avoids generator overhead
184 names = ", ".join(
185 [f"{flatten_inline(s['name'])} ({flatten_inline(s['reason'])})" for s in skipped]
186 )
187 lines.append(f"- skipped agents (never ran): {names}")
188 # The zero-config fallback (#863): said in the report, since `--quiet` drops
189 # the stderr line and the exit code alone does not say why this run passed.
190 if (metadata.get("panel") or {}).get("zero_config_fallback"):
191 lines.append(
192 "- single local seat (zero-config fallback: no jury.toml and no agent CLI): "
193 "a single-vendor panel, so the default cross-vendor guard does not fail it; "
194 "a `--min-vendors N` named on the command line is still enforced"
195 )
196 retried = metadata.get("retried") or []
197 if retried:
198 lines.append(f"- retried agents: {', '.join(retried)}")
199 # A short panel is stated, not left to be inferred from the agent table (#501).
200 # Silence is the failure mode: a run with two non-reviewing slots described
201 # itself as a full panel, and only the chair's prose said otherwise.
202 panel = metadata.get("panel") or {}
203 if panel.get("short"):
204 lines.append(
205 f"- ⚠️ **effective panel: {panel.get('effective', 0)} of "
206 f"{panel.get('configured', 0)} reviewer(s)** "
207 f"({panel.get('abstained', 0)} returned no review, "
208 f"{panel.get('failed', 0)} failed) — "
209 f"{panel.get('vendors', 0)} vendor(s) contributed "
210 "(counted by vendor identity; the table below shows each seat's "
211 "configured vendor). An abstention is not "
212 "an approval; treat cross-vendor consensus accordingly."
213 )
214 # What a downstream consumer will actually be handed, and the chair's role in
215 # it (#699). Stated on every run, not only a short one: the number that got a
216 # tier-3 review refused was produced by a panel nothing had flagged as short.
217 # Presence, not truth: an all-silent panel supplies 0 reviews, and 0 is
218 # falsy, so a truthiness guard deleted the line in the one run where a
219 # reader most needs the number — while the run's own log still announced the
220 # seats it had. Zero is a count, and it is printed as one.
221 if panel.get("reviews_supplied") is not None:
222 chair_name = panel.get("chair") or ""
223 if panel.get("chair_ballot"):
224 role = f"chair `{chair_name}` also sat on the panel — its review is one of them"
225 elif chair_name:
226 role = f"chair `{chair_name}` supplied no review of its own — synthesis only"
227 else: # pragma: no cover - a run always resolves a chair
228 role = "no chair was resolved"
229 # The four ways a seat that ran supplies no review, named separately —
230 # they ask for different fixes, and a single "N did not review" would
231 # send a reader to the CLI logs for a reviewer that answered fine and
232 # simply reviewed nothing (#700, round 2). The phrases come from
233 # `panel.CAUSE_PHRASES`, the same table the shortfall message renders
234 # from, so this line and that one describe the same seat the same way;
235 # writing them apart is how "named nothing checkable and abstained" came
236 # to be printed over a ballot that had named a file and then refused
237 # (#700, round 5).
238 missing = [
239 f"{panel[panel_mod.PANEL_METADATA_KEYS[cause]]} {panel_mod.CAUSE_PHRASES[cause]}"
240 for cause in panel_mod.ABSTENTION_CAUSES
241 if panel.get(panel_mod.PANEL_METADATA_KEYS[cause])
242 ]
243 tail = f"; {', '.join(missing)}" if missing else ""
244 lines.append(
245 f"- reviews for a downstream consumer: {panel['reviews_supplied']} of "
246 f"{panel.get('ballots', 0)} ballot(s) (a ballot counts only when it names "
247 f"what it read and votes; the chair's synthesis record is carried alongside "
248 f"them and is not a review); {role}{tail}"
249 )
250 total = metadata["total_wall_clock_s"]
251 lines.append(f"- total wall-clock (cost proxy, not $): {total:.0f}s")
252 lines.append("")
253 # The `vendor` column is PROVENANCE: the string the operator configured,
254 # verbatim, even when the gate counts the seat as `cli` (#701). The count in
255 # the short-panel line above is the gate's arithmetic, and says so.
256 lines.append("| agent | vendor | status | duration |")
257 lines.append("| --- | --- | --- | --- |")
258 for a in metadata["agents"]:
259 code = a.get("error_code")
260 status = a["status"] if not code else f"{a['status']} ({code})"
261 # "ok" alone hid a slot that returned nothing reviewable (#501).
262 review = a.get("review_status")
263 if review and review != "findings":
264 status = f"{status}, {review}"
265 # Note a retried agent inline; attempts == 1 leaves the row unchanged.
266 attempts = a.get("attempts", 1)
267 if attempts and attempts > 1:
268 status += f", {attempts} attempts"
269 lines.append(
270 f"| {flatten_inline(a['name'])} | {flatten_inline(a['vendor'])} "
271 f"| {status} | {a['duration_s']:.0f}s |"
272 )
273 lines.append("")
274 economics = metadata.get("economics")
275 if economics and economics.get("breakdown"):
276 lines.append("### 💰 Run Economics (estimated)\n")
277 lines.append(
278 f"- total tokens (est): ~{economics['total_tokens_est']:,} · "
279 f"cost (est): ~${economics['total_cost_usd_est']:.4f} USD"
280 )
281 if economics.get("local_free_slots"):
282 lines.append(
283 f"- ⚡ {economics['local_free_slots']} slot(s) powered by local models ($0.00 free offline)"
284 )
285 lines.append("")
286 lines.append(
287 "_Wall-clock seconds and token counts are approximate cost proxies (no direct billing "
288 "telemetry is extracted from CLIs), not guaranteed dollar costs._\n"
289 )
290 return lines
293def _classification_block(classification: dict) -> list[str]:
294 """Render the compact PR-level classification summary.
296 Deterministic: ``classification`` is produced by the pure
297 :mod:`ai_jury.classification` module, so the rendered section is
298 stable for a deterministic run (and golden-tested under mock).
299 """
300 return [
301 "## Classification\n",
302 _classification.summary_line(classification),
303 "",
304 ]
307def _consensus_block(groups) -> list[str]:
308 lines = ["## Consensus\n"]
309 by_bucket: dict[str, list] = {b: [] for b in _BUCKET_ORDER}
310 for g in groups:
311 by_bucket.setdefault(g.bucket, []).append(g)
312 for bucket in _BUCKET_ORDER:
313 bg = by_bucket.get(bucket) or []
314 if not bg:
315 continue
316 lines.append(f"### {_BUCKET_LABELS.get(bucket, bucket)}\n")
317 for g in bg:
318 lines.append(_group_line(g))
319 lines.append("")
320 return lines
323def _vote_block(vote) -> list[str]:
324 """Render the panel-vote verdict + tally + per-reviewer ballots (issue #220).
326 Vocabulary-agnostic: the tally renders whatever stances the vote carries
327 (code: REQUEST CHANGES/COMMENT/APPROVE; issue: NEEDS-INFO/UNCLEAR/READY).
328 """
329 lines = ["## Verdict — panel vote\n"]
330 # bolt: CPython optimization — list comprehension avoids generator expression overhead
331 tally = " · ".join([f"{n} {label.lower()}" for label, n in vote.tally.items()])
332 lines.append(f"**{vote.verdict}** — {tally}\n")
333 for b in vote.ballots:
334 lines.append(f"- `{b.reviewer}`: **{b.vote}** ({b.reason})")
335 lines.append("")
336 return lines
339def _verdict_headline(synthesis, vote) -> str | None:
340 """One-line verdict for the report's TL;DR callout (pure, deterministic).
342 Prefers the panel vote's verdict when voting; otherwise lifts the opening
343 ``## Verdict`` line out of the chair's synthesis prose — both the code and
344 issue synthesis prompts mandate a ``## Verdict\\n<LABEL> — <one sentence>``
345 first section, so the lift is reliable. The verdict sentence may wrap across
346 lines; they are joined into one. Returns ``None`` when neither source is
347 available (failed/absent synthesis, deviating output) so the caller simply
348 omits the callout — it is purely additive, never replacing a section.
349 """
350 if vote is not None and getattr(vote, "verdict", None):
351 return vote.verdict
352 if synthesis is None or not getattr(synthesis, "ok", False):
353 return None
354 rows = (synthesis.output or "").splitlines()
355 for i, row in enumerate(rows):
356 if row.strip().lower().lstrip("#").strip() == "verdict":
357 collected: list[str] = []
358 for nxt in rows[i + 1 :]:
359 if nxt.strip().startswith("#"):
360 break
361 if not nxt.strip():
362 if collected:
363 break
364 continue
365 collected.append(nxt.strip())
366 return " ".join(collected) or None
367 return None
370def render(
371 reviews: list[AgentResult],
372 debate: list[AgentResult],
373 synthesis: AgentResult | None,
374 *,
375 chair: str,
376 findings: list[Finding] | None = None,
377 warnings: list[str] | None = None,
378 groups: list | None = None,
379 verify: AgentResult | None = None,
380 context_mode: str | None = None,
381 redact_secrets: bool | None = None,
382 redaction_count: int = 0,
383 metadata: dict | None = None,
384 classification: dict | None = None,
385 review_scope: str | None = None,
386 vote=None,
387 footer: bool = True,
388) -> str:
389 findings = findings or []
390 warnings = warnings or []
391 groups = groups or []
392 lines: list[str] = []
393 lines.append("# 🏛️ AI Jury\n")
395 # TL;DR callout (issue: scannable headline): hoist the verdict to the very
396 # top so the outcome is the first thing a reader sees, before the panel and
397 # the full report. Purely additive — omitted when no verdict is available.
398 headline = _verdict_headline(synthesis, vote)
399 if headline:
400 lines.append(f"> ⚡ **TL;DR · {headline}**\n")
402 # bolt: Explicit list materialization lets join evaluate iteratively in C
403 panel = ", ".join([f"`{r.agent}` ({r.vendor})" for r in reviews])
404 lines.append(f"**Panel:** {panel}\n")
406 # Review-scope note (issue #9): only rendered when the caller supplies it
407 # (incremental mode), so the default report is unchanged.
408 if review_scope:
409 lines.append(f"{review_scope}\n")
411 # Compact, deterministic PR-level classification (issue #7). Derived from the
412 # structured findings/groups when not supplied explicitly so the section
413 # always renders for a normal run.
414 if classification is None:
415 classification = _classification.classify(findings=findings, groups=groups)
416 lines.extend(_classification_block(classification))
418 if context_mode is not None or redact_secrets is not None:
419 lines.append("## Context policy\n")
420 if context_mode is not None:
421 lines.append(f"- context mode: {context_mode}")
422 if redact_secrets is not None:
423 state = "on" if redact_secrets else "off"
424 extra = f" ({redaction_count} redacted)" if redact_secrets else ""
425 lines.append(f"- secret redaction: {state}{extra}")
426 lines.append("")
428 if groups:
429 lines.extend(_consensus_block(groups))
430 lines.append("---\n")
432 # Panel-vote verdict (issue #220): when voting, the tally is the headline
433 # verdict and the chair's synthesis becomes supporting reasoning.
434 if vote is not None:
435 lines.extend(_vote_block(vote))
436 lines.append("---\n")
438 if verify is not None:
439 lines.append("## Verification\n")
440 lines.append(f"> Verified by `{chair}`\n")
441 if verify.ok:
442 lines.append(_defuse_patch_syntax(verify.output).strip() + "\n")
443 else:
444 lines.append(f"_Verification failed: {flatten_inline(verify.error)}_\n")
445 lines.append("---\n")
447 chair_heading = "Chair's reasoning" if vote is not None else "Chair verdict"
448 if synthesis and synthesis.ok:
449 lines.append(f"## {chair_heading}\n")
450 lines.append(f"> Synthesized by `{chair}`\n")
451 lines.append(_defuse_patch_syntax(synthesis.output).strip() + "\n")
452 elif synthesis and not synthesis.ok:
453 lines.append(f"## {chair_heading}\n")
454 lines.append(f"_Synthesis failed: {flatten_inline(synthesis.error)}_\n")
456 lines.append("---\n")
457 lines.append("## Structured findings\n")
458 if findings:
459 # ``f.file``/``f.line`` may be None (a finding need not be located), so
460 # coerce in the sort key — comparing None against str/int raises TypeError.
461 ranked = sorted(
462 findings,
463 key=lambda f: (SEVERITY_ORDER.get(f.severity, 99), f.file or "", f.line or 0),
464 )
465 for f in ranked:
466 lines.append(_finding_line(f))
467 lines.append("")
468 else:
469 lines.append("_(no structured findings parsed)_\n")
471 if warnings:
472 lines.append("> ⚠️ agent output warnings\n")
473 for w in warnings:
474 lines.append(f"- {w}")
475 lines.append("")
477 lines.append("## Round 1 — independent reviews\n")
478 for r in reviews:
479 status = f"{r.duration_s:.0f}s" if r.ok else _fail_status(r)
480 lines.append(_block(f"`{r.agent}` ({r.vendor}) — {status}", r.output if r.ok else ""))
482 if debate:
483 lines.append("## Round 2 — cross-examination\n")
484 for r in debate:
485 status = f"{r.duration_s:.0f}s" if r.ok else _fail_status(r)
486 lines.append(_block(f"`{r.agent}` — {status}", r.output if r.ok else ""))
488 if metadata is not None:
489 lines.append("---\n")
490 lines.extend(_metadata_block(metadata))
492 if footer:
493 lines.append(render_footer(reviews))
494 return "\n".join(lines)
497_LIVE_LABELS = {
498 "review": "Round 1 review",
499 "debate": "Cross-examination",
500 "verify": "Verification",
501 "synthesis": "Decision — verdict & reasoning",
502}
505def render_live_step(
506 kind: str, result: AgentResult, round_no: int | None = None
507) -> tuple[str, str]:
508 """Format one streamed step as ``(title, body)`` for live output (issue #210).
510 Pure — no I/O. The CLI ``--live`` handler prints this to stdout and (with
511 ``--pr``) posts it as its own comment, as each step completes. ``kind`` is one
512 of review / debate / verify / synthesis."""
513 label = _LIVE_LABELS.get(kind, kind)
514 if kind == "debate" and round_no:
515 label = f"Cross-examination · round {round_no}"
516 if kind in ("verify", "synthesis"):
517 who = f"chair `{result.agent}`"
518 else:
519 who = f"`{result.agent}` ({result.vendor})"
520 status = f"{result.duration_s:.0f}s" if result.ok else _fail_status(result)
521 title = f"🏛️ AI Jury — {label}: {who} — {status}"
522 body = _defuse_patch_syntax(result.output).strip() if result.ok else ""
523 return title, (body or "_(no output)_")
526def _conversation_blocks(
527 reviews: list[AgentResult],
528 debate: list[AgentResult],
529 synthesis: AgentResult | None,
530 verify: AgentResult | None,
531 *,
532 chair: str,
533) -> list[str]:
534 """The chronological deliberation, foregrounded: each reviewer's raw output,
535 then the debate exchanges in order, then verification, then the chair's
536 decision *and its reasoning* — so a reader can follow who said what and why
537 the chair ruled as it did (issue: full transcript)."""
538 lines: list[str] = ["## Round 1 — independent reviews\n"]
539 for r in reviews:
540 status = f"{r.duration_s:.0f}s" if r.ok else _fail_status(r)
541 lines.append(_block(f"`{r.agent}` ({r.vendor}) — {status}", r.output if r.ok else ""))
542 if debate:
543 lines.append("## Round 2 — cross-examination (debate)\n")
544 for r in debate:
545 status = f"{r.duration_s:.0f}s" if r.ok else _fail_status(r)
546 lines.append(_block(f"`{r.agent}` — {status}", r.output if r.ok else ""))
547 if verify is not None:
548 lines.append("## Verification\n")
549 lines.append(f"> Verified by `{chair}`\n")
550 lines.append(
551 _defuse_patch_syntax(verify.output).strip() + "\n"
552 if verify.ok
553 else f"_Verification failed: {flatten_inline(verify.error)}_\n"
554 )
555 lines.append("## Decision — verdict & reasoning\n")
556 if synthesis and synthesis.ok:
557 lines.append(f"> Decided by `{chair}`\n")
558 lines.append(_defuse_patch_syntax(synthesis.output).strip() + "\n")
559 elif synthesis and not synthesis.ok:
560 lines.append(f"_Synthesis failed: {flatten_inline(synthesis.error)}_\n")
561 else:
562 lines.append("_(no synthesis produced)_\n")
563 return lines
566def _summary_blocks(
567 findings: list[Finding],
568 warnings: list[str],
569 groups: list,
570 classification: dict,
571 vote=None,
572) -> list[str]:
573 """Consensus + structured-findings recap (the auditable at-a-glance summary)."""
574 lines = list(_classification_block(classification))
575 if vote is not None:
576 lines.extend(_vote_block(vote))
577 if groups:
578 lines.extend(_consensus_block(groups))
579 lines.append("## Structured findings\n")
580 if findings:
581 ranked = sorted(
582 findings,
583 key=lambda f: (SEVERITY_ORDER.get(f.severity, 99), f.file or "", f.line or 0),
584 )
585 # bolt: CPython optimization — list comprehension instead of generator expressions in extend()
586 lines.extend([_finding_line(f) for f in ranked])
587 lines.append("")
588 else:
589 lines.append("_(no structured findings parsed)_\n")
590 if warnings: 590 ↛ 591line 590 didn't jump to line 591 because the condition on line 590 was never true
591 lines.append("> ⚠️ agent output warnings\n")
592 lines.extend([f"- {w}" for w in warnings])
593 lines.append("")
594 return lines
597def render_transcript(
598 reviews: list[AgentResult],
599 debate: list[AgentResult],
600 synthesis: AgentResult | None,
601 *,
602 chair: str,
603 findings: list[Finding] | None = None,
604 warnings: list[str] | None = None,
605 groups: list | None = None,
606 verify: AgentResult | None = None,
607 context_mode: str | None = None,
608 redact_secrets: bool | None = None,
609 redaction_count: int = 0,
610 metadata: dict | None = None,
611 classification: dict | None = None,
612 review_scope: str | None = None,
613 lead_with_summary: bool = False,
614 vote=None,
615 footer: bool = True,
616) -> str:
617 """Render the full play-by-play transcript (issue: full transcript / --verbose).
619 Two layouts from one function:
621 * ``lead_with_summary=False`` (``--transcript``) — a dedicated, conversation-first
622 document: Round 1 → debate → verification → the chair's decision & reasoning,
623 then a compact consensus/findings recap for auditability.
624 * ``lead_with_summary=True`` (``--verbose``) — the consensus/verdict summary first,
625 then the same full transcript below it, in one document.
627 The default :func:`render` (consensus-first summary with a raw appendix) is
628 unchanged, so existing reports/goldens are unaffected.
629 """
630 findings = findings or []
631 warnings = warnings or []
632 groups = groups or []
633 if classification is None:
634 classification = _classification.classify(findings=findings, groups=groups)
636 lines: list[str] = []
637 lines.append(
638 "# 🏛️ AI Jury — verbose report\n" if lead_with_summary else "# 🏛️ AI Jury — full transcript\n"
639 )
640 # TL;DR callout (parity with render()): the verdict headline leads the
641 # verbose/transcript report too, so every renderer surfaces the outcome first.
642 headline = _verdict_headline(synthesis, vote)
643 if headline:
644 lines.append(f"> ⚡ **TL;DR · {headline}**\n")
645 # bolt: explicit list enables optimized string join bypassing generator loop overhead
646 panel = ", ".join([f"`{r.agent}` ({r.vendor})" for r in reviews])
647 lines.append(f"**Panel:** {panel}\n")
648 if review_scope:
649 lines.append(f"{review_scope}\n")
651 # Disclose the context/redaction policy (parity with render()): whoever reads
652 # the shared transcript should see whether secrets were redacted before the
653 # diff reached the agents.
654 if context_mode is not None or redact_secrets is not None:
655 lines.append("## Context policy\n")
656 if context_mode is not None:
657 lines.append(f"- context mode: {context_mode}")
658 if redact_secrets is not None:
659 state = "on" if redact_secrets else "off"
660 extra = f" ({redaction_count} redacted)" if redact_secrets else ""
661 lines.append(f"- secret redaction: {state}{extra}")
662 lines.append("")
664 if lead_with_summary:
665 lines.extend(_summary_blocks(findings, warnings, groups, classification, vote=vote))
666 lines.append("---\n")
667 lines.append("# Full transcript\n")
668 lines.extend(_conversation_blocks(reviews, debate, synthesis, verify, chair=chair))
669 else:
670 lines.extend(_conversation_blocks(reviews, debate, synthesis, verify, chair=chair))
671 lines.append("---\n")
672 lines.extend(_summary_blocks(findings, warnings, groups, classification, vote=vote))
674 if metadata is not None:
675 lines.append("---\n")
676 lines.extend(_metadata_block(metadata))
678 if footer:
679 lines.append(render_footer(reviews, transcript=True))
680 return "\n".join(lines)
683def render_sections(
684 reviews: list[AgentResult],
685 debate: list[AgentResult],
686 synthesis: AgentResult | None,
687 *,
688 chair: str,
689 findings: list[Finding] | None = None,
690 warnings: list[str] | None = None,
691 groups: list | None = None,
692 verify: AgentResult | None = None,
693 classification: dict | None = None,
694 vote=None,
695) -> list[tuple[str, str]]:
696 """Split the report into ordered ``(title, body)`` sections for phased posting.
698 Returns up to three sections — **Round 1** (independent reviews), **Round 2**
699 (debate, omitted when there was none), and **Decision** (verification + chair
700 verdict + consensus + structured findings) — so a PR can show the flow as
701 separate, readable comments (issue #127). ``render()`` (the single-blob
702 report) is unchanged. Empty sections are skipped.
703 """
704 findings = findings or []
705 warnings = warnings or []
706 groups = groups or []
707 sections: list[tuple[str, str]] = []
709 # Round 1 — independent reviews.
710 # bolt: Explicitly evaluating as a list allows C-level optimizations in join
711 r1 = [f"**Panel:** {', '.join([f'`{r.agent}` ({r.vendor})' for r in reviews])}\n"]
712 for r in reviews:
713 status = f"{r.duration_s:.0f}s" if r.ok else _fail_status(r)
714 r1.append(_block(f"`{r.agent}` ({r.vendor}) — {status}", r.output if r.ok else ""))
715 sections.append(("🏛️ AI Jury — Round 1: independent reviews", "\n".join(r1).strip()))
717 # Round 2 — cross-examination (only if a debate ran).
718 if debate:
719 r2 = []
720 for r in debate:
721 status = f"{r.duration_s:.0f}s" if r.ok else _fail_status(r)
722 r2.append(_block(f"`{r.agent}` — {status}", r.output if r.ok else ""))
723 sections.append(("🏛️ AI Jury — Round 2: cross-examination (debate)", "\n".join(r2).strip()))
725 # Decision — verification + chair verdict + consensus + findings.
726 dec: list[str] = []
727 if classification is None:
728 classification = _classification.classify(findings=findings, groups=groups)
729 dec.extend(_classification_block(classification))
730 if vote is not None:
731 dec.extend(_vote_block(vote))
732 if groups:
733 dec.extend(_consensus_block(groups))
734 if verify is not None:
735 dec.append("## Verification\n")
736 dec.append(f"> Verified by `{chair}`\n")
737 dec.append(
738 _defuse_patch_syntax(verify.output).strip() + "\n"
739 if verify.ok
740 else f"_Verification failed: {flatten_inline(verify.error)}_\n"
741 )
742 chair_heading = "Chair's reasoning" if vote is not None else "Chair verdict"
743 if synthesis and synthesis.ok:
744 dec.append(f"## {chair_heading}\n")
745 dec.append(f"> Synthesized by `{chair}`\n")
746 dec.append(_defuse_patch_syntax(synthesis.output).strip() + "\n")
747 elif synthesis and not synthesis.ok:
748 dec.append(f"## {chair_heading}\n\n_Synthesis failed: {flatten_inline(synthesis.error)}_\n")
749 if findings:
750 dec.append("## Structured findings\n")
751 ranked = sorted(
752 findings,
753 key=lambda f: (SEVERITY_ORDER.get(f.severity, 99), f.file or "", f.line or 0),
754 )
755 # bolt: Optimizes speed by allowing Python C implementations of extend()
756 dec.extend([_finding_line(f) for f in ranked])
757 if warnings:
758 dec.append("\n> ⚠️ agent output warnings\n")
759 dec.extend([f"- {w}" for w in warnings])
760 sections.append(("🏛️ AI Jury — Decision: verdict & consensus", "\n".join(dec).strip()))
762 return sections