Coverage for src/ai_jury/formats.py: 100%
45 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Machine-readable renderers for the jury outcome.
3Markdown rendering lives in :mod:`ai_jury.report`. This module
4adds structured outputs intended for tooling:
6* :func:`to_json` -- a structured JSON report (schema documented in the README).
7* :func:`to_sarif` -- a SARIF 2.1.0 document for CI / code scanning upload.
8* :func:`to_keel_reviews` -- a per-reviewer review bundle (issue #663) for a
9 consumer that renders one head-pinned verdict per panelist.
11Both renderers are deterministic for a deterministic outcome (e.g. under
12``mock=True``) and only emit legitimate finding fields -- never raw diff or
13prompt text.
14"""
16from __future__ import annotations
18import json
19from typing import Any
21from . import __version__
22from .findings import SEVERITIES, Finding
23from .metadata import build_run_metadata
25#: Version of the JSON report schema produced by :func:`to_json`.
26#:
27#: 1.1 (issue #663) ADDED the top-level ``reviewers`` array. Nothing was removed,
28#: renamed or reshaped, so every existing consumer of ``findings``/``consensus``/
29#: ``verdicts``/``verdict``/``metadata`` reads an identical document.
30#:
31#: 1.2 (issue #700) adds ``scope``, ``testing``, ``model_source``,
32#: ``scope_substantive`` and ``counts_as_review`` to each ``reviewers`` entry —
33#: and, the reason this is a version bump rather than a silent addition, CHANGES
34#: two things a consumer may have keyed on:
35#:
36#: * ``reviewers[].model`` was "the configured model id, or ``""`` when the CLI's
37#: default is in force"; it is now never empty for a slot that has an agent,
38#: carrying either the id actually requested or a statement that the CLI chose
39#: and does not report which. A consumer testing ``model == ""`` to detect the
40#: default case must read ``model_source == "cli_default"`` instead.
41#: * the ``reviewers`` array now carries one entry per seat that **ran**, not per
42#: seat that returned output: a silent agent is recorded as an abstention
43#: naming it rather than dropped. A consumer counting the non-``chair`` entries
44#: as reviews must read ``counts_as_review`` (equivalently:
45#: ``scope_substantive`` and ``verdict != "ABSTAIN"``), which is the same rule
46#: its own ``review-verdict-insubstantial`` gate applies.
47#:
48#: 1.3 (issues #709, #710) changes nothing structurally and two things
49#: semantically, which is why it is a bump rather than a silent correction:
50#:
51#: * ``reviewers[].model`` under ``model_source: "requested"`` is now the id the
52#: invocation actually sent, read back off the result the adapter produced. It
53#: was recomputed from the seat's ``vendor`` while every invocation path
54#: computes it from the seat's ``adapter``, so on a seat where those differ
55#: (possible since #705) the ballot could name a model the run never asked for
56#: — under a field whose whole claim is that it is the id that was sent.
57#: * ``reviewers[].model_source`` gains a fifth value, ``recomputed``: no
58#: invocation recorded an id for this slot, so the one in ``model`` was derived
59#: from the run's config. It used to ship as ``requested``, which is the same
60#: overstatement one source short — the derivation returns
61#: ``gemini-3-pro-high`` for the seat above, whose adapter sent
62#: ``gemini-3-pro``. A consumer switching on the four known values must accept
63#: a fifth.
64#: * ``reviewers[].abstention_cause`` gains a fifth value, ``not_in_change``: the
65#: seat stated a scope and none of it is in the change under review. Those
66#: ballots were ``named_nothing`` before, when they were recorded at all — a
67#: ``Checked:`` line naming anything at all used to make the ballot a **review**
68#: (#710), so some of these entries were previously counted rather than
69#: bucketed. A consumer switching on the four known values must accept a fifth.
70#:
71#: Every top-level key, and every ``reviewers`` key, keeps its name and shape.
72#:
73#: 1.4 (issue #714) embeds metadata schema 7, which added ``metadata.routing``.
74#:
75#: 1.5 (issue #863) embeds metadata schema 8, which adds
76#: ``metadata.panel.zero_config_fallback``. Additive: nothing else changes.
77JSON_SCHEMA_VERSION = "1.5"
79#: Canonical SARIF schema URI and version emitted by :func:`to_sarif`.
80SARIF_SCHEMA = "https://json.schemastore.org/sarif-2.1.0.json"
81SARIF_VERSION = "2.1.0"
83TOOL_NAME = "ai-jury"
84TOOL_URI = "https://github.com/berkayturanci/ai-jury"
86#: Mapping from finding severity to SARIF result level.
87_SARIF_LEVEL = {
88 "critical": "error",
89 "major": "error",
90 "minor": "warning",
91 "nit": "note",
92 "info": "note",
93}
96def severity_to_sarif_level(severity: str) -> str:
97 """Map a jury severity to a SARIF result ``level``.
99 critical/major -> ``error``, minor -> ``warning``, nit/info -> ``note``.
100 Unknown severities fall back to ``note``.
101 """
102 return _SARIF_LEVEL.get(severity, "note")
105def _finding_dict(f: Finding) -> dict[str, Any]:
106 """Serialise a finding to a stable, ordered dict of legitimate fields."""
107 return {
108 "severity": f.severity,
109 "file": f.file,
110 "line": f.line,
111 "claim": f.claim,
112 "evidence": f.evidence,
113 "suggested_fix": f.suggested_fix,
114 "confidence": f.confidence,
115 "reviewer": f.reviewer,
116 }
119def _group_dict(g: Any) -> dict[str, Any]:
120 """Serialise a consensus group to a stable, ordered dict."""
121 return {
122 "representative": _finding_dict(g.representative),
123 "agreement": len(g.reviewers),
124 "reviewers": list(g.reviewers),
125 "bucket": g.bucket,
126 "verification_status": g.status or None,
127 }
130def to_json(
131 outcome: Any,
132 config: Any,
133 *,
134 decision=None,
135 vote=None,
136 mode: str = "code",
137 zero_config_fallback: bool = False,
138) -> str:
139 """Render the jury outcome as a structured, pretty-printed JSON report.
141 Top-level keys: ``schema_version``, ``metadata`` (from
142 :func:`build_run_metadata`), ``findings``, ``consensus``, ``reviewers``,
143 ``verdicts`` and ``verdict`` (the chair synthesis text, if any). The result is
144 deterministic for a deterministic outcome and contains only legitimate finding
145 fields.
147 ``decision``/``vote`` are threaded into the metadata so the JSON report
148 reflects an effective ``--decision vote`` override (issue #248); when omitted
149 the metadata falls back to ``config.decision`` as before. ``mode`` selects the
150 ballot vocabulary (``code`` → APPROVE/COMMENT/REQUEST_CHANGES, ``issue`` →
151 READY/UNCLEAR/NEEDS_INFO), matching ``--issue``. ``zero_config_fallback`` is
152 passed through to the metadata (#863).
153 """
154 synthesis = getattr(outcome, "synthesis", None)
155 verdict_text = ""
156 if synthesis is not None and getattr(synthesis, "ok", False):
157 verdict_text = (synthesis.output or "").strip()
159 # Drop the wall-clock timestamp so the report is deterministic for a
160 # deterministic run (matching report.py, which omits generated_at too).
161 metadata = build_run_metadata(
162 outcome,
163 config,
164 decision=decision,
165 vote=vote,
166 mode=mode,
167 zero_config_fallback=zero_config_fallback,
168 )
169 metadata.pop("generated_at", None)
171 # Surface the deterministic PR-level classification (issue #7) at the top
172 # level for easy machine consumption (it is also embedded in ``metadata``).
173 from .ballots import reviewer_ballots
174 from .classification import classify
176 doc: dict[str, Any] = {
177 "schema_version": JSON_SCHEMA_VERSION,
178 "metadata": metadata,
179 "classification": classify(outcome),
180 "findings": [_finding_dict(f) for f in outcome.findings],
181 "consensus": [_group_dict(g) for g in outcome.groups],
182 # Per-reviewer ballots (issue #663): who said what, with vendor/model
183 # provenance. Purely additive — every key above is unchanged.
184 "reviewers": reviewer_ballots(outcome, config, vote=vote, mode=mode),
185 "verdicts": [
186 {
187 "file": v.file,
188 "line": v.line,
189 "claim": v.claim,
190 "status": v.status,
191 "reasoning": v.reasoning,
192 }
193 for v in outcome.verdicts
194 ],
195 "verdict": verdict_text,
196 }
197 return json.dumps(doc, indent=2, sort_keys=False, ensure_ascii=False)
200def to_keel_reviews(outcome: Any, config: Any, *, vote=None, mode: str = "code") -> str:
201 """Render the panel as a per-reviewer review bundle (issue #663).
203 A JSON **array** — not an object — of ``{reviewer, verdict, scope, findings,
204 testing, vendor, model, model_source, counts_as_review}`` records, one per
205 seat that ran plus the chair as ``reviewer: "chair"``. That is the payload
206 keel's ``keel review --reviews <file>`` accepts, so a panel run can *be* the
207 review rather than merely inform one. A seat that returned nothing is present
208 as an abstention naming it, and ``counts_as_review`` is how a consumer tells
209 the reviews from the records of seats that did not review.
211 Deterministic for a deterministic outcome. Free text lifted from agent replies
212 (``scope``, ``testing``) is flattened and capped in :mod:`ai_jury.ballots`; no
213 diff text, prompt text or secret ever reaches this document.
214 """
215 from .ballots import keel_reviews
217 return json.dumps(
218 keel_reviews(outcome, config, vote=vote, mode=mode),
219 indent=2,
220 sort_keys=False,
221 ensure_ascii=False,
222 )
225def _sarif_result(f: Finding) -> dict[str, Any]:
226 """Map a finding to a SARIF result object."""
227 physical: dict[str, Any] = {"artifactLocation": {"uri": f.file or ""}}
228 # SARIF 2.1.0 requires ``region.startLine`` to be a positive (1-based)
229 # integer. A reviewer's structured output is attacker-influenced (a finding's
230 # ``line`` is parsed from agent JSON that can be steered by the diff), so a
231 # forged ``"line": 0`` / negative value would emit an invalid region and make
232 # GitHub code-scanning reject the WHOLE SARIF upload — suppressing every
233 # finding (denial-of-evidence). Drop the region in that case so the finding
234 # still surfaces at file level (security audit 2026-06-13 r5).
235 if f.line is not None and f.line >= 1:
236 physical["region"] = {"startLine": f.line}
237 return {
238 "ruleId": f"jury/{f.severity}",
239 "level": severity_to_sarif_level(f.severity),
240 "message": {"text": f.claim},
241 "locations": [{"physicalLocation": physical}],
242 }
245def to_sarif(outcome: Any, _config: Any) -> str:
246 """Render the jury outcome as a SARIF 2.1.0 document.
248 Consensus group representatives are preferred as the source of results; if
249 there are no groups the raw findings are used. Rules are derived from the
250 severities actually present. The output is deterministic.
251 """
252 if outcome.groups:
253 findings = [g.representative for g in outcome.groups]
254 else:
255 findings = list(outcome.findings)
257 # Rules: one per severity actually used, in canonical severity order.
258 used = {f.severity for f in findings}
259 rules = [
260 {
261 "id": f"jury/{sev}",
262 "name": f"jury-{sev}",
263 "shortDescription": {"text": f"{sev} finding reported by the review jury"},
264 "defaultConfiguration": {"level": severity_to_sarif_level(sev)},
265 }
266 for sev in SEVERITIES
267 if sev in used
268 ]
270 doc = {
271 "$schema": SARIF_SCHEMA,
272 "version": SARIF_VERSION,
273 "runs": [
274 {
275 "tool": {
276 "driver": {
277 "name": TOOL_NAME,
278 "informationUri": TOOL_URI,
279 "version": __version__,
280 "rules": rules,
281 }
282 },
283 "results": [_sarif_result(f) for f in findings],
284 }
285 ],
286 }
287 return json.dumps(doc, indent=2, sort_keys=False, ensure_ascii=False)