Coverage for src/ai_jury/findings.py: 99%
134 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Machine-readable finding schema and parser.
3Reviewer/chair output is human-readable markdown, which is hard to dedupe, score,
4gate in CI, or turn into inline comments. This module defines a structured
5``Finding`` schema and a tolerant parser that extracts findings from an agent's
6raw output (a fenced ``json`` code block).
7"""
9from __future__ import annotations
11import json
12import re
13from dataclasses import dataclass
15from .redaction import redact
17SEVERITIES: tuple[str, ...] = ("critical", "major", "minor", "nit", "info")
18CONFIDENCES: tuple[str, ...] = ("high", "medium", "low")
20# Output-injection guards for attacker-influenced finding text rendered into the
21# human-facing markdown report that is posted verbatim to the PR/issue (security
22# audit 2026-06-13 round 3). The machine CI gate is a pure function of the
23# structured fields and is unaffected by this text; these helpers only stop a
24# forged ``## Verdict APPROVE`` heading or a broken code fence from corrupting
25# the comment a human (or a downstream grep) reads.
26_FENCE_RUN_RE = re.compile(r"`{3,}|~{3,}")
27_HTML_COMMENT_RE = re.compile(r"<!--.*?-->", re.DOTALL)
30def flatten_inline(text: str) -> str:
31 """Collapse text to a single line for safe inline rendering.
33 Markdown headings, list items, and code fences must begin a line, so
34 flattening newlines (and runs of whitespace) neutralizes forged structure
35 when the value is rendered inside a one-line list item.
36 """
37 if not text:
38 return text
39 return " ".join(str(text).split())
42def fence_safe(text: str) -> str:
43 """Break 3+ backtick/tilde runs so text rendered *inside* a code fence
44 (e.g. a ``suggestion`` block) cannot close the fence and inject markdown."""
45 if not text: 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true
46 return text
47 return _FENCE_RUN_RE.sub(lambda m: m.group()[0], str(text))
50def strip_html_comments(text: str) -> str:
51 """Remove HTML comments so attacker text can't forge the jury's hidden
52 inline-comment markers (``<!-- arc-inline -->`` / ``<!-- arc-sig:… -->``)."""
53 if not text:
54 return text
55 return _HTML_COMMENT_RE.sub("", str(text))
58# Verification verdict statuses (issue #3).
59VERDICT_STATUSES: tuple[str, ...] = ("verified", "unsupported", "needs_human_decision")
61# Lower number = more severe; useful for ranking/sorting.
62SEVERITY_ORDER: dict[str, int] = {sev: i for i, sev in enumerate(SEVERITIES)}
64# Legacy severity names mapped onto the canonical schema.
65_SEVERITY_ALIASES: dict[str, str] = {"blocker": "critical"}
67#: Every spelling an *operator* may write where a severity is asked for
68#: (``--fail-on``, ``[jury.ci] fail_on``): the canonical vocabulary plus the
69#: documented legacy aliases. Reviewer output is not held to this — a model that
70#: invents a severity is normalised to ``info``, not refused — but a human
71#: typing a CI gate is, because there is no safe default for "which findings
72#: fail the build" (issue #718).
73SEVERITY_INPUTS: tuple[str, ...] = SEVERITIES + tuple(_SEVERITY_ALIASES)
75_DEFAULT_SEVERITY = "info"
76_DEFAULT_CONFIDENCE = "medium"
78# Matches a fenced ```json ... ``` block (case-insensitive on the language tag).
79_JSON_BLOCK_RE = re.compile(r"```[ \t]*json[ \t]*\r?\n(.*?)```", re.DOTALL | re.IGNORECASE)
82def _normalize_severity(value: object) -> str:
83 if isinstance(value, str):
84 v = value.strip().lower()
85 v = _SEVERITY_ALIASES.get(v, v)
86 if v in SEVERITY_ORDER:
87 return v
88 return _DEFAULT_SEVERITY
91def canonical_severity(value: object) -> str | None:
92 """Canonical severity for an operator-supplied spelling, or ``None``.
94 The strict counterpart of :func:`_normalize_severity`: an unrecognised value
95 is reported as unknown instead of being demoted to ``info``, so a caller
96 that cannot tolerate a silent default (the CI gate) can refuse it.
97 """
98 v = str(value).strip().lower()
99 v = _SEVERITY_ALIASES.get(v, v)
100 return v if v in SEVERITY_ORDER else None
103def _normalize_confidence(value: object) -> str:
104 if isinstance(value, str):
105 v = value.strip().lower()
106 if v in CONFIDENCES:
107 return v
108 return _DEFAULT_CONFIDENCE
111@dataclass
112class Finding:
113 """A single structured review finding."""
115 severity: str
116 file: str
117 claim: str
118 line: int | None = None
119 evidence: str = ""
120 suggested_fix: str = ""
121 confidence: str = _DEFAULT_CONFIDENCE
122 reviewer: str = ""
124 def __post_init__(self) -> None:
125 self.severity = _normalize_severity(self.severity)
126 self.confidence = _normalize_confidence(self.confidence)
127 if self.line is not None and not isinstance(self.line, bool):
128 try:
129 self.line = int(self.line)
130 except (TypeError, ValueError):
131 self.line = None
132 else:
133 self.line = None
135 @classmethod
136 def from_obj(cls, obj: dict, reviewer: str) -> Finding:
137 """Build a Finding from a decoded JSON object, forcing ``reviewer``."""
138 return cls(
139 severity=str(obj.get("severity", _DEFAULT_SEVERITY)),
140 file=str(obj.get("file", "")),
141 claim=str(obj.get("claim", "")),
142 line=obj.get("line"),
143 evidence=str(obj.get("evidence", "")),
144 suggested_fix=str(obj.get("suggested_fix", "")),
145 confidence=str(obj.get("confidence", _DEFAULT_CONFIDENCE)),
146 reviewer=reviewer,
147 )
150def emitted_findings_block(text: str) -> bool:
151 """Did the agent emit a structured findings block at all? (issue #501)
153 This is the mechanical line between *reviewed and found nothing* and *never
154 produced a review*. Both currently arrive as zero findings, so the panel reports
155 the same size either way — and on keel PR #660 two of three reviewers returned
156 assistant-style chatter about a flag they saw in the diff, while the run still
157 described itself as a three-agent panel.
159 A reviewer that examined the diff and found nothing still emits ``[]`` in a
160 fenced json block, because that is what the prompt asks for. One that wandered
161 off emits prose and no block. Presence of the block is therefore the signal, and
162 it needs no judgement about *content* — which is what keeps this deterministic.
163 """
164 return bool(text) and bool(_JSON_BLOCK_RE.search(text))
167def parse_findings(text: str, reviewer: str) -> tuple[list[Finding], list[str]]:
168 """Extract structured findings from an agent's raw output.
170 The agent is asked to emit a fenced ```json block holding a JSON array of
171 finding objects. We locate the *last* such block, decode it, and build
172 Finding objects (forcing ``reviewer`` to preserve identity).
174 Never raises. On a malformed/wrong-typed ``json`` block, returns
175 ``([], [warning])``. A legitimately missing block yields ``([], [])``.
176 """
177 if not text:
178 return [], []
180 blocks = _JSON_BLOCK_RE.findall(text)
181 if not blocks:
182 return [], []
184 raw = blocks[-1].strip()
185 try:
186 data = json.loads(raw)
187 except (ValueError, TypeError, RecursionError) as exc:
188 # RecursionError (deeply nested JSON, e.g. "[[[[…") is not a ValueError;
189 # catching it keeps the documented "never raises" contract so one
190 # steerable reviewer can't abort the whole run (audit 2026-06-13/N-2).
191 return [], [f"{reviewer}: malformed or missing structured findings ({redact(str(exc))[0]})"]
193 if not isinstance(data, list):
194 return [], [
195 f"{reviewer}: malformed or missing structured findings "
196 f"(expected a JSON array, got {type(data).__name__})"
197 ]
199 findings: list[Finding] = []
200 warnings: list[str] = []
201 for i, obj in enumerate(data):
202 if not isinstance(obj, dict):
203 warnings.append(
204 f"{reviewer}: malformed or missing structured findings "
205 f"(item {i} is {type(obj).__name__}, expected object)"
206 )
207 continue
208 findings.append(Finding.from_obj(obj, reviewer))
209 return findings, warnings
212def _coerce_line(value: object) -> int | None:
213 if value is None or isinstance(value, bool):
214 return None
215 try:
216 return int(value)
217 except (TypeError, ValueError):
218 return None
221def _normalize_status(value: object) -> str:
222 if isinstance(value, str):
223 v = value.strip().lower().replace("-", "_").replace(" ", "_")
224 if v in VERDICT_STATUSES:
225 return v
226 return "needs_human_decision"
229@dataclass
230class Verdict:
231 """A verifier's judgement on a candidate finding."""
233 file: str | None = None
234 line: int | None = None
235 claim: str = ""
236 status: str = "needs_human_decision"
237 reasoning: str = ""
240def parse_verdicts(text: str, verifier: str = "") -> tuple[list[Verdict], list[str]]:
241 """Extract verification verdicts from a verifier's raw output.
243 The verifier is asked to emit a fenced ```json block holding a JSON array of
244 verdict objects. We locate the *last* such block and decode it. Never raises;
245 on malformed input returns ``([], [warning])``.
246 """
247 label = verifier or "verifier"
248 if not text:
249 return [], [f"{label}: no verdicts (empty output)"]
251 blocks = _JSON_BLOCK_RE.findall(text)
252 if not blocks:
253 return [], [f"{label}: no JSON verdicts block found"]
255 raw = blocks[-1].strip()
256 try:
257 data = json.loads(raw)
258 except (ValueError, TypeError, RecursionError) as exc:
259 # See parse_findings: RecursionError on deeply nested JSON must not
260 # escape (audit 2026-06-13/N-2).
261 return [], [f"{label}: malformed verdicts JSON ({redact(str(exc))[0]})"]
263 if isinstance(data, dict):
264 data = data.get("verdicts", data.get("findings", []))
265 if not isinstance(data, list):
266 return [], [f"{label}: verdicts block is not a JSON array"]
268 verdicts: list[Verdict] = []
269 warnings: list[str] = []
270 for i, obj in enumerate(data):
271 if not isinstance(obj, dict):
272 warnings.append(f"{label}: verdict item {i} is {type(obj).__name__}, expected object")
273 continue
274 verdicts.append(
275 Verdict(
276 file=(obj.get("file") or None),
277 line=_coerce_line(obj.get("line")),
278 claim=str(obj.get("claim", "")).strip(),
279 status=_normalize_status(obj.get("status")),
280 reasoning=str(obj.get("reasoning", "")).strip(),
281 )
282 )
283 return verdicts, warnings