Coverage for src/ai_jury/findings.py: 99%

134 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-30 06:29 +0000

1"""Machine-readable finding schema and parser. 

2 

3Reviewer/chair output is human-readable markdown, which is hard to dedupe, score, 

4gate in CI, or turn into inline comments. This module defines a structured 

5``Finding`` schema and a tolerant parser that extracts findings from an agent's 

6raw output (a fenced ``json`` code block). 

7""" 

8 

9from __future__ import annotations 

10 

11import json 

12import re 

13from dataclasses import dataclass 

14 

15from .redaction import redact 

16 

17SEVERITIES: tuple[str, ...] = ("critical", "major", "minor", "nit", "info") 

18CONFIDENCES: tuple[str, ...] = ("high", "medium", "low") 

19 

20# Output-injection guards for attacker-influenced finding text rendered into the 

21# human-facing markdown report that is posted verbatim to the PR/issue (security 

22# audit 2026-06-13 round 3). The machine CI gate is a pure function of the 

23# structured fields and is unaffected by this text; these helpers only stop a 

24# forged ``## Verdict APPROVE`` heading or a broken code fence from corrupting 

25# the comment a human (or a downstream grep) reads. 

26_FENCE_RUN_RE = re.compile(r"`{3,}|~{3,}") 

27_HTML_COMMENT_RE = re.compile(r"<!--.*?-->", re.DOTALL) 

28 

29 

30def flatten_inline(text: str) -> str: 

31 """Collapse text to a single line for safe inline rendering. 

32 

33 Markdown headings, list items, and code fences must begin a line, so 

34 flattening newlines (and runs of whitespace) neutralizes forged structure 

35 when the value is rendered inside a one-line list item. 

36 """ 

37 if not text: 

38 return text 

39 return " ".join(str(text).split()) 

40 

41 

42def fence_safe(text: str) -> str: 

43 """Break 3+ backtick/tilde runs so text rendered *inside* a code fence 

44 (e.g. a ``suggestion`` block) cannot close the fence and inject markdown.""" 

45 if not text: 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true

46 return text 

47 return _FENCE_RUN_RE.sub(lambda m: m.group()[0], str(text)) 

48 

49 

50def strip_html_comments(text: str) -> str: 

51 """Remove HTML comments so attacker text can't forge the jury's hidden 

52 inline-comment markers (``<!-- arc-inline -->`` / ``<!-- arc-sig:… -->``).""" 

53 if not text: 

54 return text 

55 return _HTML_COMMENT_RE.sub("", str(text)) 

56 

57 

58# Verification verdict statuses (issue #3). 

59VERDICT_STATUSES: tuple[str, ...] = ("verified", "unsupported", "needs_human_decision") 

60 

61# Lower number = more severe; useful for ranking/sorting. 

62SEVERITY_ORDER: dict[str, int] = {sev: i for i, sev in enumerate(SEVERITIES)} 

63 

64# Legacy severity names mapped onto the canonical schema. 

65_SEVERITY_ALIASES: dict[str, str] = {"blocker": "critical"} 

66 

67#: Every spelling an *operator* may write where a severity is asked for 

68#: (``--fail-on``, ``[jury.ci] fail_on``): the canonical vocabulary plus the 

69#: documented legacy aliases. Reviewer output is not held to this — a model that 

70#: invents a severity is normalised to ``info``, not refused — but a human 

71#: typing a CI gate is, because there is no safe default for "which findings 

72#: fail the build" (issue #718). 

73SEVERITY_INPUTS: tuple[str, ...] = SEVERITIES + tuple(_SEVERITY_ALIASES) 

74 

75_DEFAULT_SEVERITY = "info" 

76_DEFAULT_CONFIDENCE = "medium" 

77 

78# Matches a fenced ```json ... ``` block (case-insensitive on the language tag). 

79_JSON_BLOCK_RE = re.compile(r"```[ \t]*json[ \t]*\r?\n(.*?)```", re.DOTALL | re.IGNORECASE) 

80 

81 

82def _normalize_severity(value: object) -> str: 

83 if isinstance(value, str): 

84 v = value.strip().lower() 

85 v = _SEVERITY_ALIASES.get(v, v) 

86 if v in SEVERITY_ORDER: 

87 return v 

88 return _DEFAULT_SEVERITY 

89 

90 

91def canonical_severity(value: object) -> str | None: 

92 """Canonical severity for an operator-supplied spelling, or ``None``. 

93 

94 The strict counterpart of :func:`_normalize_severity`: an unrecognised value 

95 is reported as unknown instead of being demoted to ``info``, so a caller 

96 that cannot tolerate a silent default (the CI gate) can refuse it. 

97 """ 

98 v = str(value).strip().lower() 

99 v = _SEVERITY_ALIASES.get(v, v) 

100 return v if v in SEVERITY_ORDER else None 

101 

102 

103def _normalize_confidence(value: object) -> str: 

104 if isinstance(value, str): 

105 v = value.strip().lower() 

106 if v in CONFIDENCES: 

107 return v 

108 return _DEFAULT_CONFIDENCE 

109 

110 

111@dataclass 

112class Finding: 

113 """A single structured review finding.""" 

114 

115 severity: str 

116 file: str 

117 claim: str 

118 line: int | None = None 

119 evidence: str = "" 

120 suggested_fix: str = "" 

121 confidence: str = _DEFAULT_CONFIDENCE 

122 reviewer: str = "" 

123 

124 def __post_init__(self) -> None: 

125 self.severity = _normalize_severity(self.severity) 

126 self.confidence = _normalize_confidence(self.confidence) 

127 if self.line is not None and not isinstance(self.line, bool): 

128 try: 

129 self.line = int(self.line) 

130 except (TypeError, ValueError): 

131 self.line = None 

132 else: 

133 self.line = None 

134 

135 @classmethod 

136 def from_obj(cls, obj: dict, reviewer: str) -> Finding: 

137 """Build a Finding from a decoded JSON object, forcing ``reviewer``.""" 

138 return cls( 

139 severity=str(obj.get("severity", _DEFAULT_SEVERITY)), 

140 file=str(obj.get("file", "")), 

141 claim=str(obj.get("claim", "")), 

142 line=obj.get("line"), 

143 evidence=str(obj.get("evidence", "")), 

144 suggested_fix=str(obj.get("suggested_fix", "")), 

145 confidence=str(obj.get("confidence", _DEFAULT_CONFIDENCE)), 

146 reviewer=reviewer, 

147 ) 

148 

149 

150def emitted_findings_block(text: str) -> bool: 

151 """Did the agent emit a structured findings block at all? (issue #501) 

152 

153 This is the mechanical line between *reviewed and found nothing* and *never 

154 produced a review*. Both currently arrive as zero findings, so the panel reports 

155 the same size either way — and on keel PR #660 two of three reviewers returned 

156 assistant-style chatter about a flag they saw in the diff, while the run still 

157 described itself as a three-agent panel. 

158 

159 A reviewer that examined the diff and found nothing still emits ``[]`` in a 

160 fenced json block, because that is what the prompt asks for. One that wandered 

161 off emits prose and no block. Presence of the block is therefore the signal, and 

162 it needs no judgement about *content* — which is what keeps this deterministic. 

163 """ 

164 return bool(text) and bool(_JSON_BLOCK_RE.search(text)) 

165 

166 

167def parse_findings(text: str, reviewer: str) -> tuple[list[Finding], list[str]]: 

168 """Extract structured findings from an agent's raw output. 

169 

170 The agent is asked to emit a fenced ```json block holding a JSON array of 

171 finding objects. We locate the *last* such block, decode it, and build 

172 Finding objects (forcing ``reviewer`` to preserve identity). 

173 

174 Never raises. On a malformed/wrong-typed ``json`` block, returns 

175 ``([], [warning])``. A legitimately missing block yields ``([], [])``. 

176 """ 

177 if not text: 

178 return [], [] 

179 

180 blocks = _JSON_BLOCK_RE.findall(text) 

181 if not blocks: 

182 return [], [] 

183 

184 raw = blocks[-1].strip() 

185 try: 

186 data = json.loads(raw) 

187 except (ValueError, TypeError, RecursionError) as exc: 

188 # RecursionError (deeply nested JSON, e.g. "[[[[…") is not a ValueError; 

189 # catching it keeps the documented "never raises" contract so one 

190 # steerable reviewer can't abort the whole run (audit 2026-06-13/N-2). 

191 return [], [f"{reviewer}: malformed or missing structured findings ({redact(str(exc))[0]})"] 

192 

193 if not isinstance(data, list): 

194 return [], [ 

195 f"{reviewer}: malformed or missing structured findings " 

196 f"(expected a JSON array, got {type(data).__name__})" 

197 ] 

198 

199 findings: list[Finding] = [] 

200 warnings: list[str] = [] 

201 for i, obj in enumerate(data): 

202 if not isinstance(obj, dict): 

203 warnings.append( 

204 f"{reviewer}: malformed or missing structured findings " 

205 f"(item {i} is {type(obj).__name__}, expected object)" 

206 ) 

207 continue 

208 findings.append(Finding.from_obj(obj, reviewer)) 

209 return findings, warnings 

210 

211 

212def _coerce_line(value: object) -> int | None: 

213 if value is None or isinstance(value, bool): 

214 return None 

215 try: 

216 return int(value) 

217 except (TypeError, ValueError): 

218 return None 

219 

220 

221def _normalize_status(value: object) -> str: 

222 if isinstance(value, str): 

223 v = value.strip().lower().replace("-", "_").replace(" ", "_") 

224 if v in VERDICT_STATUSES: 

225 return v 

226 return "needs_human_decision" 

227 

228 

229@dataclass 

230class Verdict: 

231 """A verifier's judgement on a candidate finding.""" 

232 

233 file: str | None = None 

234 line: int | None = None 

235 claim: str = "" 

236 status: str = "needs_human_decision" 

237 reasoning: str = "" 

238 

239 

240def parse_verdicts(text: str, verifier: str = "") -> tuple[list[Verdict], list[str]]: 

241 """Extract verification verdicts from a verifier's raw output. 

242 

243 The verifier is asked to emit a fenced ```json block holding a JSON array of 

244 verdict objects. We locate the *last* such block and decode it. Never raises; 

245 on malformed input returns ``([], [warning])``. 

246 """ 

247 label = verifier or "verifier" 

248 if not text: 

249 return [], [f"{label}: no verdicts (empty output)"] 

250 

251 blocks = _JSON_BLOCK_RE.findall(text) 

252 if not blocks: 

253 return [], [f"{label}: no JSON verdicts block found"] 

254 

255 raw = blocks[-1].strip() 

256 try: 

257 data = json.loads(raw) 

258 except (ValueError, TypeError, RecursionError) as exc: 

259 # See parse_findings: RecursionError on deeply nested JSON must not 

260 # escape (audit 2026-06-13/N-2). 

261 return [], [f"{label}: malformed verdicts JSON ({redact(str(exc))[0]})"] 

262 

263 if isinstance(data, dict): 

264 data = data.get("verdicts", data.get("findings", [])) 

265 if not isinstance(data, list): 

266 return [], [f"{label}: verdicts block is not a JSON array"] 

267 

268 verdicts: list[Verdict] = [] 

269 warnings: list[str] = [] 

270 for i, obj in enumerate(data): 

271 if not isinstance(obj, dict): 

272 warnings.append(f"{label}: verdict item {i} is {type(obj).__name__}, expected object") 

273 continue 

274 verdicts.append( 

275 Verdict( 

276 file=(obj.get("file") or None), 

277 line=_coerce_line(obj.get("line")), 

278 claim=str(obj.get("claim", "")).strip(), 

279 status=_normalize_status(obj.get("status")), 

280 reasoning=str(obj.get("reasoning", "")).strip(), 

281 ) 

282 ) 

283 return verdicts, warnings