Coverage for src/ai_jury/consensus.py: 100%
104 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Deterministic consensus grouping of findings across reviewers.
3Groups findings that refer to the same underlying issue so the report can
4distinguish issues raised by every reviewer (consensus) from those raised by a
5single reviewer (single_reviewer). Grouping is fully deterministic: identical
6input always produces identical output.
7"""
9from __future__ import annotations
11import re
12from dataclasses import dataclass, field
14from .config import normalise_vendor
15from .findings import SEVERITY_ORDER, Finding
17# Buckets describing how broadly a finding was raised.
18BUCKET_CONSENSUS = "consensus"
19BUCKET_MAJORITY = "majority"
20BUCKET_SINGLE = "single_reviewer"
21# Verification-derived buckets (issue #3).
22BUCKET_REJECTED = "rejected"
23BUCKET_DISPUTED = "disputed"
25# Line proximity threshold: findings within this many lines are treated as the
26# same location (when both have a line).
27LINE_PROXIMITY = 3
28# Token-set Jaccard threshold for treating two claims as the same.
29JACCARD_THRESHOLD = 0.6
32@dataclass
33class FindingGroup:
34 representative: Finding
35 reviewers: list[str] = field(default_factory=list)
36 severity: str = "info"
37 members: list[Finding] = field(default_factory=list)
38 bucket: str = BUCKET_SINGLE
39 # Verification status (issue #3); empty until a verdict is attached.
40 status: str = ""
41 status_reasoning: str = ""
44def _normalize_path(path, *, fold_case: bool = True) -> str:
45 """Normalize a path for comparison.
47 Strips only a real leading ``./`` prefix — NOT ``str.lstrip("./")``, which
48 removes a whole leading run of ``.``/``/`` and collides distinct paths
49 (e.g. ``.github/x.yml`` vs ``github/x.yml``, ``../auth.py`` vs ``./auth.py``;
50 audit 2026-06-13 r5/M).
52 ``fold_case`` lower-cases the result, which is fine for *grouping/dedup*
53 (a finding's display location) but DANGEROUS for the CI gate: on a
54 case-sensitive filesystem ``Config.py`` and ``config.py`` are different
55 files, so a case-folded match would let a verifier ``unsupported`` verdict on
56 a benign sibling swallow a real critical and pass the gate (audit
57 2026-06-13 r6/M). ``orchestrator._verdict_matches_group`` therefore calls
58 this with ``fold_case=False`` (case-exact, fail-closed).
59 """
60 if not path:
61 return ""
62 p = str(path).strip().replace("\\", "/")
63 while p.startswith("./"):
64 p = p[2:]
65 return p.lower() if fold_case else p
68def _normalize_claim(claim) -> str:
69 text = (claim or "").lower()
70 text = re.sub(r"[^a-z0-9\s]+", " ", text)
71 return re.sub(r"\s+", " ", text).strip()
74def _tokens(claim_norm: str) -> set[str]:
75 return set(claim_norm.split())
78def _jaccard(a: set[str], b: set[str]) -> float:
79 if not a and not b:
80 return 1.0
81 if not a or not b:
82 return 0.0
83 inter = len(a & b)
84 # Bolt ⚡: calculate union size without allocating a new set
85 union = len(a) + len(b) - inter
86 return inter / union if union else 0.0
89def _same_location(a: Finding, b: Finding) -> bool:
90 if _normalize_path(a.file) != _normalize_path(b.file):
91 return False
92 if a.line is None and b.line is None:
93 return True
94 if a.line is None or b.line is None:
95 return False
96 return abs(a.line - b.line) <= LINE_PROXIMITY
99def _same_claim(a_norm: str, b_norm: str) -> bool:
100 if a_norm == b_norm:
101 return True
102 return _jaccard(_tokens(a_norm), _tokens(b_norm)) >= JACCARD_THRESHOLD
105def _sort_key(f: Finding):
106 return (
107 _normalize_path(f.file),
108 f.line if f.line is not None else -1,
109 _normalize_claim(f.claim),
110 )
113def _max_severity(findings) -> str:
114 # Lower SEVERITY_ORDER index == more severe.
115 best = None
116 best_rank = None
117 for f in findings:
118 rank = SEVERITY_ORDER.get(f.severity, len(SEVERITY_ORDER))
119 if best_rank is None or rank < best_rank:
120 best_rank = rank
121 best = f.severity
122 return best or "info"
125def _classify(reviewer_count: int, distinct_reviewers: int) -> str:
126 if reviewer_count > 0 and distinct_reviewers >= reviewer_count:
127 return BUCKET_CONSENSUS
128 if distinct_reviewers > 1:
129 return BUCKET_MAJORITY
130 return BUCKET_SINGLE
133# Severity a demoted group is capped at — below the default CI ``fail_on``
134# threshold (critical/major) so it stops blocking by default but still shows.
135_DEMOTED_SEVERITY = "minor"
138def demote_local_only_groups(
139 groups: list[FindingGroup], vendor_by_reviewer: dict[str, str]
140) -> None:
141 """Cap severity for a finding raised only by local/free-model reviewers (issue #442).
143 A numeric per-reviewer trust weight was rejected in favor of this categorical
144 rule: it is auditable in one line ("local-only finding, demoted unless a cloud
145 reviewer concurs"), where a coefficient invites silent drift. A group is
146 demoted only when EVERY contributing reviewer resolves to vendor ``"local"``;
147 a group with at least one cloud-vendor reviewer, or with no reviewers at all
148 (e.g. an injected finding), is left untouched. Mutates ``groups`` in place and
149 is idempotent — safe to call more than once on the same groups.
150 """
151 for group in groups:
152 if not group.reviewers:
153 continue
154 # Normalised, not compared raw: the mapping is built from configured
155 # vendors, and "which vendor is this" is one question with one answer
156 # everywhere (issue #701, round 3).
157 if not all(
158 normalise_vendor(vendor_by_reviewer.get(r, "")) == "local" for r in group.reviewers
159 ):
160 continue
161 if (
162 SEVERITY_ORDER.get(group.severity, len(SEVERITY_ORDER))
163 < SEVERITY_ORDER[_DEMOTED_SEVERITY]
164 ):
165 group.severity = _DEMOTED_SEVERITY
168def group_findings(findings, reviewer_count: int) -> list[FindingGroup]:
169 """Group findings that describe the same issue.
171 Findings match when they share a normalized file path, are within
172 ``LINE_PROXIMITY`` lines (or both have no line), and have an equal or
173 sufficiently similar (token-set Jaccard) normalized claim.
175 Deterministic: findings are sorted before greedy grouping, and the resulting
176 groups are returned in a stable order.
177 """
178 ordered = sorted(findings, key=_sort_key)
179 groups: list[FindingGroup] = []
180 group_claim_norms: list[str] = []
182 for finding in ordered:
183 f_claim = _normalize_claim(finding.claim)
184 placed = False
185 for idx, group in enumerate(groups):
186 rep = group.representative
187 if _same_location(finding, rep) and _same_claim(f_claim, group_claim_norms[idx]):
188 group.members.append(finding)
189 placed = True
190 break
191 if not placed:
192 groups.append(
193 FindingGroup(
194 representative=finding,
195 members=[finding],
196 )
197 )
198 group_claim_norms.append(f_claim)
200 for group in groups:
201 reviewers = sorted({m.reviewer for m in group.members if m.reviewer})
202 group.reviewers = reviewers
203 group.severity = _max_severity(group.members)
204 # Pick the most severe member as representative for display stability.
205 group.representative = min(
206 group.members,
207 key=lambda m: (
208 SEVERITY_ORDER.get(m.severity, len(SEVERITY_ORDER)),
209 _sort_key(m),
210 ),
211 )
212 group.bucket = _classify(reviewer_count, len(reviewers))
214 # Stable, deterministic output order: severity, then location, then claim.
215 groups.sort(
216 key=lambda g: (
217 SEVERITY_ORDER.get(g.severity, len(SEVERITY_ORDER)),
218 _normalize_path(g.representative.file),
219 g.representative.line if g.representative.line is not None else -1,
220 _normalize_claim(g.representative.claim),
221 )
222 )
223 return groups