Coverage for src/ai_jury/consensus.py: 100%

104 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-30 06:29 +0000

1"""Deterministic consensus grouping of findings across reviewers. 

2 

3Groups findings that refer to the same underlying issue so the report can 

4distinguish issues raised by every reviewer (consensus) from those raised by a 

5single reviewer (single_reviewer). Grouping is fully deterministic: identical 

6input always produces identical output. 

7""" 

8 

9from __future__ import annotations 

10 

11import re 

12from dataclasses import dataclass, field 

13 

14from .config import normalise_vendor 

15from .findings import SEVERITY_ORDER, Finding 

16 

17# Buckets describing how broadly a finding was raised. 

18BUCKET_CONSENSUS = "consensus" 

19BUCKET_MAJORITY = "majority" 

20BUCKET_SINGLE = "single_reviewer" 

21# Verification-derived buckets (issue #3). 

22BUCKET_REJECTED = "rejected" 

23BUCKET_DISPUTED = "disputed" 

24 

25# Line proximity threshold: findings within this many lines are treated as the 

26# same location (when both have a line). 

27LINE_PROXIMITY = 3 

28# Token-set Jaccard threshold for treating two claims as the same. 

29JACCARD_THRESHOLD = 0.6 

30 

31 

32@dataclass 

33class FindingGroup: 

34 representative: Finding 

35 reviewers: list[str] = field(default_factory=list) 

36 severity: str = "info" 

37 members: list[Finding] = field(default_factory=list) 

38 bucket: str = BUCKET_SINGLE 

39 # Verification status (issue #3); empty until a verdict is attached. 

40 status: str = "" 

41 status_reasoning: str = "" 

42 

43 

44def _normalize_path(path, *, fold_case: bool = True) -> str: 

45 """Normalize a path for comparison. 

46 

47 Strips only a real leading ``./`` prefix — NOT ``str.lstrip("./")``, which 

48 removes a whole leading run of ``.``/``/`` and collides distinct paths 

49 (e.g. ``.github/x.yml`` vs ``github/x.yml``, ``../auth.py`` vs ``./auth.py``; 

50 audit 2026-06-13 r5/M). 

51 

52 ``fold_case`` lower-cases the result, which is fine for *grouping/dedup* 

53 (a finding's display location) but DANGEROUS for the CI gate: on a 

54 case-sensitive filesystem ``Config.py`` and ``config.py`` are different 

55 files, so a case-folded match would let a verifier ``unsupported`` verdict on 

56 a benign sibling swallow a real critical and pass the gate (audit 

57 2026-06-13 r6/M). ``orchestrator._verdict_matches_group`` therefore calls 

58 this with ``fold_case=False`` (case-exact, fail-closed). 

59 """ 

60 if not path: 

61 return "" 

62 p = str(path).strip().replace("\\", "/") 

63 while p.startswith("./"): 

64 p = p[2:] 

65 return p.lower() if fold_case else p 

66 

67 

68def _normalize_claim(claim) -> str: 

69 text = (claim or "").lower() 

70 text = re.sub(r"[^a-z0-9\s]+", " ", text) 

71 return re.sub(r"\s+", " ", text).strip() 

72 

73 

74def _tokens(claim_norm: str) -> set[str]: 

75 return set(claim_norm.split()) 

76 

77 

78def _jaccard(a: set[str], b: set[str]) -> float: 

79 if not a and not b: 

80 return 1.0 

81 if not a or not b: 

82 return 0.0 

83 inter = len(a & b) 

84 # Bolt ⚡: calculate union size without allocating a new set 

85 union = len(a) + len(b) - inter 

86 return inter / union if union else 0.0 

87 

88 

89def _same_location(a: Finding, b: Finding) -> bool: 

90 if _normalize_path(a.file) != _normalize_path(b.file): 

91 return False 

92 if a.line is None and b.line is None: 

93 return True 

94 if a.line is None or b.line is None: 

95 return False 

96 return abs(a.line - b.line) <= LINE_PROXIMITY 

97 

98 

99def _same_claim(a_norm: str, b_norm: str) -> bool: 

100 if a_norm == b_norm: 

101 return True 

102 return _jaccard(_tokens(a_norm), _tokens(b_norm)) >= JACCARD_THRESHOLD 

103 

104 

105def _sort_key(f: Finding): 

106 return ( 

107 _normalize_path(f.file), 

108 f.line if f.line is not None else -1, 

109 _normalize_claim(f.claim), 

110 ) 

111 

112 

113def _max_severity(findings) -> str: 

114 # Lower SEVERITY_ORDER index == more severe. 

115 best = None 

116 best_rank = None 

117 for f in findings: 

118 rank = SEVERITY_ORDER.get(f.severity, len(SEVERITY_ORDER)) 

119 if best_rank is None or rank < best_rank: 

120 best_rank = rank 

121 best = f.severity 

122 return best or "info" 

123 

124 

125def _classify(reviewer_count: int, distinct_reviewers: int) -> str: 

126 if reviewer_count > 0 and distinct_reviewers >= reviewer_count: 

127 return BUCKET_CONSENSUS 

128 if distinct_reviewers > 1: 

129 return BUCKET_MAJORITY 

130 return BUCKET_SINGLE 

131 

132 

133# Severity a demoted group is capped at — below the default CI ``fail_on`` 

134# threshold (critical/major) so it stops blocking by default but still shows. 

135_DEMOTED_SEVERITY = "minor" 

136 

137 

138def demote_local_only_groups( 

139 groups: list[FindingGroup], vendor_by_reviewer: dict[str, str] 

140) -> None: 

141 """Cap severity for a finding raised only by local/free-model reviewers (issue #442). 

142 

143 A numeric per-reviewer trust weight was rejected in favor of this categorical 

144 rule: it is auditable in one line ("local-only finding, demoted unless a cloud 

145 reviewer concurs"), where a coefficient invites silent drift. A group is 

146 demoted only when EVERY contributing reviewer resolves to vendor ``"local"``; 

147 a group with at least one cloud-vendor reviewer, or with no reviewers at all 

148 (e.g. an injected finding), is left untouched. Mutates ``groups`` in place and 

149 is idempotent — safe to call more than once on the same groups. 

150 """ 

151 for group in groups: 

152 if not group.reviewers: 

153 continue 

154 # Normalised, not compared raw: the mapping is built from configured 

155 # vendors, and "which vendor is this" is one question with one answer 

156 # everywhere (issue #701, round 3). 

157 if not all( 

158 normalise_vendor(vendor_by_reviewer.get(r, "")) == "local" for r in group.reviewers 

159 ): 

160 continue 

161 if ( 

162 SEVERITY_ORDER.get(group.severity, len(SEVERITY_ORDER)) 

163 < SEVERITY_ORDER[_DEMOTED_SEVERITY] 

164 ): 

165 group.severity = _DEMOTED_SEVERITY 

166 

167 

168def group_findings(findings, reviewer_count: int) -> list[FindingGroup]: 

169 """Group findings that describe the same issue. 

170 

171 Findings match when they share a normalized file path, are within 

172 ``LINE_PROXIMITY`` lines (or both have no line), and have an equal or 

173 sufficiently similar (token-set Jaccard) normalized claim. 

174 

175 Deterministic: findings are sorted before greedy grouping, and the resulting 

176 groups are returned in a stable order. 

177 """ 

178 ordered = sorted(findings, key=_sort_key) 

179 groups: list[FindingGroup] = [] 

180 group_claim_norms: list[str] = [] 

181 

182 for finding in ordered: 

183 f_claim = _normalize_claim(finding.claim) 

184 placed = False 

185 for idx, group in enumerate(groups): 

186 rep = group.representative 

187 if _same_location(finding, rep) and _same_claim(f_claim, group_claim_norms[idx]): 

188 group.members.append(finding) 

189 placed = True 

190 break 

191 if not placed: 

192 groups.append( 

193 FindingGroup( 

194 representative=finding, 

195 members=[finding], 

196 ) 

197 ) 

198 group_claim_norms.append(f_claim) 

199 

200 for group in groups: 

201 reviewers = sorted({m.reviewer for m in group.members if m.reviewer}) 

202 group.reviewers = reviewers 

203 group.severity = _max_severity(group.members) 

204 # Pick the most severe member as representative for display stability. 

205 group.representative = min( 

206 group.members, 

207 key=lambda m: ( 

208 SEVERITY_ORDER.get(m.severity, len(SEVERITY_ORDER)), 

209 _sort_key(m), 

210 ), 

211 ) 

212 group.bucket = _classify(reviewer_count, len(reviewers)) 

213 

214 # Stable, deterministic output order: severity, then location, then claim. 

215 groups.sort( 

216 key=lambda g: ( 

217 SEVERITY_ORDER.get(g.severity, len(SEVERITY_ORDER)), 

218 _normalize_path(g.representative.file), 

219 g.representative.line if g.representative.line is not None else -1, 

220 _normalize_claim(g.representative.claim), 

221 ) 

222 ) 

223 return groups