Coverage for src/ai_jury/routing.py: 100%

81 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-30 06:29 +0000

1"""Risk-aware tiered routing: which seats review, which one anchors, who escalates. 

2 

3Issue #714 (the ask was #524). ``routing = "tiered"`` / ``--tiered`` is a cost 

4lever, and a cost lever that could silently weaken a review would be worse than 

5none, so every decision here is a pure function of things the operator wrote 

6and the diff itself, and every decision is reported: 

7 

8* The **risk band** is :func:`ai_jury.diffprofile.profile_diff` — the same 

9 classifier ``--auto`` uses — so a routed run and an auto-depth run agree on 

10 what "routine" means. 

11* The **tier** of a seat is ``[[agent]] tier`` (default ``frontier``). There are 

12 no model-name heuristics: a seat is economical because the operator said so. 

13* A ``high``-risk diff (security-sensitive paths, large, many files) gets the 

14 **full** enabled panel; nothing is saved on a change that can hurt. 

15* A ``low`` or ``medium`` diff gets every **economical** seat plus **one 

16 frontier anchor** — the configured chair when it is frontier and usable, else 

17 the first usable frontier seat in config order. The other frontier seats are 

18 *benched*: they run only if round 1 escalates. 

19* Two floors always hold, in this order: the panel keeps at least 

20 ``[jury.ci] min_vendors`` distinct vendors (counted by 

21 :func:`ai_jury.config.vendor_identity`, as the gate counts them) and at least 

22 ``[jury.ci] min_reviews`` seats; benched seats are added back in config order 

23 until both do. A bench with no economical seat, or no frontier seat, runs 

24 unchanged — there is nothing to save, or nothing to anchor with — and the 

25 plan says which. 

26* **Escalation** after round 1: when the consensus groups carry a ``critical`` 

27 or ``major`` finding, the benched frontier seats join the debate as 

28 cross-examiners and the chair for verification/synthesis is drawn from the 

29 frontier seats. Without escalation the benched seats stay benched. What they 

30 contribute is a debate voice the chair reads, not structured findings — the 

31 debate round has never been parsed for findings, for any seat — so a routed 

32 run can miss a finding a full panel would have surfaced. That is the trade 

33 the flag is, and the report names every seat it did not seat. 

34 

35Pure and deterministic; the orchestrator owns applying it and logging it. 

36""" 

37 

38from __future__ import annotations 

39 

40from dataclasses import dataclass, field 

41 

42from .config import DEFAULT_TIER, AgentSpec, vendor_identity 

43from .diffprofile import RISK_HIGH, RISK_LOW, RISK_MEDIUM 

44 

45MODE_STANDARD = "standard" 

46MODE_TIERED = "tiered" 

47TIER_ECONOMICAL = "economical" 

48 

49#: Severities that escalate a routed run after round 1. 

50ESCALATING_SEVERITIES: tuple[str, ...] = ("critical", "major") 

51 

52 

53@dataclass 

54class RoutingPlan: 

55 """What tiered routing decided, and why — the record the report carries.""" 

56 

57 mode: str 

58 risk: str 

59 panel: list[str] = field(default_factory=list) 

60 benched: list[str] = field(default_factory=list) 

61 anchor: str | None = None 

62 reason: str = "" 

63 escalated: bool = False 

64 escalation_reason: str = "" 

65 

66 def as_dict(self) -> dict: 

67 return { 

68 "mode": self.mode, 

69 "risk": self.risk, 

70 "panel": list(self.panel), 

71 "benched": list(self.benched), 

72 "anchor": self.anchor, 

73 "reason": self.reason, 

74 "escalated": self.escalated, 

75 "escalation_reason": self.escalation_reason, 

76 } 

77 

78 

79def standard_plan(usable: list[str]) -> RoutingPlan: 

80 """The plan a ``standard`` run records: everybody sits, nothing is benched.""" 

81 return RoutingPlan(mode=MODE_STANDARD, risk="", panel=list(usable), reason="standard routing") 

82 

83 

84def _tier_of(spec: AgentSpec) -> str: 

85 return getattr(spec, "tier", DEFAULT_TIER) or DEFAULT_TIER 

86 

87 

88def plan_panel( 

89 specs: list[AgentSpec], 

90 usable: list[str], 

91 risk: str, 

92 *, 

93 chair: str, 

94 min_vendors: int = 0, 

95 min_reviews: int = 0, 

96) -> RoutingPlan: 

97 """Decide the round-1 panel for a ``tiered`` run (pure). 

98 

99 ``specs`` is the enabled bench in config order; ``usable`` the names whose 

100 CLI answered. ``risk`` is a :mod:`ai_jury.diffprofile` band. The result 

101 lists the panel and the bench in config order. 

102 """ 

103 usable_set = set(usable) 

104 ordered = [s for s in specs if s.name in usable_set] 

105 names = [s.name for s in ordered] 

106 economical = [s.name for s in ordered if _tier_of(s) == TIER_ECONOMICAL] 

107 frontier = [s.name for s in ordered if _tier_of(s) != TIER_ECONOMICAL] 

108 

109 def full(reason: str) -> RoutingPlan: 

110 return RoutingPlan(mode=MODE_TIERED, risk=risk, panel=names, reason=reason) 

111 

112 if risk not in (RISK_LOW, RISK_MEDIUM): 

113 band = risk if risk == RISK_HIGH else f"{risk!r} (unknown band, treated as high)" 

114 return full(f"risk={band}: full panel, nothing benched") 

115 if not economical: 

116 return full(f"risk={risk}: no economical seat configured, full panel") 

117 if not frontier: 

118 return full(f"risk={risk}: no frontier seat to anchor with, full panel") 

119 

120 anchor = chair if chair in frontier else frontier[0] 

121 keep = set(economical) | {anchor} 

122 benched = [n for n in frontier if n != anchor] 

123 vendor_of = {s.name: vendor_identity(s.vendor) for s in ordered} 

124 notes: list[str] = [] 

125 

126 def distinct(panel: set[str]) -> int: 

127 return len({vendor_of[n] for n in panel}) 

128 

129 # Floor 1: the cross-vendor gate must still be reachable. Prefer a benched 

130 # seat that brings a vendor the panel lacks; fall back to config order. 

131 while benched and distinct(keep) < min_vendors: 

132 adding = next( 

133 (n for n in benched if vendor_of[n] not in {vendor_of[k] for k in keep}), None 

134 ) 

135 if adding is None: 

136 adding = benched[0] 

137 benched.remove(adding) 

138 keep.add(adding) 

139 notes.append(f"{adding} unbenched for min_vendors={min_vendors}") 

140 # Floor 2: the review count a consumer requires. 

141 while benched and len(keep) < min_reviews: 

142 adding = benched.pop(0) 

143 keep.add(adding) 

144 notes.append(f"{adding} unbenched for min_reviews={min_reviews}") 

145 

146 panel = [n for n in names if n in keep] 

147 reason = ( 

148 f"risk={risk}: {len(economical)} economical seat(s) + anchor {anchor}; " 

149 f"benched {', '.join(benched) if benched else 'nobody'}" 

150 ) 

151 if notes: 

152 reason += "; " + "; ".join(notes) 

153 return RoutingPlan( 

154 mode=MODE_TIERED, risk=risk, panel=panel, benched=benched, anchor=anchor, reason=reason 

155 ) 

156 

157 

158def escalation_effect(joined: list[str], debate_ran: bool = True) -> str: 

159 """What escalation actually did, read off the run rather than predicted. 

160 

161 Called **after** the debate section, with the benched seats that produced a 

162 debate result. Predicting this is what two review rounds caught: a 

163 single-round run has no debate to join (``--auto`` sets exactly one round on 

164 the ``low`` band that benched the seats), and an adaptive run can converge 

165 after round 1 and skip the debate it was going to have (``--auto`` sets 

166 ``early_stop`` on ``medium``). Escalation still moves the chair in both 

167 cases — verification and synthesis are a frontier seat reading the diff and 

168 the findings — but the bench did not review, and a record that says it did 

169 is a record that lies about the run (#714, review rounds 1 and 2). 

170 """ 

171 if joined: 

172 return f"{', '.join(joined)} joined the debate" 

173 if debate_ran: 

174 # The debate happened; the bench is simply not in it, because every 

175 # benched call failed. Saying "no debate round ran" would be false 

176 # about the seated voices that did debate (review round 4). 

177 return "the debate ran without the bench, so only the chair was escalated" 

178 return "no debate round ran, so only the chair was escalated" 

179 

180 

181def should_escalate(groups) -> tuple[bool, str]: 

182 """Whether round 1 warrants the benched frontier seats (pure). 

183 

184 ``groups`` are the consensus groups after round 1. Any ``critical`` or 

185 ``major`` group escalates: those are the findings a cheaper panel is most 

186 likely to have got wrong in either direction, and the ones the frontier 

187 seats were kept back for. 

188 """ 

189 hot = [g for g in groups if getattr(g, "severity", "") in ESCALATING_SEVERITIES] 

190 if not hot: 

191 return False, "no critical or major finding after round 1" 

192 worst = min(hot, key=lambda g: ESCALATING_SEVERITIES.index(g.severity)) 

193 return True, f"{len(hot)} {worst.severity}-or-worse finding group(s) after round 1" 

194 

195 

196def frontier_names(specs: list[AgentSpec], usable: list[str]) -> list[str]: 

197 """Usable frontier seats in config order — the chair pool once escalated.""" 

198 usable_set = set(usable) 

199 return [s.name for s in specs if s.name in usable_set and _tier_of(s) != TIER_ECONOMICAL] 

200 

201 

202def describe(plan: RoutingPlan) -> str: 

203 """One log line naming the decision.""" 

204 if plan.mode != MODE_TIERED: 

205 return "routing: standard (full panel)" 

206 return f"tiered routing: {plan.reason} → panel {', '.join(plan.panel)}"