Coverage for src/ai_jury/routing.py: 100%
81 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Risk-aware tiered routing: which seats review, which one anchors, who escalates.
3Issue #714 (the ask was #524). ``routing = "tiered"`` / ``--tiered`` is a cost
4lever, and a cost lever that could silently weaken a review would be worse than
5none, so every decision here is a pure function of things the operator wrote
6and the diff itself, and every decision is reported:
8* The **risk band** is :func:`ai_jury.diffprofile.profile_diff` — the same
9 classifier ``--auto`` uses — so a routed run and an auto-depth run agree on
10 what "routine" means.
11* The **tier** of a seat is ``[[agent]] tier`` (default ``frontier``). There are
12 no model-name heuristics: a seat is economical because the operator said so.
13* A ``high``-risk diff (security-sensitive paths, large, many files) gets the
14 **full** enabled panel; nothing is saved on a change that can hurt.
15* A ``low`` or ``medium`` diff gets every **economical** seat plus **one
16 frontier anchor** — the configured chair when it is frontier and usable, else
17 the first usable frontier seat in config order. The other frontier seats are
18 *benched*: they run only if round 1 escalates.
19* Two floors always hold, in this order: the panel keeps at least
20 ``[jury.ci] min_vendors`` distinct vendors (counted by
21 :func:`ai_jury.config.vendor_identity`, as the gate counts them) and at least
22 ``[jury.ci] min_reviews`` seats; benched seats are added back in config order
23 until both do. A bench with no economical seat, or no frontier seat, runs
24 unchanged — there is nothing to save, or nothing to anchor with — and the
25 plan says which.
26* **Escalation** after round 1: when the consensus groups carry a ``critical``
27 or ``major`` finding, the benched frontier seats join the debate as
28 cross-examiners and the chair for verification/synthesis is drawn from the
29 frontier seats. Without escalation the benched seats stay benched. What they
30 contribute is a debate voice the chair reads, not structured findings — the
31 debate round has never been parsed for findings, for any seat — so a routed
32 run can miss a finding a full panel would have surfaced. That is the trade
33 the flag is, and the report names every seat it did not seat.
35Pure and deterministic; the orchestrator owns applying it and logging it.
36"""
38from __future__ import annotations
40from dataclasses import dataclass, field
42from .config import DEFAULT_TIER, AgentSpec, vendor_identity
43from .diffprofile import RISK_HIGH, RISK_LOW, RISK_MEDIUM
45MODE_STANDARD = "standard"
46MODE_TIERED = "tiered"
47TIER_ECONOMICAL = "economical"
49#: Severities that escalate a routed run after round 1.
50ESCALATING_SEVERITIES: tuple[str, ...] = ("critical", "major")
53@dataclass
54class RoutingPlan:
55 """What tiered routing decided, and why — the record the report carries."""
57 mode: str
58 risk: str
59 panel: list[str] = field(default_factory=list)
60 benched: list[str] = field(default_factory=list)
61 anchor: str | None = None
62 reason: str = ""
63 escalated: bool = False
64 escalation_reason: str = ""
66 def as_dict(self) -> dict:
67 return {
68 "mode": self.mode,
69 "risk": self.risk,
70 "panel": list(self.panel),
71 "benched": list(self.benched),
72 "anchor": self.anchor,
73 "reason": self.reason,
74 "escalated": self.escalated,
75 "escalation_reason": self.escalation_reason,
76 }
79def standard_plan(usable: list[str]) -> RoutingPlan:
80 """The plan a ``standard`` run records: everybody sits, nothing is benched."""
81 return RoutingPlan(mode=MODE_STANDARD, risk="", panel=list(usable), reason="standard routing")
84def _tier_of(spec: AgentSpec) -> str:
85 return getattr(spec, "tier", DEFAULT_TIER) or DEFAULT_TIER
88def plan_panel(
89 specs: list[AgentSpec],
90 usable: list[str],
91 risk: str,
92 *,
93 chair: str,
94 min_vendors: int = 0,
95 min_reviews: int = 0,
96) -> RoutingPlan:
97 """Decide the round-1 panel for a ``tiered`` run (pure).
99 ``specs`` is the enabled bench in config order; ``usable`` the names whose
100 CLI answered. ``risk`` is a :mod:`ai_jury.diffprofile` band. The result
101 lists the panel and the bench in config order.
102 """
103 usable_set = set(usable)
104 ordered = [s for s in specs if s.name in usable_set]
105 names = [s.name for s in ordered]
106 economical = [s.name for s in ordered if _tier_of(s) == TIER_ECONOMICAL]
107 frontier = [s.name for s in ordered if _tier_of(s) != TIER_ECONOMICAL]
109 def full(reason: str) -> RoutingPlan:
110 return RoutingPlan(mode=MODE_TIERED, risk=risk, panel=names, reason=reason)
112 if risk not in (RISK_LOW, RISK_MEDIUM):
113 band = risk if risk == RISK_HIGH else f"{risk!r} (unknown band, treated as high)"
114 return full(f"risk={band}: full panel, nothing benched")
115 if not economical:
116 return full(f"risk={risk}: no economical seat configured, full panel")
117 if not frontier:
118 return full(f"risk={risk}: no frontier seat to anchor with, full panel")
120 anchor = chair if chair in frontier else frontier[0]
121 keep = set(economical) | {anchor}
122 benched = [n for n in frontier if n != anchor]
123 vendor_of = {s.name: vendor_identity(s.vendor) for s in ordered}
124 notes: list[str] = []
126 def distinct(panel: set[str]) -> int:
127 return len({vendor_of[n] for n in panel})
129 # Floor 1: the cross-vendor gate must still be reachable. Prefer a benched
130 # seat that brings a vendor the panel lacks; fall back to config order.
131 while benched and distinct(keep) < min_vendors:
132 adding = next(
133 (n for n in benched if vendor_of[n] not in {vendor_of[k] for k in keep}), None
134 )
135 if adding is None:
136 adding = benched[0]
137 benched.remove(adding)
138 keep.add(adding)
139 notes.append(f"{adding} unbenched for min_vendors={min_vendors}")
140 # Floor 2: the review count a consumer requires.
141 while benched and len(keep) < min_reviews:
142 adding = benched.pop(0)
143 keep.add(adding)
144 notes.append(f"{adding} unbenched for min_reviews={min_reviews}")
146 panel = [n for n in names if n in keep]
147 reason = (
148 f"risk={risk}: {len(economical)} economical seat(s) + anchor {anchor}; "
149 f"benched {', '.join(benched) if benched else 'nobody'}"
150 )
151 if notes:
152 reason += "; " + "; ".join(notes)
153 return RoutingPlan(
154 mode=MODE_TIERED, risk=risk, panel=panel, benched=benched, anchor=anchor, reason=reason
155 )
158def escalation_effect(joined: list[str], debate_ran: bool = True) -> str:
159 """What escalation actually did, read off the run rather than predicted.
161 Called **after** the debate section, with the benched seats that produced a
162 debate result. Predicting this is what two review rounds caught: a
163 single-round run has no debate to join (``--auto`` sets exactly one round on
164 the ``low`` band that benched the seats), and an adaptive run can converge
165 after round 1 and skip the debate it was going to have (``--auto`` sets
166 ``early_stop`` on ``medium``). Escalation still moves the chair in both
167 cases — verification and synthesis are a frontier seat reading the diff and
168 the findings — but the bench did not review, and a record that says it did
169 is a record that lies about the run (#714, review rounds 1 and 2).
170 """
171 if joined:
172 return f"{', '.join(joined)} joined the debate"
173 if debate_ran:
174 # The debate happened; the bench is simply not in it, because every
175 # benched call failed. Saying "no debate round ran" would be false
176 # about the seated voices that did debate (review round 4).
177 return "the debate ran without the bench, so only the chair was escalated"
178 return "no debate round ran, so only the chair was escalated"
181def should_escalate(groups) -> tuple[bool, str]:
182 """Whether round 1 warrants the benched frontier seats (pure).
184 ``groups`` are the consensus groups after round 1. Any ``critical`` or
185 ``major`` group escalates: those are the findings a cheaper panel is most
186 likely to have got wrong in either direction, and the ones the frontier
187 seats were kept back for.
188 """
189 hot = [g for g in groups if getattr(g, "severity", "") in ESCALATING_SEVERITIES]
190 if not hot:
191 return False, "no critical or major finding after round 1"
192 worst = min(hot, key=lambda g: ESCALATING_SEVERITIES.index(g.severity))
193 return True, f"{len(hot)} {worst.severity}-or-worse finding group(s) after round 1"
196def frontier_names(specs: list[AgentSpec], usable: list[str]) -> list[str]:
197 """Usable frontier seats in config order — the chair pool once escalated."""
198 usable_set = set(usable)
199 return [s.name for s in specs if s.name in usable_set and _tier_of(s) != TIER_ECONOMICAL]
202def describe(plan: RoutingPlan) -> str:
203 """One log line naming the decision."""
204 if plan.mode != MODE_TIERED:
205 return "routing: standard (full panel)"
206 return f"tiered routing: {plan.reason} → panel {', '.join(plan.panel)}"