Coverage for src/ai_jury/metadata.py: 99%
108 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Run metadata and cost-awareness (wall-clock proxy) reporting.
3Builds a machine-readable metadata dict describing a jury run: which
4agents participated, their per-agent status and wall-clock duration, how many
5rounds ran, whether verification was enabled, and timestamps.
7IMPORTANT: This metadata deliberately contains NO diff text, NO prompt text,
8NO agent output, and NO secrets -- only structural/operational signals.
10There are no token counts available from the underlying CLIs, so wall-clock
11seconds are used as an approximate cost *proxy*, not a dollar cost.
12"""
14from __future__ import annotations
16from datetime import UTC, datetime
17from typing import TYPE_CHECKING
19from . import panel
20from .config import DEFAULT_MIN_VENDORS, normalise_vendor, vendor_identity
22if TYPE_CHECKING: # pragma: no cover - typing only
23 from .config import JuryConfig
24 from .orchestrator import JuryOutcome
26# v2 (issue #30/#40) added: stop_reason, skipped, retried, budget_exhausted,
27# execution{...}, and per-agent ``attempts``.
28# v4 (issue #501) added: ``panel`` (configured vs effective size, abstentions) and
29# per-agent ``review_status``. A slot that returns no review is an abstention, not an
30# approval, and until now nothing in the output said so.
31# v5 (issues #699/#700) added, inside ``panel``: ``ballots``, ``reviews_supplied``,
32# ``silent``, ``insubstantial``, ``refused``, ``adapter_failed``, ``chair`` and
33# ``chair_ballot`` — the number of reviews a downstream consumer actually
34# receives (the ballots that named something and voted; the chair's synthesis
35# record is not one of them, and neither is an abstaining ballot), the ways
36# a seat that ran produced no review, and whether the chairing agent's own ballot
37# is one of the counted reviews. The causes and ``reviews_supplied`` sum to
38# ``ballots``. Purely additive: every v4 key keeps its name and meaning.
39# v6 (issue #710) added, inside ``panel``: ``not_in_change`` — the seats whose
40# stated scope named only paths or symbols the change does not contain. Split
41# out of ``insubstantial``, which keeps meaning exactly "answered and named
42# nothing checkable": naming `src/made/up.py` is a different failure from naming
43# nothing, and the two ask for different fixes. Additive, and the buckets still
44# sum to ``ballots`` — the seats moved between buckets, none was added or lost.
45# v7 (issue #714) added ``routing``.
46# v8 (issue #863) added, inside ``panel``: ``zero_config_fallback`` — ``true`` when
47# the run had no config and no usable agent CLI, so the panel was the one local
48# seat the zero-config fallback seated. That run is single-vendor, and the default
49# cross-vendor guard does not fail it; this says so where ``--quiet`` cannot hide
50# it. Always present. Additive: every v7 key keeps its name and meaning.
51SCHEMA_VERSION = 8
54#: What a reviewer slot actually contributed (issue #501). ``clean`` and
55#: ``abstained`` both carry zero findings and used to be reported identically,
56#: which is how a run with two non-reviewing slots still described itself as a
57#: three-agent panel.
58REVIEW_STATUSES = ("findings", "clean", "abstained", "failed")
61def review_status(result) -> str:
62 """Classify one reviewer slot's contribution. Pure, and never judges content.
64 * ``failed`` — the adapter did not return a result at all.
65 * ``findings`` — produced at least one structured finding.
66 * ``clean`` — emitted a findings block that was empty: examined, found nothing.
67 * ``abstained`` — returned successfully with no findings block. Not an approval:
68 nothing reviewable came back, so this slot contributed no evidence either way.
69 """
70 if not getattr(result, "ok", False):
71 return "failed"
72 if getattr(result, "findings", None):
73 return "findings"
74 return "clean" if getattr(result, "structured", False) else "abstained"
77def panel_accounting(reviews, chair: str = "", ballots=None) -> dict:
78 """Configured versus *effective* panel size, and the per-status breakdown.
80 A consumer gating on the panel needs the effective number — keel downgrades a
81 jury to advisory below two participating vendors, and can only do that if the
82 report says the panel was short. ``vendors`` counts distinct vendors that
83 actually contributed a review, which is the number that matters for
84 cross-vendor consensus: three slots from one vendor are not three perspectives.
86 Vendors are counted by :func:`config.vendor_identity`, not by the raw
87 string: a seat whose vendor the tool does not recognise ran on the generic
88 ``cli`` fallback and is counted as ``cli``, so two unidentifiable seats are
89 one vendor here even though the report still names each one honestly
90 (issue #701).
92 ``reviews_supplied`` is a different number again, and the one #699 was about:
93 how many reviews the bundle hands on. It is not ``configured`` (every seat
94 that ran, silent ones included), not ``effective`` (a slot that returned
95 prose but no findings block may still have named nothing), and not
96 ``ballots`` plus the chair record — a consumer splits the ``reviewers`` array
97 on ``role`` and reads the ``chair`` entry as the panel's consensus rather
98 than as one more review.
100 It is :func:`ai_jury.panel.review_count` over the **ballot records**, which
101 is why they are passed in rather than derived here: whether a seat reviewed
102 is a fact about the record it produced — its scope and its verdict — and no
103 predicate over the raw round-1 result can see it. That was the defect at this
104 line (#700, round 2): a prose-only non-review abstained on its ballot and was
105 still counted here, so ``--min-reviews`` could be satisfied by seats that
106 reviewed nothing. With ``ballots`` omitted the count is reported as ``None``
107 rather than guessed, because every guess available over-counts.
109 ``silent``, ``insubstantial``, ``refused`` and ``adapter_failed`` split the
110 seats that produced no review by cause, for the shortfall message: a silent
111 agent is a CLI that broke or a budget that ran out, a seat that answered and
112 named nothing is a reviewer that did not review, a refusal is a model
113 declining the task, and a failed adapter is an invocation to fix. They come
114 from :func:`ai_jury.panel.abstention_buckets` — each ballot classified by the
115 cause it carries — so that they and ``reviews_supplied`` add up to
116 ``ballots`` with nothing left over.
118 ``insubstantial`` was derived by subtraction until #700's fifth round, and
119 the subtraction was the defect: ``ballots - silent - supplied`` swept up
120 every seat that was neither silent nor a counted review, and once a ballot
121 could carry a substantive scope and still abstain that included the seat that
122 named a file and then refused, and the one whose adapter died holding a file
123 name. Both were then rendered as "named nothing checkable" — a cause their
124 own ballot contradicted. It now means exactly ``named_nothing``: the seats
125 that answered and named nothing a reader could check, and only those.
127 ``chair_ballot`` says whether the chairing agent's own ballot is one of the
128 *counted* reviews — a chairing agent that ran and abstained has a ballot in
129 the bundle and has supplied no review.
130 """
131 reviews = list(reviews or [])
132 seats = panel.ballot_seats(reviews)
133 records = list(ballots) if ballots is not None else None
134 counted = [r for r in (records or []) if panel.is_review(r)]
135 supplied = len(counted) if records is not None else None
136 # Without the records there is nothing to classify: the cause is a fact the
137 # ballot carries, and every guess available from the raw results is the
138 # subtraction this replaced. ``silent`` is the exception, and only because
139 # :func:`ai_jury.ballots.abstention_cause` tests silence first and on the
140 # same predicate — the two readings are the same number by construction.
141 buckets = panel.abstention_buckets(records) if records is not None else None
142 silent = sum(1 for r in seats if not panel.responded(r))
144 # bolt: Consolidate multiple metrics into a single-pass O(N) explicit loop
145 effective_count = 0
146 abstained_count = 0
147 failed_count = 0
148 contributing_vendors = set()
150 for r in reviews:
151 st = review_status(r)
152 if st in ("findings", "clean"):
153 effective_count += 1
154 vendor = vendor_identity(getattr(r, "vendor", ""))
155 if vendor: 155 ↛ 150line 155 didn't jump to line 150 because the condition on line 155 was always true
156 contributing_vendors.add(vendor)
157 elif st == "abstained":
158 abstained_count += 1
159 elif st == "failed": 159 ↛ 150line 159 didn't jump to line 150 because the condition on line 159 was always true
160 failed_count += 1
162 return {
163 "configured": len(reviews),
164 "effective": effective_count,
165 "vendors": len(contributing_vendors),
166 "abstained": abstained_count,
167 "failed": failed_count,
168 "short": effective_count < len(reviews),
169 # One ballot record per seat that ran — including a silent one, recorded
170 # as an abstention so the report can name it. Deliberately no longer the
171 # same number as ``reviews_supplied``.
172 "ballots": len(seats),
173 "reviews_supplied": supplied,
174 # One entry per cause, from the one mapping, so a bucket cannot be added
175 # in `panel` and go unpublished here. ``silent`` is the only one that
176 # survives a call with no ballots, because it is the only one the raw
177 # results can answer.
178 **{
179 key: (buckets[cause] if buckets is not None else None)
180 for cause, key in panel.PANEL_METADATA_KEYS.items()
181 },
182 "silent": buckets[panel.SILENT] if buckets is not None else silent,
183 "chair": chair or "",
184 "chair_ballot": bool(chair) and any(r.get("name", "") == chair for r in counted),
185 }
188def distinct_vendors(specs) -> int:
189 """How many distinct vendors a set of agent specs represents (pure).
191 Slots, not vendors, is the mistake this exists to prevent: three
192 ``[[agent]]`` entries all pointing at one vendor are one perspective, so a
193 run configured that way never claimed cross-vendor consensus and must not be
194 failed for not delivering it.
196 Counted by :func:`config.vendor_identity`, so a pair of seats naming
197 vendors this build does not know collapses to the single ``cli`` identity
198 they actually share (issue #701) rather than reading as two.
199 """
200 return len({vendor_identity(getattr(s, "vendor", "")) for s in specs or []} - {""})
203def resolve_min_vendors(cli_value, config) -> tuple[int, bool]:
204 """The effective cross-vendor threshold, and whether a person asked for it.
206 PURE. ``cli_value`` is ``args.min_vendors``: ``None`` when neither
207 ``--min-vendors`` nor ``--no-min-vendors`` was passed, in which case the
208 value comes from ``[jury.ci] min_vendors`` (shipped as 2, #682).
210 The second element says whether the threshold was named on the command line.
211 Only an unnamed (default) threshold is scoped down to runs that actually
212 claimed cross-vendor consensus — someone who types ``--min-vendors 3`` on a
213 two-vendor panel is asking for the failure and gets it.
215 Lives here, beside the gate, so a run and ``jury --doctor`` resolve the same
216 threshold from the same flag (#863).
217 """
218 if cli_value is None:
219 return max(0, int(getattr(config.ci, "min_vendors", DEFAULT_MIN_VENDORS))), False
220 return max(0, int(cli_value)), True
223def claimed_vendors(enabled_agents, local_fallback=None) -> int:
224 """How many distinct vendors a run claimed, which scopes the default guard (pure).
226 The enabled seats, except when the zero-config local fallback seated a model
227 (#863): it does so only when none of the built-in seats can run, so the panel
228 that run can form is that one seat, and the missing seats claimed nothing.
229 The run's gate and ``jury --doctor`` both count through here.
230 """
231 return distinct_vendors([local_fallback] if local_fallback is not None else enabled_agents)
234def vendor_guard_fails(contributed: int, required: int, configured_vendors: int | None) -> bool:
235 """Whether ``contributed`` vendors fail a threshold of ``required`` (pure).
237 ``configured_vendors`` scopes the default: fewer than ``required`` means the
238 run never claimed cross-vendor consensus and is left alone. ``None`` is a
239 threshold named on the command line, enforced as asked. The run passes the
240 vendors that reviewed; ``jury --doctor`` passes the vendors it can reach, a
241 ceiling on that — so the two apply one rule and cannot drift (#863).
242 """
243 if required <= 0:
244 return False
245 if configured_vendors is not None and configured_vendors < required:
246 return False
247 return contributed < required
250def collapse_reason(reviews, required: int, configured_vendors: int | None = None) -> str | None:
251 """Why this run may not stand as cross-vendor consensus, or ``None``.
253 PURE. ``required`` is the number of distinct vendors that must have
254 *contributed* a review (:func:`panel_accounting`'s ``vendors``), not the
255 number configured — an agent that was installed, probed clean and then
256 returned nothing is exactly the failure this guards (#635/#682).
258 ``configured_vendors`` scopes the DEFAULT: when fewer distinct vendors are
259 enabled than ``required``, the run never claimed cross-vendor consensus and
260 is left alone, so turning the guard on by default cannot fail a
261 single-vendor install that was always honest about being one. Pass ``None``
262 for an explicitly requested threshold, which is enforced as asked.
264 The message NAMES the opt-out. Whoever reads it is looking at a red CI step
265 on a gate that ships on by default, quite possibly for the first time, and a
266 failure that does not say how to accept it sends them to the issue tracker
267 for a flag the tool already has.
268 """
269 contributed = panel_accounting(reviews).get("vendors", 0)
270 if not vendor_guard_fails(contributed, required, configured_vendors):
271 return None
272 return (
273 f"panel collapsed: {contributed} vendor(s) contributed a review, "
274 f"{required} required. An abstention is not an approval; "
275 f"cross-vendor consensus was not formed. To accept a collapsed panel, "
276 f"pass --no-min-vendors (or set [jury.ci] min_vendors = 0); to catch a "
277 f"missing CLI at startup instead, run with --strict."
278 )
281def _agent_entry(result) -> dict:
282 """Build a single agent metadata entry.
284 Only operational fields are copied -- never ``output`` or ``error`` text,
285 which could contain raw prompt/diff content or secrets.
286 """
287 return {
288 "name": result.agent,
289 # PROVENANCE: the vendor string as configured, never the collapsed
290 # identity. `panel.vendors` is the gate's count of `vendor_identity`;
291 # this is what the seat said it was (#701).
292 "vendor": result.vendor,
293 "status": "ok" if result.ok else "failed",
294 "duration_s": round(float(result.duration_s), 3),
295 "error_code": result.error_code,
296 # Number of attempts made (issue #30): >1 means a transient failure was
297 # retried before this outcome.
298 "attempts": int(getattr(result, "attempts", 1) or 1),
299 # What this slot contributed, not merely whether the CLI exited 0 (#501).
300 "review_status": review_status(result),
301 }
304def _rounds_executed(outcome: JuryOutcome) -> int:
305 # Prefer the orchestrator's authoritative count (adaptive rounds, issue #40);
306 # fall back to inferring it from the phases that produced output.
307 recorded = getattr(outcome, "rounds_executed", None)
308 if isinstance(recorded, int) and recorded >= 1:
309 return recorded
310 rounds = 1 if outcome.reviews else 0
311 if outcome.debate:
312 rounds += 1
313 return rounds
316def estimate_economics(results: list) -> dict:
317 """Estimate token counts and USD dollar cost across all executed agent slots (issue #528).
319 Uses conservative token heuristics (~4 chars/token from output + base context)
320 and published per-vendor pricing tiers. Local models (Ollama, local) are computed
321 at $0.00 (free offline).
322 """
323 vendor_rates_per_1m = {
324 "local": 0.0,
325 "ollama": 0.0,
326 "deepseek": 0.27,
327 "groq": 0.30,
328 "moonshot": 0.50,
329 "gemini": 1.25,
330 "google": 1.25,
331 "openai": 2.50,
332 "codex": 2.50,
333 "anthropic": 3.00,
334 "claude": 3.00,
335 }
336 breakdown = []
337 total_tokens = 0
338 total_cost_usd = 0.0
339 # bolt: Consolidate multiple metrics into a single-pass O(N) explicit loop
340 local_free_slots = 0
342 for r in results:
343 agent = getattr(r, "agent", "unknown")
344 vendor = normalise_vendor(getattr(r, "vendor", ""))
345 output_len = len(getattr(r, "output", "") or "")
346 # Heuristic: base prompt ~800 tokens + output tokens
347 tokens_est = max(100, 800 + (output_len // 4)) if getattr(r, "ok", False) else 200
349 # Match rate
350 rate_per_1m = 2.0 # default generic rate
351 for k, v in vendor_rates_per_1m.items():
352 if k in vendor or k in agent.lower():
353 rate_per_1m = v
354 break
356 cost_usd = (tokens_est / 1_000_000) * rate_per_1m
357 is_local = rate_per_1m == 0.0
359 total_tokens += tokens_est
360 total_cost_usd += cost_usd
361 if is_local:
362 local_free_slots += 1
364 breakdown.append(
365 {
366 "agent": agent,
367 "vendor": getattr(r, "vendor", ""),
368 "tokens_est": tokens_est,
369 "cost_usd_est": round(cost_usd, 6),
370 "is_local_free": is_local,
371 }
372 )
374 return {
375 "total_tokens_est": total_tokens,
376 "total_cost_usd_est": round(total_cost_usd, 4),
377 "local_free_slots": local_free_slots,
378 "breakdown": breakdown,
379 }
382def build_run_metadata(
383 outcome: JuryOutcome,
384 config: JuryConfig,
385 *,
386 decision=None,
387 vote=None,
388 mode: str = "code",
389 zero_config_fallback: bool = False,
390) -> dict:
391 """Return a machine-readable metadata dict for a jury run.
393 The dict is safe to serialize as JSON and contains no diff text, prompt
394 text, agent output, or secrets.
396 Per-agent entries reflect the review panel (round 1). Total wall-clock is
397 summed across every phase (review, debate, verify, synthesis) so it captures
398 the full run cost proxy even though debate/verify/synthesis are re-runs of
399 panel agents rather than distinct participants.
401 ``mode`` selects the ballot vocabulary, matching ``--issue``. It reaches here
402 because ``panel.reviews_supplied`` is counted over the **ballots** (#700,
403 round 2), and the ballots are what ``--issue`` changes: the metadata and the
404 ``reviewers`` array must be derived under the same mode, or the run's own
405 gate would count a different document than the one it printed.
407 ``zero_config_fallback`` says the panel is the zero-config fallback's one
408 local seat (#863); the caller knows, because it seated it.
409 """
410 # The panel is the set of round-1 participants; this is the canonical
411 # per-agent view and avoids duplicating the chair across later phases.
412 agents = [_agent_entry(r) for r in outcome.reviews]
414 all_results = list(outcome.reviews) + list(outcome.debate)
415 if outcome.synthesis is not None:
416 all_results.append(outcome.synthesis)
417 if outcome.verify is not None:
418 all_results.append(outcome.verify)
419 total_wall_clock_s = round(sum(float(r.duration_s) for r in all_results), 3)
421 # Reproducibility signals (issue #41): the run seed and a stable hash of the
422 # effective config let a run be reproduced/explained. The seed is whatever
423 # the run was configured with (may be None when unseeded). The config hash
424 # is a pure function of config, so it is stable across runs and over time.
425 from .ballots import reviewer_ballots
426 from .classification import classify
427 from .config import config_hash
429 # Execution / partial-result signals (issue #30) and adaptive-round signals
430 # (issue #40). ``skipped`` lists agents whose CLI was unavailable so they
431 # never ran; ``budget_exhausted`` flags a run that stopped early on the total
432 # timeout; ``stop_reason`` explains why debate ran or stopped.
433 skipped = [
434 {"name": name, "reason": reason} for name, reason in getattr(outcome, "skipped", []) or []
435 ]
436 retried = [a["name"] for a in agents if a["attempts"] > 1]
438 # Final-verdict mode (issue #220). ``decision`` is the effective mode (CLI
439 # override else config); ``vote`` is the tally dict when voting, else None.
440 decision = decision or config.decision
441 vote_meta = None
442 if vote is not None:
443 vote_meta = {
444 "verdict": vote.verdict,
445 "tally": vote.tally,
446 "ballots": [
447 {"reviewer": b.reviewer, "vote": b.vote, "reason": b.reason} for b in vote.ballots
448 ],
449 }
451 return {
452 "schema_version": SCHEMA_VERSION,
453 "decision": decision,
454 "vote": vote_meta,
455 "agents": agents,
456 # Configured vs effective panel size (issue #501): a slot that returned no
457 # review is an abstention, not an approval, and must not inflate the panel.
458 "panel": {
459 **panel_accounting(
460 outcome.reviews,
461 chair=getattr(outcome, "chair", "") or "",
462 ballots=reviewer_ballots(outcome, config, vote=vote, mode=mode),
463 ),
464 "zero_config_fallback": bool(zero_config_fallback),
465 },
466 "economics": estimate_economics(all_results),
467 "rounds_executed": _rounds_executed(outcome),
468 "from_cache": bool(getattr(outcome, "from_cache", False)),
469 "stop_reason": getattr(outcome, "stop_reason", "") or "",
470 # What routing decided (#714): mode, risk band, who sat, who was benched,
471 # the anchor, and whether round 1 escalated. The standard mode records
472 # {mode: "standard", panel: [...]} so the key is always present.
473 "routing": dict(getattr(outcome, "routing", None) or {"mode": "standard"}),
474 "skipped": skipped,
475 "retried": retried,
476 "budget_exhausted": bool(getattr(outcome, "budget_exhausted", False)),
477 "execution": {
478 "total_timeout": config.total_timeout,
479 "phase_timeout": config.phase_timeout,
480 "retries": config.retries,
481 "early_stop": config.early_stop,
482 "max_rounds": config.effective_max_rounds,
483 },
484 "verify_enabled": bool(config.verify),
485 "context_mode": outcome.context_mode,
486 "redact_secrets": bool(outcome.redact_secrets),
487 "redaction_count": outcome.redaction_count,
488 "seed": config.seed,
489 "config_hash": config_hash(config),
490 # PR-level classification (issue #7): deterministic summary derived from
491 # the structured findings + consensus groups. No diff text is included.
492 "classification": classify(outcome),
493 # Wall-clock is an approximate COST PROXY, not a dollar cost. No token
494 # counts are available from the underlying CLIs.
495 "total_wall_clock_s": total_wall_clock_s,
496 "cost_signal": "wall-clock-proxy",
497 "generated_at": datetime.now(UTC).isoformat(),
498 }
501# end