Coverage for src/ai_jury/metadata.py: 99%

108 statements  

« prev     ^ index     » next       coverage.py v7.16.1, created at 2026-09-30 06:29 +0000

1"""Run metadata and cost-awareness (wall-clock proxy) reporting. 

2 

3Builds a machine-readable metadata dict describing a jury run: which 

4agents participated, their per-agent status and wall-clock duration, how many 

5rounds ran, whether verification was enabled, and timestamps. 

6 

7IMPORTANT: This metadata deliberately contains NO diff text, NO prompt text, 

8NO agent output, and NO secrets -- only structural/operational signals. 

9 

10There are no token counts available from the underlying CLIs, so wall-clock 

11seconds are used as an approximate cost *proxy*, not a dollar cost. 

12""" 

13 

14from __future__ import annotations 

15 

16from datetime import UTC, datetime 

17from typing import TYPE_CHECKING 

18 

19from . import panel 

20from .config import DEFAULT_MIN_VENDORS, normalise_vendor, vendor_identity 

21 

22if TYPE_CHECKING: # pragma: no cover - typing only 

23 from .config import JuryConfig 

24 from .orchestrator import JuryOutcome 

25 

26# v2 (issue #30/#40) added: stop_reason, skipped, retried, budget_exhausted, 

27# execution{...}, and per-agent ``attempts``. 

28# v4 (issue #501) added: ``panel`` (configured vs effective size, abstentions) and 

29# per-agent ``review_status``. A slot that returns no review is an abstention, not an 

30# approval, and until now nothing in the output said so. 

31# v5 (issues #699/#700) added, inside ``panel``: ``ballots``, ``reviews_supplied``, 

32# ``silent``, ``insubstantial``, ``refused``, ``adapter_failed``, ``chair`` and 

33# ``chair_ballot`` — the number of reviews a downstream consumer actually 

34# receives (the ballots that named something and voted; the chair's synthesis 

35# record is not one of them, and neither is an abstaining ballot), the ways 

36# a seat that ran produced no review, and whether the chairing agent's own ballot 

37# is one of the counted reviews. The causes and ``reviews_supplied`` sum to 

38# ``ballots``. Purely additive: every v4 key keeps its name and meaning. 

39# v6 (issue #710) added, inside ``panel``: ``not_in_change`` — the seats whose 

40# stated scope named only paths or symbols the change does not contain. Split 

41# out of ``insubstantial``, which keeps meaning exactly "answered and named 

42# nothing checkable": naming `src/made/up.py` is a different failure from naming 

43# nothing, and the two ask for different fixes. Additive, and the buckets still 

44# sum to ``ballots`` — the seats moved between buckets, none was added or lost. 

45# v7 (issue #714) added ``routing``. 

46# v8 (issue #863) added, inside ``panel``: ``zero_config_fallback`` — ``true`` when 

47# the run had no config and no usable agent CLI, so the panel was the one local 

48# seat the zero-config fallback seated. That run is single-vendor, and the default 

49# cross-vendor guard does not fail it; this says so where ``--quiet`` cannot hide 

50# it. Always present. Additive: every v7 key keeps its name and meaning. 

51SCHEMA_VERSION = 8 

52 

53 

54#: What a reviewer slot actually contributed (issue #501). ``clean`` and 

55#: ``abstained`` both carry zero findings and used to be reported identically, 

56#: which is how a run with two non-reviewing slots still described itself as a 

57#: three-agent panel. 

58REVIEW_STATUSES = ("findings", "clean", "abstained", "failed") 

59 

60 

61def review_status(result) -> str: 

62 """Classify one reviewer slot's contribution. Pure, and never judges content. 

63 

64 * ``failed`` — the adapter did not return a result at all. 

65 * ``findings`` — produced at least one structured finding. 

66 * ``clean`` — emitted a findings block that was empty: examined, found nothing. 

67 * ``abstained`` — returned successfully with no findings block. Not an approval: 

68 nothing reviewable came back, so this slot contributed no evidence either way. 

69 """ 

70 if not getattr(result, "ok", False): 

71 return "failed" 

72 if getattr(result, "findings", None): 

73 return "findings" 

74 return "clean" if getattr(result, "structured", False) else "abstained" 

75 

76 

77def panel_accounting(reviews, chair: str = "", ballots=None) -> dict: 

78 """Configured versus *effective* panel size, and the per-status breakdown. 

79 

80 A consumer gating on the panel needs the effective number — keel downgrades a 

81 jury to advisory below two participating vendors, and can only do that if the 

82 report says the panel was short. ``vendors`` counts distinct vendors that 

83 actually contributed a review, which is the number that matters for 

84 cross-vendor consensus: three slots from one vendor are not three perspectives. 

85 

86 Vendors are counted by :func:`config.vendor_identity`, not by the raw 

87 string: a seat whose vendor the tool does not recognise ran on the generic 

88 ``cli`` fallback and is counted as ``cli``, so two unidentifiable seats are 

89 one vendor here even though the report still names each one honestly 

90 (issue #701). 

91 

92 ``reviews_supplied`` is a different number again, and the one #699 was about: 

93 how many reviews the bundle hands on. It is not ``configured`` (every seat 

94 that ran, silent ones included), not ``effective`` (a slot that returned 

95 prose but no findings block may still have named nothing), and not 

96 ``ballots`` plus the chair record — a consumer splits the ``reviewers`` array 

97 on ``role`` and reads the ``chair`` entry as the panel's consensus rather 

98 than as one more review. 

99 

100 It is :func:`ai_jury.panel.review_count` over the **ballot records**, which 

101 is why they are passed in rather than derived here: whether a seat reviewed 

102 is a fact about the record it produced — its scope and its verdict — and no 

103 predicate over the raw round-1 result can see it. That was the defect at this 

104 line (#700, round 2): a prose-only non-review abstained on its ballot and was 

105 still counted here, so ``--min-reviews`` could be satisfied by seats that 

106 reviewed nothing. With ``ballots`` omitted the count is reported as ``None`` 

107 rather than guessed, because every guess available over-counts. 

108 

109 ``silent``, ``insubstantial``, ``refused`` and ``adapter_failed`` split the 

110 seats that produced no review by cause, for the shortfall message: a silent 

111 agent is a CLI that broke or a budget that ran out, a seat that answered and 

112 named nothing is a reviewer that did not review, a refusal is a model 

113 declining the task, and a failed adapter is an invocation to fix. They come 

114 from :func:`ai_jury.panel.abstention_buckets` — each ballot classified by the 

115 cause it carries — so that they and ``reviews_supplied`` add up to 

116 ``ballots`` with nothing left over. 

117 

118 ``insubstantial`` was derived by subtraction until #700's fifth round, and 

119 the subtraction was the defect: ``ballots - silent - supplied`` swept up 

120 every seat that was neither silent nor a counted review, and once a ballot 

121 could carry a substantive scope and still abstain that included the seat that 

122 named a file and then refused, and the one whose adapter died holding a file 

123 name. Both were then rendered as "named nothing checkable" — a cause their 

124 own ballot contradicted. It now means exactly ``named_nothing``: the seats 

125 that answered and named nothing a reader could check, and only those. 

126 

127 ``chair_ballot`` says whether the chairing agent's own ballot is one of the 

128 *counted* reviews — a chairing agent that ran and abstained has a ballot in 

129 the bundle and has supplied no review. 

130 """ 

131 reviews = list(reviews or []) 

132 seats = panel.ballot_seats(reviews) 

133 records = list(ballots) if ballots is not None else None 

134 counted = [r for r in (records or []) if panel.is_review(r)] 

135 supplied = len(counted) if records is not None else None 

136 # Without the records there is nothing to classify: the cause is a fact the 

137 # ballot carries, and every guess available from the raw results is the 

138 # subtraction this replaced. ``silent`` is the exception, and only because 

139 # :func:`ai_jury.ballots.abstention_cause` tests silence first and on the 

140 # same predicate — the two readings are the same number by construction. 

141 buckets = panel.abstention_buckets(records) if records is not None else None 

142 silent = sum(1 for r in seats if not panel.responded(r)) 

143 

144 # bolt: Consolidate multiple metrics into a single-pass O(N) explicit loop 

145 effective_count = 0 

146 abstained_count = 0 

147 failed_count = 0 

148 contributing_vendors = set() 

149 

150 for r in reviews: 

151 st = review_status(r) 

152 if st in ("findings", "clean"): 

153 effective_count += 1 

154 vendor = vendor_identity(getattr(r, "vendor", "")) 

155 if vendor: 155 ↛ 150line 155 didn't jump to line 150 because the condition on line 155 was always true

156 contributing_vendors.add(vendor) 

157 elif st == "abstained": 

158 abstained_count += 1 

159 elif st == "failed": 159 ↛ 150line 159 didn't jump to line 150 because the condition on line 159 was always true

160 failed_count += 1 

161 

162 return { 

163 "configured": len(reviews), 

164 "effective": effective_count, 

165 "vendors": len(contributing_vendors), 

166 "abstained": abstained_count, 

167 "failed": failed_count, 

168 "short": effective_count < len(reviews), 

169 # One ballot record per seat that ran — including a silent one, recorded 

170 # as an abstention so the report can name it. Deliberately no longer the 

171 # same number as ``reviews_supplied``. 

172 "ballots": len(seats), 

173 "reviews_supplied": supplied, 

174 # One entry per cause, from the one mapping, so a bucket cannot be added 

175 # in `panel` and go unpublished here. ``silent`` is the only one that 

176 # survives a call with no ballots, because it is the only one the raw 

177 # results can answer. 

178 **{ 

179 key: (buckets[cause] if buckets is not None else None) 

180 for cause, key in panel.PANEL_METADATA_KEYS.items() 

181 }, 

182 "silent": buckets[panel.SILENT] if buckets is not None else silent, 

183 "chair": chair or "", 

184 "chair_ballot": bool(chair) and any(r.get("name", "") == chair for r in counted), 

185 } 

186 

187 

188def distinct_vendors(specs) -> int: 

189 """How many distinct vendors a set of agent specs represents (pure). 

190 

191 Slots, not vendors, is the mistake this exists to prevent: three 

192 ``[[agent]]`` entries all pointing at one vendor are one perspective, so a 

193 run configured that way never claimed cross-vendor consensus and must not be 

194 failed for not delivering it. 

195 

196 Counted by :func:`config.vendor_identity`, so a pair of seats naming 

197 vendors this build does not know collapses to the single ``cli`` identity 

198 they actually share (issue #701) rather than reading as two. 

199 """ 

200 return len({vendor_identity(getattr(s, "vendor", "")) for s in specs or []} - {""}) 

201 

202 

203def resolve_min_vendors(cli_value, config) -> tuple[int, bool]: 

204 """The effective cross-vendor threshold, and whether a person asked for it. 

205 

206 PURE. ``cli_value`` is ``args.min_vendors``: ``None`` when neither 

207 ``--min-vendors`` nor ``--no-min-vendors`` was passed, in which case the 

208 value comes from ``[jury.ci] min_vendors`` (shipped as 2, #682). 

209 

210 The second element says whether the threshold was named on the command line. 

211 Only an unnamed (default) threshold is scoped down to runs that actually 

212 claimed cross-vendor consensus — someone who types ``--min-vendors 3`` on a 

213 two-vendor panel is asking for the failure and gets it. 

214 

215 Lives here, beside the gate, so a run and ``jury --doctor`` resolve the same 

216 threshold from the same flag (#863). 

217 """ 

218 if cli_value is None: 

219 return max(0, int(getattr(config.ci, "min_vendors", DEFAULT_MIN_VENDORS))), False 

220 return max(0, int(cli_value)), True 

221 

222 

223def claimed_vendors(enabled_agents, local_fallback=None) -> int: 

224 """How many distinct vendors a run claimed, which scopes the default guard (pure). 

225 

226 The enabled seats, except when the zero-config local fallback seated a model 

227 (#863): it does so only when none of the built-in seats can run, so the panel 

228 that run can form is that one seat, and the missing seats claimed nothing. 

229 The run's gate and ``jury --doctor`` both count through here. 

230 """ 

231 return distinct_vendors([local_fallback] if local_fallback is not None else enabled_agents) 

232 

233 

234def vendor_guard_fails(contributed: int, required: int, configured_vendors: int | None) -> bool: 

235 """Whether ``contributed`` vendors fail a threshold of ``required`` (pure). 

236 

237 ``configured_vendors`` scopes the default: fewer than ``required`` means the 

238 run never claimed cross-vendor consensus and is left alone. ``None`` is a 

239 threshold named on the command line, enforced as asked. The run passes the 

240 vendors that reviewed; ``jury --doctor`` passes the vendors it can reach, a 

241 ceiling on that — so the two apply one rule and cannot drift (#863). 

242 """ 

243 if required <= 0: 

244 return False 

245 if configured_vendors is not None and configured_vendors < required: 

246 return False 

247 return contributed < required 

248 

249 

250def collapse_reason(reviews, required: int, configured_vendors: int | None = None) -> str | None: 

251 """Why this run may not stand as cross-vendor consensus, or ``None``. 

252 

253 PURE. ``required`` is the number of distinct vendors that must have 

254 *contributed* a review (:func:`panel_accounting`'s ``vendors``), not the 

255 number configured — an agent that was installed, probed clean and then 

256 returned nothing is exactly the failure this guards (#635/#682). 

257 

258 ``configured_vendors`` scopes the DEFAULT: when fewer distinct vendors are 

259 enabled than ``required``, the run never claimed cross-vendor consensus and 

260 is left alone, so turning the guard on by default cannot fail a 

261 single-vendor install that was always honest about being one. Pass ``None`` 

262 for an explicitly requested threshold, which is enforced as asked. 

263 

264 The message NAMES the opt-out. Whoever reads it is looking at a red CI step 

265 on a gate that ships on by default, quite possibly for the first time, and a 

266 failure that does not say how to accept it sends them to the issue tracker 

267 for a flag the tool already has. 

268 """ 

269 contributed = panel_accounting(reviews).get("vendors", 0) 

270 if not vendor_guard_fails(contributed, required, configured_vendors): 

271 return None 

272 return ( 

273 f"panel collapsed: {contributed} vendor(s) contributed a review, " 

274 f"{required} required. An abstention is not an approval; " 

275 f"cross-vendor consensus was not formed. To accept a collapsed panel, " 

276 f"pass --no-min-vendors (or set [jury.ci] min_vendors = 0); to catch a " 

277 f"missing CLI at startup instead, run with --strict." 

278 ) 

279 

280 

281def _agent_entry(result) -> dict: 

282 """Build a single agent metadata entry. 

283 

284 Only operational fields are copied -- never ``output`` or ``error`` text, 

285 which could contain raw prompt/diff content or secrets. 

286 """ 

287 return { 

288 "name": result.agent, 

289 # PROVENANCE: the vendor string as configured, never the collapsed 

290 # identity. `panel.vendors` is the gate's count of `vendor_identity`; 

291 # this is what the seat said it was (#701). 

292 "vendor": result.vendor, 

293 "status": "ok" if result.ok else "failed", 

294 "duration_s": round(float(result.duration_s), 3), 

295 "error_code": result.error_code, 

296 # Number of attempts made (issue #30): >1 means a transient failure was 

297 # retried before this outcome. 

298 "attempts": int(getattr(result, "attempts", 1) or 1), 

299 # What this slot contributed, not merely whether the CLI exited 0 (#501). 

300 "review_status": review_status(result), 

301 } 

302 

303 

304def _rounds_executed(outcome: JuryOutcome) -> int: 

305 # Prefer the orchestrator's authoritative count (adaptive rounds, issue #40); 

306 # fall back to inferring it from the phases that produced output. 

307 recorded = getattr(outcome, "rounds_executed", None) 

308 if isinstance(recorded, int) and recorded >= 1: 

309 return recorded 

310 rounds = 1 if outcome.reviews else 0 

311 if outcome.debate: 

312 rounds += 1 

313 return rounds 

314 

315 

316def estimate_economics(results: list) -> dict: 

317 """Estimate token counts and USD dollar cost across all executed agent slots (issue #528). 

318 

319 Uses conservative token heuristics (~4 chars/token from output + base context) 

320 and published per-vendor pricing tiers. Local models (Ollama, local) are computed 

321 at $0.00 (free offline). 

322 """ 

323 vendor_rates_per_1m = { 

324 "local": 0.0, 

325 "ollama": 0.0, 

326 "deepseek": 0.27, 

327 "groq": 0.30, 

328 "moonshot": 0.50, 

329 "gemini": 1.25, 

330 "google": 1.25, 

331 "openai": 2.50, 

332 "codex": 2.50, 

333 "anthropic": 3.00, 

334 "claude": 3.00, 

335 } 

336 breakdown = [] 

337 total_tokens = 0 

338 total_cost_usd = 0.0 

339 # bolt: Consolidate multiple metrics into a single-pass O(N) explicit loop 

340 local_free_slots = 0 

341 

342 for r in results: 

343 agent = getattr(r, "agent", "unknown") 

344 vendor = normalise_vendor(getattr(r, "vendor", "")) 

345 output_len = len(getattr(r, "output", "") or "") 

346 # Heuristic: base prompt ~800 tokens + output tokens 

347 tokens_est = max(100, 800 + (output_len // 4)) if getattr(r, "ok", False) else 200 

348 

349 # Match rate 

350 rate_per_1m = 2.0 # default generic rate 

351 for k, v in vendor_rates_per_1m.items(): 

352 if k in vendor or k in agent.lower(): 

353 rate_per_1m = v 

354 break 

355 

356 cost_usd = (tokens_est / 1_000_000) * rate_per_1m 

357 is_local = rate_per_1m == 0.0 

358 

359 total_tokens += tokens_est 

360 total_cost_usd += cost_usd 

361 if is_local: 

362 local_free_slots += 1 

363 

364 breakdown.append( 

365 { 

366 "agent": agent, 

367 "vendor": getattr(r, "vendor", ""), 

368 "tokens_est": tokens_est, 

369 "cost_usd_est": round(cost_usd, 6), 

370 "is_local_free": is_local, 

371 } 

372 ) 

373 

374 return { 

375 "total_tokens_est": total_tokens, 

376 "total_cost_usd_est": round(total_cost_usd, 4), 

377 "local_free_slots": local_free_slots, 

378 "breakdown": breakdown, 

379 } 

380 

381 

382def build_run_metadata( 

383 outcome: JuryOutcome, 

384 config: JuryConfig, 

385 *, 

386 decision=None, 

387 vote=None, 

388 mode: str = "code", 

389 zero_config_fallback: bool = False, 

390) -> dict: 

391 """Return a machine-readable metadata dict for a jury run. 

392 

393 The dict is safe to serialize as JSON and contains no diff text, prompt 

394 text, agent output, or secrets. 

395 

396 Per-agent entries reflect the review panel (round 1). Total wall-clock is 

397 summed across every phase (review, debate, verify, synthesis) so it captures 

398 the full run cost proxy even though debate/verify/synthesis are re-runs of 

399 panel agents rather than distinct participants. 

400 

401 ``mode`` selects the ballot vocabulary, matching ``--issue``. It reaches here 

402 because ``panel.reviews_supplied`` is counted over the **ballots** (#700, 

403 round 2), and the ballots are what ``--issue`` changes: the metadata and the 

404 ``reviewers`` array must be derived under the same mode, or the run's own 

405 gate would count a different document than the one it printed. 

406 

407 ``zero_config_fallback`` says the panel is the zero-config fallback's one 

408 local seat (#863); the caller knows, because it seated it. 

409 """ 

410 # The panel is the set of round-1 participants; this is the canonical 

411 # per-agent view and avoids duplicating the chair across later phases. 

412 agents = [_agent_entry(r) for r in outcome.reviews] 

413 

414 all_results = list(outcome.reviews) + list(outcome.debate) 

415 if outcome.synthesis is not None: 

416 all_results.append(outcome.synthesis) 

417 if outcome.verify is not None: 

418 all_results.append(outcome.verify) 

419 total_wall_clock_s = round(sum(float(r.duration_s) for r in all_results), 3) 

420 

421 # Reproducibility signals (issue #41): the run seed and a stable hash of the 

422 # effective config let a run be reproduced/explained. The seed is whatever 

423 # the run was configured with (may be None when unseeded). The config hash 

424 # is a pure function of config, so it is stable across runs and over time. 

425 from .ballots import reviewer_ballots 

426 from .classification import classify 

427 from .config import config_hash 

428 

429 # Execution / partial-result signals (issue #30) and adaptive-round signals 

430 # (issue #40). ``skipped`` lists agents whose CLI was unavailable so they 

431 # never ran; ``budget_exhausted`` flags a run that stopped early on the total 

432 # timeout; ``stop_reason`` explains why debate ran or stopped. 

433 skipped = [ 

434 {"name": name, "reason": reason} for name, reason in getattr(outcome, "skipped", []) or [] 

435 ] 

436 retried = [a["name"] for a in agents if a["attempts"] > 1] 

437 

438 # Final-verdict mode (issue #220). ``decision`` is the effective mode (CLI 

439 # override else config); ``vote`` is the tally dict when voting, else None. 

440 decision = decision or config.decision 

441 vote_meta = None 

442 if vote is not None: 

443 vote_meta = { 

444 "verdict": vote.verdict, 

445 "tally": vote.tally, 

446 "ballots": [ 

447 {"reviewer": b.reviewer, "vote": b.vote, "reason": b.reason} for b in vote.ballots 

448 ], 

449 } 

450 

451 return { 

452 "schema_version": SCHEMA_VERSION, 

453 "decision": decision, 

454 "vote": vote_meta, 

455 "agents": agents, 

456 # Configured vs effective panel size (issue #501): a slot that returned no 

457 # review is an abstention, not an approval, and must not inflate the panel. 

458 "panel": { 

459 **panel_accounting( 

460 outcome.reviews, 

461 chair=getattr(outcome, "chair", "") or "", 

462 ballots=reviewer_ballots(outcome, config, vote=vote, mode=mode), 

463 ), 

464 "zero_config_fallback": bool(zero_config_fallback), 

465 }, 

466 "economics": estimate_economics(all_results), 

467 "rounds_executed": _rounds_executed(outcome), 

468 "from_cache": bool(getattr(outcome, "from_cache", False)), 

469 "stop_reason": getattr(outcome, "stop_reason", "") or "", 

470 # What routing decided (#714): mode, risk band, who sat, who was benched, 

471 # the anchor, and whether round 1 escalated. The standard mode records 

472 # {mode: "standard", panel: [...]} so the key is always present. 

473 "routing": dict(getattr(outcome, "routing", None) or {"mode": "standard"}), 

474 "skipped": skipped, 

475 "retried": retried, 

476 "budget_exhausted": bool(getattr(outcome, "budget_exhausted", False)), 

477 "execution": { 

478 "total_timeout": config.total_timeout, 

479 "phase_timeout": config.phase_timeout, 

480 "retries": config.retries, 

481 "early_stop": config.early_stop, 

482 "max_rounds": config.effective_max_rounds, 

483 }, 

484 "verify_enabled": bool(config.verify), 

485 "context_mode": outcome.context_mode, 

486 "redact_secrets": bool(outcome.redact_secrets), 

487 "redaction_count": outcome.redaction_count, 

488 "seed": config.seed, 

489 "config_hash": config_hash(config), 

490 # PR-level classification (issue #7): deterministic summary derived from 

491 # the structured findings + consensus groups. No diff text is included. 

492 "classification": classify(outcome), 

493 # Wall-clock is an approximate COST PROXY, not a dollar cost. No token 

494 # counts are available from the underlying CLIs. 

495 "total_wall_clock_s": total_wall_clock_s, 

496 "cost_signal": "wall-clock-proxy", 

497 "generated_at": datetime.now(UTC).isoformat(), 

498 } 

499 

500 

501# end