Coverage for src/ai_jury/cache.py: 91%
170 statements
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
« prev ^ index » next coverage.py v7.16.1, created at 2026-09-30 06:29 +0000
1"""Optional local result cache for repeated jury runs (issue #33).
3Re-running the jury against an unchanged diff with an unchanged config
4re-spends time and tokens for an identical result. This module adds an opt-in,
5on-disk cache keyed by everything that can change the outcome: the diff, the
6effective config hash, the prompt-template version, the package version, the
7context policy *and the context text it admits*, and the run seed.
9Privacy note: a cache entry stores the full structured outcome — including agent
10review/debate/synthesis text, which is derived from the diff. Treat the cache
11directory as sensitive (same trust level as the diff itself). The cache is OFF
12by default and only writes when explicitly enabled with ``--cache``; clear it
13with ``--clear-cache`` (or ``jury cache clear``).
14"""
16from __future__ import annotations
18import contextlib
19import hashlib
20import hmac
21import json
22import os
23import re
24import secrets
25import stat
26import tempfile
27from dataclasses import asdict
28from pathlib import Path
30from . import __version__, prompts
31from .adapters import AgentResult
32from .config import JuryConfig, config_hash
33from .consensus import FindingGroup
34from .findings import Finding, Verdict
35from .injection import InjectionHit
36from .largediff import ChangeIndex
37from .orchestrator import JuryOutcome
39#: The record *format* version, bumped when a stored outcome gains a field a
40#: renderer reads back. 2 (issue #709, round 2): ``AgentResult`` gained
41#: ``model``, the id the invocation sent, and the ballot reads it to justify
42#: ``model_source: requested``. An entry written without the field cannot
43#: support that label — the run that wrote it did not record what it sent — and
44#: recomputing an id to fill the gap puts a derived value under a token whose
45#: whole claim is that it came off the wire, which is #709 itself. So such an
46#: entry is **not read**: a cache exists to be an exact stand-in for a fresh
47#: run, and where it cannot be, one re-run is the honest price.
48#:
49#: The cache *key* does not settle this on its own. It happens to invalidate
50#: every pre-#709 entry in this release, because ``prompts.PROMPT_VERSION`` went
51#: 7 → 8 for #710 in the same change — but that is a coincidence of two fixes
52#: shipping together. Had #709 landed alone the key would have been unchanged
53#: and every existing entry would have come back a field short. A change to the
54#: record's format belongs in the field that versions the format.
55CACHE_SCHEMA = 2
56_ENV_DIR = "JURY_CACHE_DIR"
58# Cache files are named `<64-hex sha256>.json` (entries) or
59# `<64-hex>.json.<rand>.tmp` (in-flight atomic writes). `clear()` only touches
60# files matching this shape (issue #316/L-3) so it never deletes unrelated files
61# when JURY_CACHE_DIR points at a shared/populated directory.
62_CACHE_NAME_RE = re.compile(r"^[0-9a-f]{64}\.json")
65def default_cache_dir() -> Path:
66 """Cache directory: ``$JURY_CACHE_DIR`` or ``~/.cache/ai-jury``."""
67 override = os.environ.get(_ENV_DIR)
68 if override:
69 return Path(override)
70 base = os.environ.get("XDG_CACHE_HOME") or str(Path.home() / ".cache")
71 return Path(base) / "ai-jury"
74def _policy_fingerprint(policy) -> str:
75 """Stable fingerprint of a review policy for the cache key (issue #122).
77 Returns "none" for no/empty policy. The policy is maintainer-authored review
78 guidance injected into the prompts, so it changes the outcome and must be
79 part of the key.
80 """
81 if policy is None or (hasattr(policy, "is_empty") and policy.is_empty()):
82 return "none"
83 from dataclasses import asdict, is_dataclass
85 data = asdict(policy) if is_dataclass(policy) else policy
86 return hashlib.sha256(json.dumps(data, sort_keys=True, default=str).encode("utf-8")).hexdigest()
89def cache_key(
90 config: JuryConfig,
91 diff: str,
92 *,
93 context: str = "",
94 hints: str = "",
95 seed: int | None = None,
96 mock: bool = False,
97 policy=None,
98 mode: str = "code",
99) -> str:
100 """Stable cache key for a run.
102 A pure function of the inputs that determine the outcome. The seed is part of
103 the key (it changes randomized orchestration), unlike in ``config_hash``
104 which describes configuration independent of seed. ``mock`` is included so a
105 ``--mock`` run (deterministic canned findings) can NEVER be served as a real
106 review for the same diff+config, and vice versa. ``policy`` (the repository
107 review policy) is fingerprinted in too, since it is injected into the prompts
108 and changes the result (issue #122).
110 ``context`` is the PR context block (issue #738). The *policy* — the mode and
111 ``redact_secrets`` — was already in the payload, but the text it admits was
112 not, so under ``--context-mode expanded`` the PR title and body were rendered
113 into every Round 1 prompt (and read by the injection scanner) while the key
114 stayed put: editing a PR description, or a third party editing it, changed
115 what the panel was shown and ``--cache`` replayed the previous outcome.
117 It is hashed **pre-redaction**, exactly as ``diff`` is: ``run_jury`` redacts
118 both itself, after this key is computed, so the pre-redaction string is what
119 every caller has. Two contexts differing only in a secret therefore key
120 differently even though the panel would see the same redacted text — a
121 conservative split (one extra run, never a stale hit) and not a leak: what is
122 stored is a digest, not the text, it never leaves the machine, and a secret
123 appearing in a PR body is a change to the input this key exists to notice.
124 ``redact_secrets`` is separately in the payload, so the redaction *policy*
125 cannot change under a fixed key either.
127 The digest is present only when the panel will actually be shown context —
128 ``run_jury`` clears ``context`` under ``diff-only``, the default mode, before
129 anything reads it. Omitting the field there (rather than hashing ``""``)
130 keeps the payload byte-identical for every run that sends no context, so no
131 existing entry is invalidated for a string the panel never saw. Same
132 precedent as ``[[agent]] adapter`` in ``config_hash`` (#705).
134 ``hints`` is the static-analysis pre-pass block (issue #745), and it is here
135 for the reason ``context`` is: it is joined into every Round 1 prompt. The
136 ``hints`` *flag* was already in ``config_hash`` (#715), but the flag says
137 only that the linters ran — the block itself is a function of the **working
138 tree**, not of the diff or the config, so fixing a lint error anywhere in the
139 tree changed what the panel was shown while every other input to this key
140 stood still.
142 Unlike ``context`` it takes no mode filter: ``run_jury`` joins the block into
143 the Round 1 prompt *after* the ``diff-only`` filter, precisely so the default
144 context mode cannot discard it (#715), and it is not redacted either. So the
145 string the caller passes is the string the panel is shown, and hashing it
146 when it is non-empty — absent otherwise, by the rule above — leaves a run
147 with ``hints = false``, the default, keyed exactly as it was.
148 """
149 # What ``run_jury`` will hand the panel: nothing under "diff-only".
150 panel_context = "" if config.context.mode == "diff-only" else context
151 payload = {
152 "cache_schema": CACHE_SCHEMA,
153 "package_version": __version__,
154 "prompt_version": prompts.PROMPT_VERSION,
155 "config_hash": config_hash(config),
156 "diff_sha256": hashlib.sha256(diff.encode("utf-8")).hexdigest(),
157 "context_mode": config.context.mode,
158 "redact_secrets": config.context.redact_secrets,
159 "verify": config.verify,
160 "seed": seed if seed is not None else config.seed,
161 "mock": bool(mock),
162 "policy": _policy_fingerprint(policy),
163 # Review mode (issue #221): "code" vs "issue" select different prompt
164 # rubrics, so the same text must never be served across modes.
165 "mode": mode,
166 }
167 if panel_context:
168 payload["context_sha256"] = hashlib.sha256(panel_context.encode("utf-8")).hexdigest()
169 if hints:
170 payload["hints_sha256"] = hashlib.sha256(hints.encode("utf-8")).hexdigest()
171 blob = json.dumps(payload, sort_keys=True, separators=(",", ":"))
172 return hashlib.sha256(blob.encode("utf-8")).hexdigest()
175def _finding(d: dict) -> Finding:
176 return Finding(
177 severity=d.get("severity", "info"),
178 file=d.get("file", ""),
179 claim=d.get("claim", ""),
180 line=d.get("line"),
181 evidence=d.get("evidence", ""),
182 suggested_fix=d.get("suggested_fix", ""),
183 confidence=d.get("confidence", "medium"),
184 reviewer=d.get("reviewer", ""),
185 )
188def _verdict(d: dict) -> Verdict:
189 return Verdict(
190 file=d.get("file"),
191 line=d.get("line"),
192 claim=d.get("claim", ""),
193 status=d.get("status", "needs_human_decision"),
194 reasoning=d.get("reasoning", ""),
195 )
198def _agent_result(d: dict | None) -> AgentResult | None:
199 if d is None:
200 return None
201 return AgentResult(
202 agent=d["agent"],
203 vendor=d["vendor"],
204 ok=d["ok"],
205 output=d["output"],
206 duration_s=d["duration_s"],
207 error=d.get("error"),
208 findings=[_finding(f) for f in d.get("findings", [])],
209 warnings=list(d.get("warnings", [])),
210 error_code=d.get("error_code"),
211 attempts=d.get("attempts", 1),
212 # The id the invocation sent (#709). Restored so a cached ballot quotes
213 # the same string a fresh one does — the cache key already pins the
214 # config, so the id cannot have been anything else. The default is for a
215 # dict that never came from a cache entry (`jury replay` takes any
216 # `outcome_to_dict` dump): a stored entry missing the field is refused
217 # by `CACHE_SCHEMA` before it reaches here, rather than balloting a
218 # recomputed id under a label that claims it was sent.
219 model=d.get("model", ""),
220 )
223def _group(d: dict) -> FindingGroup:
224 return FindingGroup(
225 representative=_finding(d["representative"]),
226 reviewers=list(d.get("reviewers", [])),
227 severity=d.get("severity", "info"),
228 members=[_finding(m) for m in d.get("members", [])],
229 bucket=d.get("bucket", "single_reviewer"),
230 status=d.get("status", ""),
231 status_reasoning=d.get("status_reasoning", ""),
232 )
235def _hit(d: dict) -> InjectionHit:
236 return InjectionHit(
237 kind=d.get("kind", ""),
238 source=d.get("source", ""),
239 line=d.get("line"),
240 snippet=d.get("snippet", ""),
241 )
244def _change_index(d: dict | None) -> ChangeIndex | None:
245 """Rebuild a :class:`ai_jury.largediff.ChangeIndex`, or None for a legacy entry."""
246 if not isinstance(d, dict):
247 return None
248 return ChangeIndex(
249 paths=tuple(d.get("paths") or ()),
250 symbols=tuple(d.get("symbols") or ()),
251 )
254def outcome_to_dict(outcome: JuryOutcome) -> dict:
255 """Serialize a JuryOutcome to a JSON-safe dict (dataclasses all the way down)."""
256 return asdict(outcome)
259def outcome_from_dict(data: dict) -> JuryOutcome:
260 """Rebuild a JuryOutcome from :func:`outcome_to_dict` output."""
261 return JuryOutcome(
262 reviews=[_agent_result(r) for r in data.get("reviews", [])],
263 debate=[_agent_result(r) for r in data.get("debate", [])],
264 synthesis=_agent_result(data.get("synthesis")),
265 chair=data.get("chair", ""),
266 findings=[_finding(f) for f in data.get("findings", [])],
267 warnings=list(data.get("warnings", [])),
268 groups=[_group(g) for g in data.get("groups", [])],
269 verify=_agent_result(data.get("verify")),
270 verdicts=[_verdict(v) for v in data.get("verdicts", [])],
271 context_mode=data.get("context_mode", "diff-only"),
272 redact_secrets=data.get("redact_secrets", True),
273 redaction_count=data.get("redaction_count", 0),
274 injection_hits=[_hit(h) for h in data.get("injection_hits", [])],
275 skipped=[tuple(s) for s in data.get("skipped", [])],
276 budget_exhausted=data.get("budget_exhausted", False),
277 rounds_executed=data.get("rounds_executed", 1),
278 stop_reason=data.get("stop_reason", ""),
279 from_cache=data.get("from_cache", False),
280 # What the change contained (#710). The cache key already pins the diff
281 # by digest, so a restored index describes the same bytes as the run
282 # that wrote it; a legacy entry has none and the scope rule falls back.
283 changed=_change_index(data.get("changed")),
284 # The routing record rides along like every other field (#714). An
285 # entry written before this key existed is restored as the standard
286 # record naming the seats it holds, which is what those runs did: the
287 # panel was never trimmed. That is truthful without invalidating a
288 # single cache entry — `routing` and every seat's `tier` are already in
289 # `config_hash`, so a plan that would route differently has a different
290 # key and cannot hit an old entry at all.
291 routing=dict(data["routing"])
292 if data.get("routing")
293 else {
294 "mode": "standard",
295 "risk": "",
296 "panel": [r.get("agent", "") for r in data.get("reviews", [])],
297 "benched": [],
298 "anchor": None,
299 "reason": "standard routing",
300 "escalated": False,
301 "escalation_reason": "",
302 },
303 )
306_HMAC_KEY_FILE = ".hmac_key"
308# Upper bound on a cache entry read before MAC verification (issue #303/L-5). A
309# jury outcome is small (a few KB); reject a multi-MB file so an attacker-planted
310# giant entry can't be fully parsed into memory before the MAC rejects it.
311_MAX_CACHE_BYTES = 8 * 1024 * 1024
314def _dir_is_untrusted(directory: Path) -> bool:
315 """True when ``directory`` is group/other-writable (issue #295).
317 A world-/group-writable cache dir lets another local user plant entries (and
318 even swap the HMAC key file), so we fail closed rather than trust it. POSIX
319 only — Windows ACLs are not represented in ``st_mode``, so the check is
320 skipped there.
321 """
322 if os.name == "nt": 322 ↛ 323line 322 didn't jump to line 323 because the condition on line 322 was never true
323 return False
324 try:
325 mode = directory.stat().st_mode
326 except OSError:
327 return False
328 return bool(mode & (stat.S_IWGRP | stat.S_IWOTH))
331def _hmac_key(directory: Path) -> bytes | None:
332 """Return the per-user cache MAC secret, creating it 0o600 on first use.
334 The key lives in ``<cache_dir>/.hmac_key`` readable only by the owner, so a
335 forged entry can't carry a valid MAC unless the attacker can read the secret
336 (blocked by 0o600) or replace it (blocked by ``_dir_is_untrusted``). Returns
337 None if the secret can't be read/created, in which case callers skip MACing.
338 """
339 key_path = directory / _HMAC_KEY_FILE
340 with contextlib.suppress(OSError):
341 return key_path.read_bytes()
342 # Not present (or unreadable) — try to create it atomically, owner-only.
343 try:
344 fd = os.open(str(key_path), os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
345 except FileExistsError:
346 # Lost a race with a concurrent writer; read what they wrote.
347 try:
348 return key_path.read_bytes()
349 except OSError:
350 return None
351 except OSError:
352 return None
353 try:
354 key = secrets.token_bytes(32)
355 with os.fdopen(fd, "wb") as handle:
356 handle.write(key)
357 return key
358 except OSError:
359 return None
362def _canonical(entry: dict) -> str:
363 """Deterministic serialization of an entry (minus its ``mac``) for MACing."""
364 return json.dumps(
365 {k: v for k, v in entry.items() if k != "mac"},
366 separators=(",", ":"),
367 sort_keys=True,
368 )
371def _compute_mac(key: bytes, entry: dict) -> str:
372 return hmac.new(key, _canonical(entry).encode("utf-8"), hashlib.sha256).hexdigest()
375class Cache:
376 """A simple on-disk JSON cache of jury outcomes."""
378 def __init__(self, directory: Path | str | None = None):
379 self.dir = Path(directory) if directory else default_cache_dir()
381 def _path(self, key: str) -> Path:
382 return self.dir / f"{key}.json"
384 def load(self, key: str) -> JuryOutcome | None:
385 """Return the cached outcome for ``key`` (marked ``from_cache``), or None.
387 A corrupt or unreadable entry is treated as a miss rather than an error,
388 so a bad cache file never breaks a run.
389 """
390 path = self._path(key)
391 if not path.exists():
392 return None
393 # Fail closed: never trust an entry read from a world/group-writable dir
394 # (issue #295) — an attacker who can write the dir can forge entries.
395 if _dir_is_untrusted(self.dir):
396 return None
397 try:
398 # Size-cap the READ itself (issue #303/L-5, review): read at most
399 # _MAX_CACHE_BYTES+1 chars rather than stat-then-read (which is a
400 # TOCTOU and still reads the whole file). A giant attacker-planted
401 # entry is rejected without being pulled into memory.
402 with path.open("r", encoding="utf-8") as fh:
403 raw = fh.read(_MAX_CACHE_BYTES + 1)
404 if len(raw) > _MAX_CACHE_BYTES:
405 return None
406 data = json.loads(raw)
407 except (OSError, ValueError, RecursionError):
408 # RecursionError on deeply nested JSON is not a ValueError; catch it
409 # so a planted entry can't crash the fail-closed read (audit
410 # 2026-06-13 r3, mirrors findings.py).
411 return None
412 if data.get("cache_schema") != CACHE_SCHEMA:
413 return None
414 # Integrity: the entry must name the key it was written for (issue
415 # #293/F-10). A file dropped at <digest>.json with mismatched content
416 # (e.g. a forged verdict copied from another key) is treated as a miss.
417 if data.get("cache_key") != key:
418 return None
419 # Integrity: verify the per-user HMAC (issue #295). Fail closed — if the
420 # key can't be read/created we cannot authenticate the entry, so treat it
421 # as a miss rather than trusting an unsigned blob. A missing or wrong MAC
422 # (a forgery, or a legacy pre-MAC entry) is likewise a miss.
423 mac_key = _hmac_key(self.dir)
424 if mac_key is None:
425 return None
426 stored_mac = data.get("mac")
427 if not isinstance(stored_mac, str) or not hmac.compare_digest(
428 stored_mac, _compute_mac(mac_key, data)
429 ):
430 return None
431 outcome = outcome_from_dict(data.get("outcome", {}))
432 outcome.from_cache = True
433 return outcome
435 def store(self, key: str, outcome: JuryOutcome) -> None:
436 """Persist ``outcome`` under ``key`` (best-effort; ignores write errors)."""
437 with contextlib.suppress(OSError):
438 # Owner-only cache dir so another local user cannot plant entries
439 # (issue #293/F-10); best-effort tighten if it already exists.
440 self.dir.mkdir(parents=True, exist_ok=True, mode=0o700)
441 with contextlib.suppress(OSError):
442 self.dir.chmod(0o700)
443 # Fail closed (#295): we just tried to tighten the dir to 0700. If it
444 # is STILL group/other-writable, the chmod failed (we don't own it) —
445 # an attacker could swap entries or the MAC key, so refuse to write
446 # rather than trust it (the audit's "don't suppress and continue").
447 if _dir_is_untrusted(self.dir):
448 return
449 payload = {
450 "cache_schema": CACHE_SCHEMA,
451 "cache_key": key,
452 "outcome": outcome_to_dict(outcome),
453 }
454 # Fail closed (#295): if we can't obtain the MAC key, do NOT write an
455 # unsigned entry — an unsigned blob would be accepted as trusted only
456 # if MACing were optional, which it is not. Skip caching instead.
457 mac_key = _hmac_key(self.dir)
458 if mac_key is None:
459 return
460 payload["mac"] = _compute_mac(mac_key, payload)
461 # Atomic write (issue #303/L-4, hardened in #316/L-4): mkstemp gives a
462 # UNIQUE name created with O_EXCL and no symlink-follow — safe against
463 # same-PID/thread concurrency and a pre-planted temp symlink — then an
464 # atomic replace. A crash mid-write can't leave a truncated entry and
465 # a reader never sees a partial file.
466 path = self._path(key)
467 blob = json.dumps(payload, separators=(",", ":")) + "\n"
468 fd, tmp_name = tempfile.mkstemp(dir=self.dir, prefix=f"{path.name}.", suffix=".tmp")
469 tmp = Path(tmp_name)
470 try:
471 with os.fdopen(fd, "w", encoding="utf-8") as handle:
472 handle.write(blob)
473 tmp.replace(path)
474 except OSError:
475 with contextlib.suppress(OSError):
476 tmp.unlink()
477 raise # caught by the outer contextlib.suppress(OSError)
479 def clear(self) -> int:
480 """Remove all cache entries; return the number deleted.
482 Also rotates (deletes) the per-user MAC key (issue #303/L-3) so a clear
483 after a suspected compromise starts fresh; the count covers only the
484 ``*.json`` entries.
485 """
486 if not self.dir.exists():
487 return 0
488 removed = 0
489 # Only touch files matching the cache-name shape (#316/L-3) so a clear on
490 # a shared JURY_CACHE_DIR never deletes unrelated files. Entries
491 # (`<hex>.json`) count; leftover atomic-write temps (`<hex>.json.*.tmp`)
492 # are reaped but not counted.
493 for path in self.dir.glob("*.json"):
494 if _CACHE_NAME_RE.match(path.name):
495 with contextlib.suppress(OSError):
496 path.unlink()
497 removed += 1
498 for tmp in self.dir.glob("*.tmp"):
499 if _CACHE_NAME_RE.match(tmp.name): 499 ↛ 500line 499 didn't jump to line 500 because the condition on line 499 was never true
500 with contextlib.suppress(OSError):
501 tmp.unlink()
502 with contextlib.suppress(OSError):
503 (self.dir / _HMAC_KEY_FILE).unlink()
504 return removed