{
 "copied_as_is": "Every listed JSON file, as committed. Only these changes: tenant ids are replaced by role labels (test tenants by \"withheld (disposable test tenant)\" in tenant fields); prereg names the commit that holds the pre-registration; internal switch names, admin routes, file names and hosts inside text are replaced by the plain words listed in scrub; renamed fields are listed in rename.",
 "cost_basis": "US dollars at API list price, from the token counts Anthropic returned (see usd_source in each pair file).",
 "dates": "2026-10-03 to 2026-10-04",
 "rename": [
  {"to": "ledger_usd_sum", "what": "the gateway's sum of its ledger rows, renamed from its internal field name"}
 ],
 "schema": "context-mode/recall-evidence/v1",
 "scrub": [
  {"from_kind": "internal host", "to": "the deployed gateway"},
  {"from_kind": "internal host", "to": "the gateway"},
  {"from_kind": "internal file or class name", "to": "the deploy smoke test"},
  {"from_kind": "internal file or class name", "to": "the pairs replay script"},
  {"from_kind": "internal file or class name", "to": "the visibility replay script"},
  {"from_kind": "internal file or class name", "to": "the gateway's model price table"},
  {"from_kind": "internal file or class name", "to": "the harness's exact-match checker"},
  {"from_kind": "internal file or class name", "to": "the gateway's request path"},
  {"from_kind": "internal file or class name", "to": "the relocation script"},
  {"from_kind": "internal admin route", "to": "the savings ledger read"},
  {"from_kind": "internal admin route", "to": "the gateway's exec log, mode gateway"},
  {"from_kind": "internal admin route", "to": "the gateway's exec log"},
  {"from_kind": "internal admin route", "to": "a whole-doc archive read"},
  {"from_kind": "internal admin route", "to": "archive search"},
  {"from_kind": "internal admin route", "to": "the gateway's ops rows"},
  {"from_kind": "internal admin route", "to": "the switch table"},
  {"from_kind": "internal switch name", "to": "chain-memo"},
  {"from_kind": "internal switch name", "to": "recall-confirm"},
  {"from_kind": "internal switch name", "to": "recall-quality-rules"},
  {"from_kind": "internal switch name", "to": "thread-object"},
  {"from_kind": "internal switch name", "to": "recall-yield"},
  {"from_kind": "internal switch name", "to": "auto-recall"},
  {"from_kind": "internal switch name", "to": "ledger-v2-write"},
  {"from_kind": "internal switch name", "to": "ledger-v2-read"},
  {"from_kind": "internal switch name", "to": "place-beside-gateway"},
  {"from_kind": "internal switch name", "to": "body-guard"},
  {"from_kind": "internal file or class name", "to": "thread object"},
  {"from_kind": "word (no invoice exists for these runs)", "to": "ledger"}
 ],
 "sets": [
  {"fields_changed": ["checker","ledger_usd_sum (renamed)","prereg","source","tenant","topo","usd_source"], "files": ["/benchmarks/recall/200k/ab.json","/benchmarks/recall/200k/dry4-plain.json","/benchmarks/recall/200k/dry4-recall-b4k.json","/benchmarks/recall/200k/dry4.json","/benchmarks/recall/200k/pair-1-plain.json","/benchmarks/recall/200k/pair-1-recall.json","/benchmarks/recall/200k/pair-1.json","/benchmarks/recall/200k/pair-2-plain.json","/benchmarks/recall/200k/pair-2-recall.json","/benchmarks/recall/200k/pair-2.json","/benchmarks/recall/200k/pair-3-plain.json","/benchmarks/recall/200k/pair-3-recall.json","/benchmarks/recall/200k/pair-3.json","/benchmarks/recall/200k/pair-4-attempt1-plain.json","/benchmarks/recall/200k/pair-4-attempt1-recall.json","/benchmarks/recall/200k/pair-4-attempt1.json","/benchmarks/recall/200k/pair-4-plain.json","/benchmarks/recall/200k/pair-4-recall.json","/benchmarks/recall/200k/pair-4.json","/benchmarks/recall/200k/spend.json"], "id": "200k", "prereg": "pre-registered in gateway commit 1c07881f (2026-10-04T01:34:16+03:00), before the first pair", "result_file": "/benchmarks/recall/200k/ab.json", "source_commit": "2fbde5ab", "tenant_labels_used": ["test-account"], "title": "Long agentic coding session on a real 200K window, Opus 5.5 (2026-10-04)", "what": "Plain Claude Code against Claude Code through Context Mode Gateway, the client's window held at 200K. 30 steps: 20 work steps, then 10 memory questions. Pre-registered. The session stayed under Recall's window line, so Recall did not re-lay it out; run 2 lowers the line."},
  {"fields_changed": ["checker","ledger_usd_sum (renamed)","prereg","source","tenant","topo","usd_source"], "files": ["/benchmarks/recall/200k-r2/ab.json","/benchmarks/recall/200k-r2/pair-1-plain.json","/benchmarks/recall/200k-r2/pair-1-recall.json","/benchmarks/recall/200k-r2/pair-1.json","/benchmarks/recall/200k-r2/pair-2-plain.json","/benchmarks/recall/200k-r2/pair-2-recall.json","/benchmarks/recall/200k-r2/pair-2.json","/benchmarks/recall/200k-r2/pair-3-plain.json","/benchmarks/recall/200k-r2/pair-3-recall.json","/benchmarks/recall/200k-r2/pair-3.json","/benchmarks/recall/200k-r2/spend.json"], "id": "200k-r2", "prereg": "pre-registered in gateway commit 34acbe8b (2026-10-04T02:54:04+03:00), before the first pair", "result_file": "/benchmarks/recall/200k-r2/ab.json", "source_commit": "2fbde5ab", "tenant_labels_used": ["test-account"], "title": "Long agentic coding session on a real 200K window, run 2: Recall acting (2026-10-04)", "what": "Run 1's workload and seed, with the gateway account's Context limit set to 100,000 so Recall's window line is crossed. Pre-registered; 3 of 3 valid pairs; Recall re-laid out in every pair. 37.6% lower cost, 9 of 10 exact answers against 4 of 10."},
  {"fields_changed": ["generated_by","note"], "files": ["/benchmarks/recall/replay/visibility.json","/benchmarks/recall/replay/pairs.json"], "id": "replay", "result_file": "/benchmarks/recall/replay/pairs.json", "source_commit": "2fbde5ab", "tenant_labels_used": [], "title": "Recall quality: offline replays of the long-session A/B (2026-10-03)", "what": "The A/B conversation rebuilt and run through the gateway's real Recall code, request by request. No model call, no network, $0. 'Visible' means the exact answer is in what the gateway forwards; it is not a model's answer."},
  {"fields_changed": ["gw","tenant"], "files": ["/benchmarks/recall/live/claude-code-check.json","/benchmarks/recall/live/codex-check.json"], "id": "live", "result_file": "/benchmarks/recall/live/claude-code-check.json", "source_commit": "2fbde5ab", "tenant_labels_used": ["owner-account"], "title": "Recall quality rules live: real client checks (2026-10-03)", "what": "A real Claude Code turn (claude -p, Haiku 4.5) and a real Codex run through the deployed gateway on the owner's account, after auto-recall and the quality rules went on."},
  {"fields_changed": ["chain-memo","commit","isolate_colo","keys","place-beside-gateway","recall-confirm","recall-yield","source","tenant","thread-object","what"], "files": ["/benchmarks/recall/speed/placement.json","/benchmarks/recall/speed/load-rerun.json"], "id": "speed", "result_file": "/benchmarks/recall/speed/placement.json", "source_commit": "2fbde5ab", "tenant_labels_used": ["owner-account","owner-account-2","customer-account","load-test-account"], "title": "Recall's thread read with objects beside the gateway (2026-10-03)", "what": "Read live from the deployed gateway's ops rows and Cloudflare's analytics: where new thread objects run, and how fast Recall reads them, before and after; then a re-run under load with three more switches on."}
 ]
}
