{
 "files": [
  {
   "date": "2026-09-29",
   "file": "v3-opus55-short.json",
   "model": "claude-opus-5-5",
   "prereg": "/benchmarks/prereg/v3-opus55-short-prereg",
   "url": "/benchmarks/data/v3-opus55-short.json",
   "workloads": [
    {"id": "tests", "label": "npm test fix loop, short", "pct": -21.8, "verdict": "less"},
    {"id": "json", "label": "One large JSON file", "pct": -21.4, "verdict": "less"},
    {"id": "log", "label": "One large log", "pct": -22.4, "verdict": "less"},
    {"id": "git", "label": "Repo history question (git log, 320 commits)", "pct": -25.2, "verdict": "less"},
    {"id": "grep", "label": "Code search (grep over 150 files)", "pct": -28.7, "verdict": "less"},
    {"id": "edit", "label": "One-line edit", "pct": -27.8, "verdict": "less"}
   ]
  },
  {
   "date": "2026-09-29",
   "file": "v4-opus55-W1-W2-W3-W4.json",
   "model": "claude-opus-5-5",
   "prereg": "/benchmarks/prereg/v4-opus55-prereg",
   "url": "/benchmarks/data/v4-opus55-W1-W2-W3-W4.json",
   "workloads": [
    {"id": "W2", "label": "Three parallel subagents, then one more", "pct": -15.2, "verdict": "less"},
    {"id": "W3", "label": "Large documents processed with a script", "pct": -9.9, "verdict": "less"}
   ]
  },
  {
   "date": "2026-09-29",
   "file": "final-opus55-tests-W3-minitest-W4.json",
   "model": "claude-opus-5-5",
   "prereg": "/benchmarks/prereg/final-opus55-tests-W3-minitest-W4-prereg",
   "url": "/benchmarks/data/final-opus55-tests-W3-minitest-W4.json",
   "workloads": [
    {"id": "tests", "label": "npm test fix loop, long", "pct": -21.3, "verdict": "less"},
    {"id": "W3", "label": "Large documents processed with a script", "pct": -9.3, "verdict": "less"},
    {"id": "minitest", "label": "Test fix loop (minitest)", "pct": -61, "verdict": "less"}
   ]
  },
  {
   "date": "2026-09-29",
   "file": "final-haiku45.json",
   "model": "claude-haiku-4-5-20251001",
   "prereg": "/benchmarks/prereg/final-haiku45-prereg",
   "url": "/benchmarks/data/final-haiku45.json",
   "workloads": [
    {"id": "tests", "label": "npm test fix loop", "pct": -52.8, "verdict": "less"},
    {"id": "minitest", "label": "Test fix loop (minitest)", "pct": -45.5, "verdict": "less"},
    {"id": "W3", "label": "Large documents processed with a script", "pct": -64.4, "verdict": "less"},
    {"id": "git", "label": "Repo history question (git log)", "pct": -19.2, "verdict": "less"},
    {"id": "grep", "label": "Code search (grep)", "pct": -28.9, "verdict": "less"},
    {"id": "json", "label": "One large JSON file", "pct": -16.1, "verdict": "less"},
    {"id": "edit", "label": "One-line edit", "pct": -20, "verdict": "less"}
   ]
  }
 ],
 "kept": ["per pair: the pair index, and per arm the cost in USD at list price, success, the model turns and any named answer checks","per workload: the public label, planned pairs, invalid attempts and the pre-registered analysis result, recomputed from the pairs","per file: model, client version, gateway freeze commit, deployed commits, date"],
 "method": ["Paired A/B: each pair runs the same task twice, back to back, once direct to Anthropic and once through Context Mode. The arm order alternates per pair.","Client: Claude Code 2.1.283 (claude -p) in both arms, fresh config, no plugin, empty MCP map.","Pairs per workload fixed and pre-registered before the first pair. One warm-up pair per workload runs first and is not counted.","Each run's pre-registration (tasks, pairs, cost basis and analysis) was committed to the gateway repository before the run's first pair. The published copies under /benchmarks/prereg/ show the commit that added each file, its date, and the run's start time from its evidence file, so a reader can see the plan came first. Each copy is the file as it stood at that commit, with private fields and passages about unpublished workloads removed and marked. Listed under \"preregistrations\" below.","Deploy guard: if the deployed gateway commit changed during a pair, the pair is invalid and runs again (invalid_attempts).","Cost in US dollars at API list price, both arms: list price applied to the token counts Anthropic returned for each call. The runs used a claude.ai login, which has no per-token invoice. Plain: Claude Code's own total. Context Mode: the gateway's per-call record for the run's sessions, hidden calls included, each call at the list price of the model it ran on, checked per run against Claude Code's own figure.","Every valid measured pair, no outlier dropped. % = mean(Context Mode - plain) / mean(plain), with the 95% t interval. Verdict: \"less\" when the whole interval is below zero, \"more\" when above, else \"same\". Gateway wins: pairs where Context Mode cost less."],
 "preregistrations": [
  {"commit": "6e2b65d565aa68f3eacdcece77fe72721c18573f", "committed": "2026-09-29T11:44:26Z", "evidence": "/benchmarks/data/v3-opus55-short.json", "file": "v3-opus55-short-prereg.md", "page": "/benchmarks/prereg/v3-opus55-short-prereg", "run_started": "2026-09-29T11:44:33Z", "title": "Opus 5.5 short tasks (v3)", "url": "/benchmarks/prereg/v3-opus55-short-prereg.md"},
  {"commit": "74b3f851ff0b39794e5c4146d6f39f361da5d554", "committed": "2026-09-29T17:04:33Z", "evidence": "/benchmarks/data/v4-opus55-W1-W2-W3-W4.json", "file": "v4-opus55-prereg.md", "note": "This run also planned a workload that is not in the published evidence. The owner stopped Part B after W2 and W3 finished, before any pair of it ran; that amendment was added to the file after the runs began.", "page": "/benchmarks/prereg/v4-opus55-prereg", "run_started": "2026-09-29T18:20:50Z", "title": "Opus 5.5 multi-agent and data analysis (v4)", "url": "/benchmarks/prereg/v4-opus55-prereg.md"},
  {"commit": "83429d30dcbfff8c5b7708270875b05c812b3a80", "committed": "2026-09-29T07:18:09Z", "evidence": "/benchmarks/data/final-opus55-tests-W3-minitest-W4.json", "file": "final-opus55-tests-W3-minitest-W4-prereg.md", "note": "This run also measured a workload that is not in the published evidence; its rows in the plan are removed and marked.", "page": "/benchmarks/prereg/final-opus55-tests-W3-minitest-W4-prereg", "run_started": "2026-09-29T07:18:17Z", "title": "Opus 5.5 test loops and data analysis (final)", "url": "/benchmarks/prereg/final-opus55-tests-W3-minitest-W4-prereg.md"},
  {"commit": "af66f1bdf43cf992c9d90786f0d0a7895235dc73", "committed": "2026-09-29T07:16:38Z", "evidence": "/benchmarks/data/final-haiku45.json", "file": "final-haiku45-prereg.md", "page": "/benchmarks/prereg/final-haiku45-prereg", "run_started": "2026-09-29T07:16:50Z", "title": "Haiku 4.5 (final)", "url": "/benchmarks/prereg/final-haiku45-prereg.md"}
 ],
 "removed": ["prompt and answer text (each workload has a fixed public label instead)","session ids","tenant ids","gateway hostname and internal endpoint names","local filesystem paths (work dirs, raw dirs, binaries)","per-request provider records, token counts and tool-call lists","Cloudflare deployment and version ids","warm-up pairs and superseded or invalid attempts (counted in invalid_attempts)","runner and preflight internals","workloads whose verdict is not \"less\": only those rows are published, and the complete record stays in the gateway repository"],
 "source": "Copied from the gateway repository (private), commit cc0c5a97047323ff2e5921827de9f21baa631b23, by site/scripts/publish-evidence.mjs.",
 "title": "Context Mode launch benchmark, 2026-09-29: published evidence"
}
