{
  "_note": "Generated by scripts/build_scoreboard.py from committed artifacts; do not edit.",
  "sections": [
    {
      "benchmark": "SWE-bench Verified, hard tasks, effort max (official grader)",
      "cost_confound": {},
      "cost_note": null,
      "date": "2026-10-06",
      "effort": "max",
      "id": "swebench-verified-hard-max",
      "model": "claude-sonnet-5-5",
      "n_common": 7,
      "note": "long-horizon pilot, not a verdict: 7 tasks (the hard '1-4 hours'/'>4 hours' Verified buckets, as many as the $100 cap bought) at effort max, median 63-69 steps per task. No cost difference is shown for distil or rtk (both $-per-solved intervals cross 1). distil's lower $ per solved comes from one attempt that crashed after 4 steps on max_tokens; on the other 6 tasks distil cost 9.7% more than plain. selective and provider-cm were dropped to fit the cap (pre-registered rule)",
      "rows": [
        {
          "arm": "plain",
          "cost_basis": "cold",
          "cost_usd": 21.6298,
          "label": "plain (no compression)",
          "library": null,
          "n": 7,
          "precached_rows": [
            0,
            8
          ],
          "rate": 0.857143,
          "rate_ci": [
            0.486872,
            0.97432
          ],
          "resolved": 6,
          "status": "measured",
          "steps": 463,
          "tokens_in": 52588959,
          "tokens_out": 825483,
          "usd_per_solved": 3.605,
          "verdict": null,
          "version": null,
          "vs_plain_ci_pts": null,
          "vs_plain_pts": null
        },
        {
          "arm": "distil",
          "cost_basis": "cold",
          "cost_usd": 18.1602,
          "label": "distil",
          "library": "distil",
          "n": 7,
          "precached_rows": [
            0,
            7
          ],
          "rate": 0.857143,
          "rate_ci": [
            0.486872,
            0.97432
          ],
          "resolved": 6,
          "status": "measured",
          "steps": 475,
          "tokens_in": 46957534,
          "tokens_out": 658560,
          "usd_per_solved": 3.0267,
          "verdict": "PILOT (no verdict)",
          "version": "1.57.0",
          "vs_plain_ci_pts": [
            0.0,
            0.0
          ],
          "vs_plain_pts": 0.0
        },
        {
          "arm": "rtk",
          "cost_basis": "cold",
          "cost_usd": 21.5408,
          "label": "RTK",
          "library": "rtk",
          "n": 7,
          "precached_rows": [
            0,
            9
          ],
          "rate": 1.0,
          "rate_ci": [
            0.64567,
            1.0
          ],
          "resolved": 7,
          "status": "measured",
          "steps": 512,
          "tokens_in": 55288254,
          "tokens_out": 773387,
          "usd_per_solved": 3.0773,
          "verdict": "PILOT (no verdict)",
          "version": "0.51.0",
          "vs_plain_ci_pts": [
            -11.64,
            40.21
          ],
          "vs_plain_pts": 14.29
        },
        {
          "arm": "headroom",
          "label": "Headroom",
          "status": "pending run"
        },
        {
          "arm": "selective",
          "label": "Selective Context",
          "status": "pending run"
        },
        {
          "arm": "provider-cm",
          "label": "Anthropic context editing",
          "status": "pending run"
        }
      ],
      "source": "benchmarks/results/swebench-verified-hard-max"
    },
    {
      "benchmark": "SWE-bench Lite (official grader)",
      "cost_confound": {
        "rtk": {
          "per_task_vs_plain_ci_pct_cold": [
            -13.1,
            0.1
          ],
          "per_task_vs_plain_pct_billed": -20.7,
          "per_task_vs_plain_pct_cold": -6.5
        }
      },
      "cost_note": "rtk, selective and provider-cm ran the same day with byte-identical first requests and read each other's prompt cache; plain and distil ran alone and did not. Re-pricing as a cold run keeps each task's exact prompt and output tokens and moves only the cache write/read split (benchmarks/why_rtk_wins.py, docs/research/why-rtk-wins.md). The analysis did not re-price selective, and re-priced provider-cm only on the 192 of 298 tasks without server-side edits, so neither shows a dollar figure. Success rates and verdicts are unaffected.",
      "date": "2026-10-05",
      "effort": "medium",
      "id": "swebench-lite-300-h2h",
      "model": "claude-sonnet-5-5",
      "n_common": 298,
      "note": "plain and distil are the swebench-outcome-300-medium run (2026-10-03), reused unchanged, so this section supersedes that run's own table; rtk, selective and provider-cm ran 2026-10-05 on the same model and effort, provider drift between the two dates is not controlled, and provider-cm used an aggressive clearing setting rather than Anthropic's defaults",
      "rows": [
        {
          "arm": "plain",
          "cost_basis": "cold",
          "cost_usd": 8.4416,
          "label": "plain (no compression)",
          "library": null,
          "n": 298,
          "precached_rows": [
            0,
            300
          ],
          "rate": 0.731544,
          "rate_ci": [
            0.678516,
            0.778677
          ],
          "resolved": 218,
          "status": "measured",
          "steps": 1360,
          "tokens_in": 6810575,
          "tokens_out": 357562,
          "usd_per_solved": 0.0387,
          "verdict": null,
          "version": null,
          "vs_plain_ci_pts": null,
          "vs_plain_pts": null
        },
        {
          "arm": "distil",
          "cost_basis": "cold",
          "cost_usd": 8.3653,
          "label": "distil",
          "library": null,
          "n": 298,
          "precached_rows": [
            0,
            300
          ],
          "rate": 0.714765,
          "rate_ci": [
            0.661021,
            0.763043
          ],
          "resolved": 213,
          "status": "measured",
          "steps": 1427,
          "tokens_in": 7081760,
          "tokens_out": 351388,
          "usd_per_solved": 0.0393,
          "verdict": "INCONCLUSIVE",
          "version": "1.56.3",
          "vs_plain_ci_pts": [
            -5.09,
            1.73
          ],
          "vs_plain_pts": -1.68
        },
        {
          "arm": "rtk",
          "cost_basis": "cold re-priced",
          "cost_usd": 7.894,
          "cost_usd_billed": 6.6917,
          "label": "RTK",
          "library": "rtk",
          "n": 298,
          "precached_rows": [
            190,
            300
          ],
          "rate": 0.734899,
          "rate_ci": [
            0.682026,
            0.781794
          ],
          "resolved": 219,
          "status": "measured",
          "steps": 1307,
          "tokens_in": 5811151,
          "tokens_out": 340032,
          "usd_per_solved": 0.036,
          "usd_per_solved_billed": 0.0306,
          "verdict": "NON-INFERIOR",
          "version": "0.51.0",
          "vs_plain_ci_pts": [
            -2.67,
            3.34
          ],
          "vs_plain_pts": 0.33
        },
        {
          "arm": "headroom",
          "label": "Headroom",
          "status": "pending run"
        },
        {
          "arm": "selective",
          "cost_basis": "confounded",
          "cost_usd": null,
          "cost_usd_billed": 9.8588,
          "label": "Selective Context",
          "library": "selective-context",
          "n": 298,
          "precached_rows": [
            167,
            300
          ],
          "rate": 0.704698,
          "rate_ci": [
            0.650564,
            0.753622
          ],
          "resolved": 210,
          "status": "measured",
          "steps": 2053,
          "tokens_in": 10771687,
          "tokens_out": 495426,
          "usd_per_solved": null,
          "usd_per_solved_billed": 0.0469,
          "verdict": "INCONCLUSIVE",
          "version": "0.1.4",
          "vs_plain_ci_pts": [
            -6.71,
            1.35
          ],
          "vs_plain_pts": -2.68
        },
        {
          "arm": "provider-cm",
          "cost_basis": "confounded",
          "cost_usd": null,
          "cost_usd_billed": 8.1546,
          "label": "Anthropic context editing",
          "library": "anthropic",
          "n": 298,
          "precached_rows": [
            139,
            300
          ],
          "rate": 0.704698,
          "rate_ci": [
            0.650564,
            0.753622
          ],
          "resolved": 210,
          "status": "measured",
          "steps": 1419,
          "tokens_in": 5643391,
          "tokens_out": 340589,
          "usd_per_solved": null,
          "usd_per_solved_billed": 0.0388,
          "verdict": "INCONCLUSIVE",
          "version": "1.11.0",
          "vs_plain_ci_pts": [
            -5.59,
            0.24
          ],
          "vs_plain_pts": -2.68
        }
      ],
      "source": "benchmarks/results/swebench-outcome-300-h2h"
    },
    {
      "benchmark": "SWE-bench Lite (official grader)",
      "cost_confound": {},
      "cost_note": null,
      "date": "2026-10-02",
      "effort": "low",
      "id": "swebench-lite-300-low",
      "model": "claude-sonnet-5-5",
      "n_common": 299,
      "note": "plain rows are reused from swebench-outcome-300 (same model, effort and tasks), as that run's report.md says",
      "rows": [
        {
          "arm": "plain",
          "cost_basis": "cold",
          "cost_usd": 7.1662,
          "label": "plain (no compression)",
          "library": null,
          "n": 299,
          "precached_rows": [
            0,
            300
          ],
          "rate": 0.702341,
          "rate_ci": [
            0.648215,
            0.751334
          ],
          "resolved": 210,
          "status": "measured",
          "steps": 1251,
          "tokens_in": 5114011,
          "tokens_out": 303228,
          "usd_per_solved": 0.0341,
          "verdict": null,
          "version": null,
          "vs_plain_ci_pts": null,
          "vs_plain_pts": null
        },
        {
          "arm": "distil",
          "cost_basis": "cold",
          "cost_usd": 7.1663,
          "label": "distil",
          "library": null,
          "n": 299,
          "precached_rows": [
            0,
            300
          ],
          "rate": 0.682274,
          "rate_ci": [
            0.627473,
            0.732451
          ],
          "resolved": 204,
          "status": "measured",
          "steps": 1290,
          "tokens_in": 5527181,
          "tokens_out": 296148,
          "usd_per_solved": 0.0351,
          "verdict": "INCONCLUSIVE",
          "version": "1.56.3",
          "vs_plain_ci_pts": [
            -5.47,
            1.45
          ],
          "vs_plain_pts": -2.01
        },
        {
          "arm": "rtk",
          "label": "RTK",
          "status": "pending run"
        },
        {
          "arm": "headroom",
          "label": "Headroom",
          "status": "pending run"
        },
        {
          "arm": "selective",
          "label": "Selective Context",
          "status": "pending run"
        },
        {
          "arm": "provider-cm",
          "label": "Anthropic context editing",
          "status": "pending run"
        }
      ],
      "source": "benchmarks/results/swebench-outcome-300-grepfix"
    },
    {
      "benchmark": "Terminal-Bench 2.1 pilot (cost_truth, neutral meter)",
      "date": "2026-09-25",
      "effort": null,
      "id": "terminal-bench-pilot",
      "model": "claude-sonnet-5",
      "n_common": 10,
      "note": "pilot, not a comparison (protocol section 7): 18 paired attempts per arm after 10 infra_error runs were excluded, run under emulated x86_64, and billed while distil priced claude-sonnet-5 at $3/$15 instead of the official $2/$10, so absolute dollars are 1.5x too high and ratios are unaffected. analysis.json carries no passing verdict for any arm",
      "rows": [
        {
          "arm": "plain",
          "cost_usd": 11.3228,
          "label": "control (no compression)",
          "n": 18,
          "rate": 0.777778,
          "resolved": 14,
          "status": "measured",
          "usd_per_solved": 0.8088,
          "version": null
        },
        {
          "arm": "distil",
          "cost_usd": 13.752,
          "label": "distil",
          "n": 18,
          "rate": 0.777778,
          "resolved": 14,
          "status": "measured",
          "usd_per_solved": 0.9823,
          "verdict": "pilot, not shown",
          "version": "1.54.0",
          "vs_plain_ci_pts": [
            -33.3,
            35.0
          ],
          "vs_plain_pts": 0.0
        },
        {
          "arm": "rtk",
          "cost_usd": 8.9744,
          "label": "RTK",
          "n": 18,
          "rate": 0.833333,
          "resolved": 15,
          "status": "measured",
          "usd_per_solved": 0.5983,
          "verdict": "pilot, not shown",
          "version": "0.50.0",
          "vs_plain_ci_pts": [
            -17.6,
            25.0
          ],
          "vs_plain_pts": 5.6
        },
        {
          "arm": "headroom",
          "cost_usd": 10.1557,
          "label": "Headroom",
          "n": 18,
          "rate": 0.722222,
          "resolved": 13,
          "status": "measured",
          "usd_per_solved": 0.7812,
          "verdict": "pilot, not shown",
          "version": "0.38.0",
          "vs_plain_ci_pts": [
            -29.4,
            11.8
          ],
          "vs_plain_pts": -5.6
        },
        {
          "arm": "selective",
          "label": "Selective Context",
          "status": "not in this harness"
        },
        {
          "arm": "provider-cm",
          "label": "Anthropic context editing",
          "status": "not in this harness"
        }
      ],
      "source": "benchmarks/results/cost_truth/pilot-20260925-145427"
    }
  ]
}
