{
  "timestamp_utc": "2026-06-18T00:00:00.000Z",
  "benchmark_name": "real-developer-transcripts",
  "rows_evaluated": 42,
  "savings_median_percent": 25.26,
  "savings_headline_percent": 43.14,
  "savings_p10": null,
  "savings_p50": 25.26,
  "savings_p90": null,
  "parity_rate_percent": 26.2,
  "p_value": 0.2114,
  "quality_judge_pass_rate": 26.2,
  "gate_verdict": "NARROW",
  "stratified_breakdown": [
    {
      "name": "real-developer-transcripts",
      "rows": 42,
      "median_savings_pct": 25.26,
      "parity_rate_pct": 26.2,
      "p_value": 0.2114,
      "dataset_revision": "claude-code-transcripts-2026-06-18"
    }
  ],
  "stratified_axes": {
    "source": "Per-axis cohorts for the real Claude Code transcript run. Each axis partitions the run's 42 valid replayed pairs, so the per-cohort rows sum to the rows evaluated rather than to a larger superseded run. Savings is the measured per-cohort median input-token reduction, and the large-context cohort holds its measured count of 31 pairs over 8K input tokens. Per-cohort parity and Wilcoxon p were not measured at the cohort level on this run (parity was measured only at the run level, 26.2 percent over 42 pairs), so they are null here rather than a fabricated per-cohort number. The tool_count and output_size axes were not stratified on this transcript run, so they carry no cohorts and render the honest absent view.",
    "task_category": [
      {
        "name": "code_gen",
        "rows": 33,
        "median_savings_pct": 39.47,
        "parity_rate_pct": null,
        "p_value": null
      },
      {
        "name": "chat",
        "rows": 9,
        "median_savings_pct": 0.51,
        "parity_rate_pct": null,
        "p_value": null
      }
    ],
    "input_complexity": [
      {
        "name": "large (input over 8K tokens)",
        "rows": 31,
        "median_savings_pct": 43.17,
        "parity_rate_pct": null,
        "p_value": null
      },
      {
        "name": "medium (input 2K to 8K tokens)",
        "rows": 6,
        "median_savings_pct": 18.89,
        "parity_rate_pct": null,
        "p_value": null
      },
      {
        "name": "small (input under 2K tokens)",
        "rows": 5,
        "median_savings_pct": 0.0,
        "parity_rate_pct": null,
        "p_value": null
      }
    ],
    "tool_count": [],
    "output_size": []
  },
  "worker_model": "gemini-3.1-pro-preview",
  "judge_model": "gemini-3.1-pro-preview",
  "commit_sha": "",
  "run_id": "cl-realdata-cc-2026-06-18",
  "source_run_tag": "2026-W25"
}
