{
 "generated": "2026-09-05",
 "source": "forensic-rca-calibration-study (private); values recomputed from telemetry/fit.json, figdata.all_rows(), Langfuse trace latency; Wave 2 block is the locked public sheet of 2026-08-17 verbatim",
 "study": {
  "incidents": 102,
  "clean_paired_cases": 87,
  "arms": 5,
  "judge_passes": 3,
  "judge_panel_models": 2,
  "agent_runs_total": 510,
  "wave2_cells": 408,
  "wave2_standard_runs": 368,
  "evaluation_set": "102 closed support incidents with confirmed root causes; a standing test harness for models and tools, not live traffic",
  "harness_history": "started at 25 cases; expanded to 102 after 25 proved too small to separate models",
  "standard_scored": 92,
  "dirty_excluded": 5
 },
 "wave2_public": {
  "exact_pct": {
   "gemini-3.7-flash": 37,
   "gemini-3.6-flash": 38,
   "claude-opus-4-6": 48,
   "gemini-3.1-pro-preview": 32
  },
  "grounded_pct": {
   "gemini-3.7-flash": 59,
   "gemini-3.6-flash": 76,
   "claude-opus-4-6": 43,
   "gemini-3.1-pro-preview": 60
  },
  "mcnemar_37_vs_36": {
   "a": 14,
   "b": 13,
   "p": 1.0
  },
  "opus_usd_per_incident": 9.56,
  "flash_usd_per_incident": {
   "gemini-3.6-flash": 0.72,
   "gemini-3.7-flash": 1.19
  },
  "flash_price_per_M_in": 0.75,
  "input_tok_M": {
   "gemini-3.7-flash": 1.55,
   "gemini-3.6-flash": 0.91
  },
  "input_ratio": 1.7,
  "tool_calls": {
   "gemini-3.7-flash": 22,
   "gemini-3.6-flash": 16
  },
  "tool_calls_delta_pct": 37,
  "auroc_combined_proxy_vs_usable": 0.46,
  "auroc_combined_p": 0.315,
  "auroc_grounded_flag_vs_exact": 0.518,
  "auroc_flag_p": 0.661,
  "auroc_flag_ci": [
   0.447,
   0.58
  ],
  "solved_by_nobody": {
   "k": 32,
   "n": 92,
   "pct": 35
  },
  "permutation_reps": 10000,
  "input_ratio_medians": 1.6
 },
 "self_correction": {
  "n_small": 25,
  "auroc_small": 0.62,
  "mcc_small": 0.36,
  "n_large": 102,
  "n_large_standard": 92,
  "auroc_large": 0.473,
  "mcc_large": -0.075,
  "confidently_wrong_pct_small": 16,
  "confidently_wrong_pct_same25_later": 24,
  "confidently_wrong_pct_large": 20
 },
 "wave3": {
  "gemini-3.8-flash": {
   "n": 87,
   "usable_pct": 89.66,
   "exact_pct": 52.87,
   "wrong_pct": 10.34,
   "grounded_pct": 57.47,
   "turns_median": 37.0,
   "calls_median": 36.5,
   "calls_mean": 38.48,
   "calls_per_active_turn": 1.011,
   "single_call_pct": 99.0,
   "active_turns": 3884,
   "input_tok_median_M": 3.09,
   "c0": 15884,
   "s_per_turn": 3853,
   "wall_median_min": 6.1,
   "wall_p90_min": 14.5,
   "concurrent_turns": 40,
   "same_tool_in_turn": 0,
   "same_tool_opps": 1929,
   "post_cd_calls_per_turn": 1.005
  },
  "claude-opus-4-6": {
   "n": 87,
   "usable_pct": 78.16,
   "exact_pct": 48.28,
   "wrong_pct": 21.84,
   "grounded_pct": 44.83,
   "turns_median": 19.0,
   "calls_median": 33.0,
   "calls_mean": 34.56,
   "calls_per_active_turn": 1.85,
   "single_call_pct": 44.2,
   "active_turns": 1905,
   "input_tok_median_M": 1.38,
   "c0": 22197,
   "s_per_turn": 6134,
   "wall_median_min": 6.6,
   "wall_p90_min": 24.9,
   "concurrent_turns": 1063,
   "same_tool_in_turn": 625,
   "same_tool_opps": 1226,
   "post_cd_calls_per_turn": 1.377
  },
  "gemini-3.7-flash": {
   "n": 87,
   "usable_pct": 78.16,
   "exact_pct": 37.93,
   "wrong_pct": 21.84,
   "grounded_pct": 59.77,
   "turns_median": 22.0,
   "calls_median": 22.5,
   "calls_mean": 22.68,
   "calls_per_active_turn": 1.053,
   "single_call_pct": 94.8,
   "active_turns": 2196,
   "input_tok_median_M": 1.36,
   "c0": 15884,
   "s_per_turn": 4478,
   "wall_median_min": 3.6,
   "wall_p90_min": 19.3,
   "concurrent_turns": 115,
   "same_tool_in_turn": 4,
   "same_tool_opps": 806,
   "post_cd_calls_per_turn": 1.018
  },
  "gemini-3.6-flash": {
   "n": 87,
   "usable_pct": 77.01,
   "exact_pct": 39.08,
   "wrong_pct": 22.99,
   "grounded_pct": 74.71,
   "turns_median": 16.0,
   "calls_median": 16.0,
   "calls_mean": 16.34,
   "calls_per_active_turn": 1.049,
   "single_call_pct": 95.2,
   "active_turns": 1589,
   "input_tok_median_M": 0.87,
   "c0": 15884,
   "s_per_turn": 4582,
   "wall_median_min": 2.7,
   "wall_p90_min": 14.5,
   "concurrent_turns": 76,
   "same_tool_in_turn": 0,
   "same_tool_opps": 452,
   "post_cd_calls_per_turn": 1.005
  },
  "gemini-3.1-pro-preview": {
   "n": 87,
   "usable_pct": 67.82,
   "exact_pct": 29.89,
   "wrong_pct": 32.18,
   "grounded_pct": 60.92,
   "turns_median": 15.0,
   "calls_median": 16.5,
   "calls_mean": 17.47,
   "calls_per_active_turn": 1.169,
   "single_call_pct": 84.2,
   "active_turns": 1524,
   "input_tok_median_M": 0.77,
   "c0": 15884,
   "s_per_turn": 4784,
   "wall_median_min": 3.8,
   "wall_p90_min": 17.3,
   "concurrent_turns": 240,
   "same_tool_in_turn": 68,
   "same_tool_opps": 450,
   "post_cd_calls_per_turn": 1.087
  }
 },
 "wave3_paired": {
  "usable_38_vs_opus": {
   "a": 14,
   "b": 4,
   "p": 0.0192
  },
  "exact_38_vs_opus": {
   "a": 12,
   "b": 8,
   "p": 0.383
  },
  "mcnemar_usable_vs_38": {
   "claude-opus-4-6": 0.01921,
   "gemini-3.7-flash": 0.00342,
   "gemini-3.6-flash": 0.01182,
   "gemini-3.1-pro-preview": 4e-05
  },
  "holm_all_pass": true,
  "more_turns_38_vs_opus": {
   "more": 95,
   "of": 102
  },
  "usable_38_vs_36": {
   "a": 15,
   "b": 4
  },
  "mid_p_note": "mid-p = exact two-sided minus the full point mass (Fagerland, Lydersen & Laake 2013). Values before 2026-09-05 subtracted half the point mass and were conservative."
 },
 "prefill_s_per_100k": {
  "gemini-3.6-flash": 1.17,
  "gemini-3.7-flash": 1.18,
  "gemini-3.8-flash": 1.19
 },
 "wallclock_share": {
  "gemini-3.8-flash": {
   "generation": 65,
   "tools": 34
  },
  "gemini-3.6-flash": {
   "generation": 38,
   "tools": 62
  },
  "gemini-3.7-flash": {
   "generation": 44,
   "tools": 55
  }
 },
 "context": {
  "descending_runs": 0,
  "runs": 510,
  "closed_form_arm_err_max_pct": 8,
  "per_run_abs_err_pct": [
   18,
   26
  ]
 },
 "excluded_from_release": [
  "s38_3_calibration (held; DM only)",
  "w2_calibration_dead (title uses banned wording)",
  "per-incident tables",
  "cache hit rate (unmeasured)",
  "w4_ensemble_coverage (not referenced by the report)",
  "judge variance decomposition beyond the standard-cell canary figures (all-20 cohort includes abstention cells)",
  "s1_overclaim_dumbbell (vendor-negative headline; duplicate of w1 at the wrong n)",
  "s2_cost_parity (footer n; dollar figures; superseded cost story)",
  "w3/w3b (pre-parity price panel)",
  "s3_zero_abstention (reserved for LinkedIn Post 2)"
 ],
 "died": {
  "early_turns_estimate_from_page_capped_cohort": 40.6,
  "early_turns_closed_form_misfit_pct": 23
 },
 "literature": {
  "brynjolfsson_li_raymond": {
   "source": "Generative AI at Work, NBER w31161 / QJE 140(2)",
   "resolutions_per_hour_pct": 13.8,
   "lowest_skill_quintile_pct": 34,
   "nps_effect": "null"
  }
 },
 "variance_canary_standard": {
  "n": 16,
  "total_label_flips": 5,
  "up": 3,
  "down": 2,
  "judge_only_flips": 2,
  "source": "research/variance_decomp.py, standard cells of the 20-case gemini-3.6-flash drift canary"
 },
 "output_s_per_1k_tokens": {
  "gemini-3.6-flash": 3.0,
  "gemini-3.7-flash": 4.6,
  "gemini-3.8-flash": 6.5
 }
}