{
  "name": "MediationBench",
  "snapshot": "2026-07-24",
  "source_commit": "27c14ebcac7b49dcd3da569bce5cc52916f7fdf3",
  "source_artifact": "docs/research/analysis/compute_claims_2026-07-24.json",
  "source_artifact_sha256": "d9a6c84f6c6bb448a348bc1fd02c8ed9e68490bf3320b9cdcbea9001d0ddbeff",
  "download_url": "/data/mediationbench-2026-07-24.json",
  "design_registry": "2.5.0",
  "suite_provenance": "The v2.5 information-study registry combines contemporaneous v2.5 cells with protocol-matched historical full-brief skilled arms that retain their original suite stamps.",
  "joint_scenarios": 62,
  "breadth_generation_status": "All six newly frozen breadth cells completed 62 of 62 scenarios after repair, with zero remaining failures.",
  "estimand": "Each point estimate is a difference between arm medians of per-scenario replicate means across 62 scenarios jointly completed in all four arms. Each 90% percentile interval comes from the frozen four-arm scenario-co-resampled bootstrap (B=10000; seed=20260706), drawing one observed replicate per (scenario, arm) within each draw. All intervals are conditional on the observed generation runs and operational judge; they are not run-sampling or human-efficacy intervals.",
  "noninferiority_rule": "E3 passes when its 90% interval lower bound is greater than -4.0 composite points. This is a non-inferiority rule, not an equivalence test; an interval spanning zero does not establish equivalence.",
  "historical_reuse": "For every tested backbone, E1 and the reflective-policy information contrast use contemporaneously generated v2.5 cells. E2, E3, and the structured-policy information contrast reuse one or more protocol-matched historical full-brief skilled runs.",
  "participant": {
    "configuration": "High-resistance open-weight simulated disputants",
    "model_id": "Sao10K/L3-8B-Stheno-v3.2"
  },
  "operational_judge": {
    "label": "Qwen3-235B-A22B-Instruct",
    "model_id": "Qwen/Qwen3-235B-A22B-Instruct-2507"
  },
  "methodology_version": "seeded_25_v1",
  "sampling": "unseeded",
  "temperatures": {
    "judge": 0.3,
    "mediator": 0.7,
    "participant": 0.8
  },
  "publication_policy": {
    "generation_arms": "Every registered clean generation arm used by this frozen study is included, including public-context-only ('blind') mediator arms and null or negative outcomes.",
    "primary_analysis": "The operational Qwen judge drives the E1-E4 study estimates.",
    "robustness": "Preplanned fixed-transcript robustness analyses are separated from clearly labeled exploratory diagnostics.",
    "rejudges": "A rejudge is a fixed-transcript sensitivity measurement, never a new generation run or generation replicate."
  },
  "information_conditions": {
    "full_bilateral_brief": "The mediator receives both parties' researcher-authored private scenario facts verbatim alongside the public scenario and conversation.",
    "public_context_only": "The mediator receives the public scenario, shared facts, and conversation, and must elicit relevant private information during the exchange.",
    "preparation_analogy": "The full bilateral brief is a controlled analogue of preparation through intake or caucus; it is not an observed human interview."
  },
  "backbones": [
    {
      "id": "qwen",
      "label": "Qwen3-235B",
      "evidence": "Replicated prospectively frozen core",
      "replicates": "2 / 2 / 2 / 2",
      "run_ids": {
        "skilled_full": [
          "0928d6af-a940-4a55-8ba5-20bf21445c87",
          "546966c9-f3e0-4d9b-948e-d0f81a6eae29"
        ],
        "skilled_public_context": [
          "986c4f8c-bbf4-4c11-9f09-11b3c7a39d43",
          "0e3e0f47-e0cf-4a82-be0e-42bd2432e8d2"
        ],
        "reflective_full": [
          "df31ef89-7280-4ce6-b625-db84c31687f0",
          "83752dc7-8edf-4a3d-8a5e-aceeb6e29cb0"
        ],
        "reflective_public_context": [
          "77e6b874-3163-4955-ba13-aa6d0f4031d4",
          "f0b1d261-9f3d-40b6-95df-f1b8b6245149"
        ]
      },
      "conversation_only_skill": {
        "point": 13.17,
        "low": 9.76,
        "high": 20.18
      },
      "prepared_skill": {
        "point": 7.27,
        "low": 1.68,
        "high": 15.07
      },
      "interaction": {
        "point": 5.9,
        "low": -0.42,
        "high": 16.11,
        "frozen_read": "Met the prospectively frozen four-point non-inferiority rule"
      },
      "conversation_only_minus_prepared": {
        "reflective_control": {
          "point": -5.89,
          "low": -17.56,
          "high": -2.32
        },
        "skilled": {
          "point": 0.01,
          "low": -6.24,
          "high": 0.85
        }
      }
    },
    {
      "id": "kimi",
      "label": "Kimi K2.5",
      "evidence": "Conditional single-run breadth",
      "replicates": "2 / 1 / 1 / 1",
      "run_ids": {
        "skilled_full": [
          "2068ccb2-818c-4b74-bb75-5993dd4e5503",
          "d05fcf37-7270-40d1-a077-0ee9de582ec7"
        ],
        "skilled_public_context": [
          "726730c8-6791-43ad-ad3b-02e50bd7465c"
        ],
        "reflective_full": [
          "83a48744-cb7b-4794-a5a2-93c7c930603d"
        ],
        "reflective_public_context": [
          "1b711261-7c3d-47c3-866b-2f7dc415a490"
        ]
      },
      "conversation_only_skill": {
        "point": 9.66,
        "low": 2.2,
        "high": 16.39
      },
      "prepared_skill": {
        "point": 1.68,
        "low": -0.45,
        "high": 7.56
      },
      "interaction": {
        "point": 7.98,
        "low": -1.74,
        "high": 13.65,
        "frozen_read": "Met the inherited frozen four-point rule; exploratory"
      },
      "conversation_only_minus_prepared": {
        "reflective_control": {
          "point": -6.45,
          "low": -12.08,
          "high": 0.74
        },
        "skilled": {
          "point": 1.53,
          "low": -3.0,
          "high": 3.28
        }
      }
    },
    {
      "id": "gptoss",
      "label": "gpt-oss-120b",
      "evidence": "Conditional single-run breadth",
      "replicates": "1 / 1 / 1 / 1",
      "run_ids": {
        "skilled_full": [
          "579b3609-7270-48b5-b2ec-08ec46d7233e"
        ],
        "skilled_public_context": [
          "d82f5d2a-1fd3-4c5b-b011-a28661c7e4d5"
        ],
        "reflective_full": [
          "cd46badc-69cc-4bec-a22a-60604567c7fb"
        ],
        "reflective_public_context": [
          "5f378266-2439-40ac-aedc-24aceb79ef3e"
        ]
      },
      "conversation_only_skill": {
        "point": -1.02,
        "low": -4.51,
        "high": 5.01
      },
      "prepared_skill": {
        "point": 4.74,
        "low": 2.15,
        "high": 15.53
      },
      "interaction": {
        "point": -5.75,
        "low": -16.63,
        "high": 0.32,
        "frozen_read": "Did not meet the frozen non-inferiority rule"
      },
      "conversation_only_minus_prepared": {
        "reflective_control": {
          "point": 4.12,
          "low": 0.33,
          "high": 15.1
        },
        "skilled": {
          "point": -1.64,
          "low": -4.57,
          "high": 3.38
        }
      }
    }
  ]
}
