{
  "name": "MediationBench",
  "snapshot": "2026-07-24",
  "source_commit": "abaff7086941f2d4a0da129e38b7019d88933db3",
  "source_commit_status": "Pinned to HAI research commit abaff7086941f2d4a0da129e38b7019d88933db3 (internal repository).",
  "source_artifact": "docs/research/analysis/compute_claims_2026-07-28.json",
  "source_artifact_sha256": "c1c53164a1441186e7369b8a1a3592a3dcf55021510322a68f0ef1d0c7e266b9",
  "download_url": "/data/mediationbench-2026-07-26-r3.json",
  "design_registry": "2.5.0",
  "suite_provenance": "The v2.5 information-study registry combines contemporaneous v2.5 cells with protocol-matched historical full-brief skilled arms that retain their original suite stamps.",
  "joint_scenarios": 62,
  "breadth_generation_status": "All six newly frozen generation cells completed 62 of 62 scenarios after repair, with zero remaining failures. Every regeneration was triggered by completion status alone - a failed scenario or a zero-token generation - and never by the score a run produced (attested by the study author, 2026-07-26).",
  "estimand": "Each point estimate is a difference between arm medians of per-scenario replicate means across 62 scenarios jointly completed in all four arms. Each 90% percentile interval comes from the four-arm scenario-co-resampled bootstrap (B=10000; seed=20260706): scenarios are resampled once per draw across all arms, each arm's observed per-scenario replicate mean is held fixed, and every draw recomputes the point-estimate statistic. An earlier implementation independently resampled replicate identity within scenarios, creating hybrid pseudo-runs; this correction changes intervals only. All intervals are conditional on the observed generation runs and operational judge; they are not run-sampling or human-efficacy intervals.",
  "positivity_rule": "E1 passes when its 90% interval lower bound is greater than 0 composite points.",
  "rules_provenance": "Two decision rules were frozen 2026-07-15, before any public-context run existed: E1 positivity and E3 non-inferiority.",
  "noninferiority_rule": "E3 passes when its 90% interval lower bound is greater than -4.0 composite points. This is a non-inferiority rule, not an equivalence test; an interval spanning zero does not establish equivalence.",
  "historical_reuse": "For every tested backbone, E1 and the reflective-policy information contrast use contemporaneously generated v2.5 cells. E2, E3, and the skilled-policy information contrast reuse one or more protocol-matched historical full-brief skilled runs.",
  "participant": {
    "configuration": "High-resistance open-weight simulated disputants",
    "model_id": "Sao10K/L3-8B-Stheno-v3.2"
  },
  "operational_judge": {
    "label": "Qwen3-235B-A22B-Instruct",
    "model_id": "Qwen/Qwen3-235B-A22B-Instruct-2507"
  },
  "methodology_version": "seeded_25_v1",
  "sampling": "unseeded",
  "temperatures": {
    "judge": 0.3,
    "mediator": 0.7,
    "participant": 0.8
  },
  "publication_policy": {
    "generation_arms": "Publication is outcome-independent by rule: every registered clean generation arm is queued for publication, including public-context-only ('blind') mediator arms and null or negative outcomes. As of 2026-07-26 the published export (generated 2026-07-26) contains all 17 of this study's runs; the five participant-transport runs (Gemma-4-31B participants) are published alongside them under the same rule.",
    "primary_analysis": "The operational Qwen judge drives the E1-E4 study estimates.",
    "robustness": "Preplanned fixed-transcript robustness analyses are separated from clearly labeled exploratory diagnostics.",
    "rejudges": "A rejudge is a fixed-transcript sensitivity measurement, never a new generation run or generation replicate. The outcome-independent publication rule covers registered fixed-transcript rejudge series as well as generation arms. A graph row's date is the transcript-generation date, not the rejudge scoring date; scoring dates belong in separate release provenance."
  },
  "information_conditions": {
    "full_bilateral_brief": "The mediator receives both parties' researcher-authored private scenario facts verbatim alongside the public scenario and conversation.",
    "public_context_only": "The mediator receives the public scenario, shared facts, and conversation, and must elicit relevant private information during the exchange.",
    "preparation_analogy": "The full bilateral brief is a controlled analogue of preparation through intake or caucus; it is not an observed human interview."
  },
  "backbones": [
    {
      "id": "qwen",
      "label": "Qwen3-235B",
      "model_id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
      "evidence": "Two runs per condition — the replicated frozen core",
      "replicates": "2 / 2 / 2 / 2",
      "run_ids": {
        "skilled_full": [
          "0928d6af-a940-4a55-8ba5-20bf21445c87",
          "546966c9-f3e0-4d9b-948e-d0f81a6eae29"
        ],
        "skilled_public_context": [
          "986c4f8c-bbf4-4c11-9f09-11b3c7a39d43",
          "0e3e0f47-e0cf-4a82-be0e-42bd2432e8d2"
        ],
        "reflective_full": [
          "df31ef89-7280-4ce6-b625-db84c31687f0",
          "83752dc7-8edf-4a3d-8a5e-aceeb6e29cb0"
        ],
        "reflective_public_context": [
          "77e6b874-3163-4955-ba13-aa6d0f4031d4",
          "f0b1d261-9f3d-40b6-95df-f1b8b6245149"
        ]
      },
      "conversation_only_skill": {
        "point": 13.17,
        "low": 9.67,
        "high": 17.55
      },
      "prepared_skill": {
        "point": 7.27,
        "low": 1.31,
        "high": 10.52
      },
      "interaction": {
        "point": 5.9,
        "low": 1.34,
        "high": 13.81,
        "frozen_read": "Met the prospectively frozen four-point non-inferiority rule"
      },
      "conversation_only_minus_prepared": {
        "reflective_control": {
          "point": -5.89,
          "low": -12.89,
          "high": -2.52
        },
        "skilled": {
          "point": 0.01,
          "low": -4.15,
          "high": 4.28
        }
      }
    },
    {
      "id": "kimi",
      "label": "Kimi K2.5",
      "model_id": "moonshotai/Kimi-K2.5",
      "evidence": "One new run per condition; conditional",
      "replicates": "2 / 1 / 1 / 1",
      "run_ids": {
        "skilled_full": [
          "2068ccb2-818c-4b74-bb75-5993dd4e5503",
          "d05fcf37-7270-40d1-a077-0ee9de582ec7"
        ],
        "skilled_public_context": [
          "726730c8-6791-43ad-ad3b-02e50bd7465c"
        ],
        "reflective_full": [
          "83a48744-cb7b-4794-a5a2-93c7c930603d"
        ],
        "reflective_public_context": [
          "1b711261-7c3d-47c3-866b-2f7dc415a490"
        ]
      },
      "conversation_only_skill": {
        "point": 9.66,
        "low": 2.32,
        "high": 16.21
      },
      "prepared_skill": {
        "point": 1.68,
        "low": -2.06,
        "high": 5.85
      },
      "interaction": {
        "point": 7.98,
        "low": -0.32,
        "high": 15.47,
        "frozen_read": "Met the inherited frozen four-point rule; exploratory"
      },
      "conversation_only_minus_prepared": {
        "reflective_control": {
          "point": -6.45,
          "low": -11.99,
          "high": 0.6
        },
        "skilled": {
          "point": 1.53,
          "low": -1.64,
          "high": 5.75
        }
      }
    },
    {
      "id": "gptoss",
      "label": "gpt-oss-120b",
      "model_id": "gpt-oss-120b",
      "evidence": "One run per condition; conditional",
      "replicates": "1 / 1 / 1 / 1",
      "run_ids": {
        "skilled_full": [
          "579b3609-7270-48b5-b2ec-08ec46d7233e"
        ],
        "skilled_public_context": [
          "d82f5d2a-1fd3-4c5b-b011-a28661c7e4d5"
        ],
        "reflective_full": [
          "cd46badc-69cc-4bec-a22a-60604567c7fb"
        ],
        "reflective_public_context": [
          "5f378266-2439-40ac-aedc-24aceb79ef3e"
        ]
      },
      "conversation_only_skill": {
        "point": -1.02,
        "low": -4.51,
        "high": 5.01
      },
      "prepared_skill": {
        "point": 4.74,
        "low": 2.19,
        "high": 15.55
      },
      "interaction": {
        "point": -5.75,
        "low": -16.75,
        "high": 0.38,
        "frozen_read": "Did not meet the frozen non-inferiority rule"
      },
      "conversation_only_minus_prepared": {
        "reflective_control": {
          "point": 4.12,
          "low": 0.23,
          "high": 15.22
        },
        "skilled": {
          "point": -1.64,
          "low": -4.59,
          "high": 3.38
        }
      }
    }
  ],
  "license": {
    "id": "CC-BY-4.0",
    "name": "Creative Commons Attribution 4.0 International",
    "url": "https://creativecommons.org/licenses/by/4.0/",
    "attribution": "Human Assisted Intelligence, PBC. MediationBench. Snapshot 2026-07-24. https://mediationbench.com/results/",
    "scope": "Covers the aggregate estimates, decision rules, and provenance in this file. The narrative text and figures on mediationbench.com remain CC BY-NC-SA 4.0. No license covers the held-out evaluation set, private transcripts, raw prompts, or scoring internals."
  },
  "errata": "r3 (2026-07-28): metadata-only successor to mediationbench-2026-07-26-r2.json; adds the committed HAI source revision abaff7086941f2d4a0da129e38b7019d88933db3 without rewriting r2. The r2 correction replaced hybrid pseudo-run resampling with four-arm scenario co-resampling while holding observed per-scenario replicate means fixed. Point estimates, intervals, estimands, margins, and frozen-rule verdicts are unchanged in r3.",
  "participant_transport": {
    "axis": "participant configuration — mediator backbone, both policy prompts, information conditions, judge, scenario set, and suite held fixed",
    "participant_model": "gemma-4-31b",
    "participant_provider": "cerebras",
    "mediator_backbone_model_id": "Qwen/Qwen3-235B-A22B-Instruct-2507",
    "snapshot": "2026-07-26",
    "design": "Policy x information 2x2 plus a scenario-matched no-mediator baseline; one run per arm (1/1/1/1); exploratory post-launch, pre-outcome amendment frozen 2026-07-25T04:19Z before any arm had evaluated scenarios.",
    "run_ids": {
      "baseline": "d7275905-c038-4ce6-bf58-66efaa6e399d",
      "skilled_full": "29ee60cc-7c87-4cf0-acb9-37fca9c014db",
      "skilled_public_context": "1650e2f0-9a95-4422-9aa9-97397c1114d9",
      "reflective_full": "c4f0de97-f99a-4775-a3ff-eaa7a80051d7",
      "reflective_public_context": "0f77a820-61f0-4b75-9b7c-3fd51af35589"
    },
    "run_scores": {
      "baseline": 15,
      "skilled_full": 41,
      "skilled_public_context": 38,
      "reflective_full": 34,
      "reflective_public_context": 35
    },
    "conversation_only_skill": {
      "point": 5.7,
      "low": -3.79,
      "high": 12.9,
      "frozen_read": "Failed the prospectively frozen positivity rule"
    },
    "prepared_skill": {
      "point": 11.39,
      "low": 5.77,
      "high": 16.92
    },
    "interaction": {
      "point": -5.68,
      "low": -15.95,
      "high": 2.3,
      "frozen_read": "Failed the prospectively frozen four-point non-inferiority rule"
    },
    "conversation_only_minus_prepared": {
      "reflective_control": {
        "point": 0.73,
        "low": -3.66,
        "high": 6.34
      },
      "skilled": {
        "point": -4.95,
        "low": -13.69,
        "high": 1.94
      }
    },
    "baseline_deltas": {
      "skilled_full": {
        "point": 29.29,
        "low": 23.88,
        "high": 34.72
      },
      "skilled_public_context": {
        "point": 24.33,
        "low": 16.99,
        "high": 30.71
      },
      "reflective_full": {
        "point": 17.9,
        "low": 14.35,
        "high": 22.22
      },
      "reflective_public_context": {
        "point": 18.63,
        "low": 14.15,
        "high": 25.06
      }
    },
    "note": "First failed frozen-rule outcome on the participant axis (gpt-oss failed both rules on the mediator-backbone axis). Mediation still cleared the no-mediator baseline in all four arms. One run per arm; this does not establish that the private brief is necessary. This is a participant column, not a mediator-backbone row."
  }
}
