{
  "schema_version": "future-shock.cmfc.public-summary.v1",
  "release_version": "0.1.0-stage1",
  "publication_status": "public_stage_1_release",
  "title": "When Do Diverse Models Fail Together?",
  "subtitle": "Difficulty, Shared Evidence, and Common Shocks Across Language-Model Fleets",
  "authors": [
    "Nicholas Zinner",
    "Beacon Bot"
  ],
  "date": "2026-08",
  "retrospective": {
    "models": 395,
    "exact_common_item_corpus": 27842,
    "support_qualified_model_pairs_reported": 3253,
    "possible_unordered_model_pairs": 77815,
    "reported_pair_selection_fraction": 0.04180427938058215,
    "pair_selection_reproducible_from_stage1": false,
    "raw_median_capa": 0.498,
    "raw_median_capa_ci": [
      0.494,
      0.503
    ],
    "median_2pl_residual_correlation": -0.018,
    "boundary": "The 3,253-pair selection count is reported from the execution-side audit. Stage 1 does not include the event-level support matrix needed to regenerate that count or reconstruct every pair-selection diagnostic."
  },
  "prospective": {
    "endpoints": 10,
    "fresh_items": 1500,
    "retained_exact_common_support_items": 1495,
    "cross_family_beta": 0.068,
    "cross_family_beta_ci": [
      0.039,
      0.093
    ]
  },
  "evidence_path_confirmations": {
    "country_primary_delta_h4": -0.46,
    "country_primary_ci": [
      -0.589,
      -0.327
    ],
    "city_secondary_delta_h4": -0.15,
    "city_secondary_ci": [
      -0.182,
      -0.119
    ],
    "interpretation": "The fixed endpoint substitution produced a larger decorrelation than the evidence-path contrast, but changed model identity, family, provider, size, and precision together."
  },
  "adversarial_transfer": {
    "semantic_instruction_conflict_amplification": 4.337,
    "semantic_instruction_conflict_ci": [
      3.358,
      5.885
    ],
    "token_suffix_amplification": 0.826,
    "token_suffix_ci": [
      0.642,
      1.086
    ],
    "boundary": "These are separate mechanisms from the recoverable common-shock experiment and are not pooled."
  },
  "common_shock_primary_contrasts": [
    {
      "population": "legacy",
      "shock": "evidence corruption",
      "shared_condition": "E_shared",
      "independent_condition": "E_independent",
      "planned_items": 800,
      "matched_items": 665,
      "matched_fraction": 0.83125,
      "support_classification": "estimable_degraded",
      "dependence_inflation_ratio": 1.0951136227087999,
      "simultaneous_ci_low": 1.015811478913401,
      "simultaneous_ci_high": 1.1806067085648981,
      "mean_loss_difference": 0.0018379281537176384,
      "absolute_volatility_ratio": 1.0600081254773286
    },
    {
      "population": "legacy",
      "shock": "instruction anchor",
      "shared_condition": "A_shared",
      "independent_condition": "A_independent",
      "planned_items": 800,
      "matched_items": 682,
      "matched_fraction": 0.8525,
      "support_classification": "estimable_degraded",
      "dependence_inflation_ratio": 0.9787596741380908,
      "simultaneous_ci_low": 0.9248247736765747,
      "simultaneous_ci_high": 1.0358400066539726,
      "mean_loss_difference": 0.0029325513196480912,
      "absolute_volatility_ratio": 0.9893697369608665
    },
    {
      "population": "contemporary",
      "shock": "evidence corruption",
      "shared_condition": "E_shared",
      "independent_condition": "E_independent",
      "planned_items": 100,
      "matched_items": 97,
      "matched_fraction": 0.97,
      "support_classification": "confirmatory_quality",
      "dependence_inflation_ratio": 1.0071561093030503,
      "simultaneous_ci_low": 0.8883296621984531,
      "simultaneous_ci_high": 1.1418772463322842,
      "mean_loss_difference": -0.0068728522336770626,
      "absolute_volatility_ratio": 1.0062903351186923
    },
    {
      "population": "contemporary",
      "shock": "instruction anchor",
      "shared_condition": "A_shared",
      "independent_condition": "A_independent",
      "planned_items": 100,
      "matched_items": 94,
      "matched_fraction": 0.94,
      "support_classification": "confirmatory_quality",
      "dependence_inflation_ratio": 1.0329067823678943,
      "simultaneous_ci_low": 0.925055067007234,
      "simultaneous_ci_high": 1.153332876185687,
      "mean_loss_difference": -0.015957446808510634,
      "absolute_volatility_ratio": 1.0304171508971272
    }
  ],
  "release_posture": "Finite-panel reliability study. Not a model ranking, deployment rule, or demonstration that shared evidence generally erases model diversity.",
  "reproducibility_level": "report-level inspection and receipt reconstruction; not historical model-call reproduction"
}
