{
 "kind": "backtest_out_of_sample",
 "market": "runs",
 "split": "2026-06-01",
 "holdout_end": "2026-07-16",
 "n_games": 558,
 "n_train_games": 3317,
 "date_min": "2026-06-01",
 "date_max": "2026-07-16",
 "features": [
  "t_runs_pg_30",
  "t_xwoba_30",
  "t_k_pct_30",
  "t_bb_pct_30",
  "t_hr_pa_30",
  "opp_runs_allowed_pg_30",
  "osp_k_prior",
  "osp_hr_prior",
  "osp_xwoba_prior",
  "opp_bp_xwoba_30",
  "opp_bp_k_pct_30",
  "opp_bp_pitches_3d",
  "park_runs_prior",
  "wx_air_density",
  "wx_temp_c",
  "wx_wind_out_cf",
  "wx_is_dome",
  "is_home"
 ],
 "run_cap": 14,
 "total_cap": 28,
 "store_coverage": {
  "advertised": "2026-06-01..2026-07-16 (n=558)",
  "actual_first": "2026-06-01",
  "actual_last": "2026-07-16",
  "last_full_slate": "2026-07-12",
  "n_slate_dates": 43,
  "games_on_final_date": 1,
  "store_max_date": "2026-07-18",
  "note": "frozen references are pinned to this coverage; if the store ingests past HOLDOUT_END the reproduction assertion fires by design rather than silently rescoring the gate"
 },
 "pairing": {
  "copula_scale": -0.016,
  "ci95": [
   -0.0465,
   0.0149
  ],
  "n_games": 3875,
  "raw_scale": -0.0029,
  "residual_scale": -0.0054,
  "decision": "independence (outer product)",
  "rationale": "CI bounds |rho| below 0.047, which cannot move the joint materially; measured, not assumed"
 },
 "extras": {
  "p_home_wins_given_tie": 0.5258,
  "n_extra_inning_games_train": 291,
  "applied": false,
  "note": "NO LONGER APPLIED. The joint is conditioned on H != A, so there is no tie mass left for an extras constant to resolve; conditioning splits it by the model's own asymmetry instead of a league constant. Retained as a published diagnostic. Whether a v1.1 winners model should reintroduce an extras-specific tilt is open."
 },
 "seed_stability": {
  "seeds": 100,
  "w1": {
   "pass_rate": 0.73,
   "mean": 0.2483,
   "sd": 0.0007,
   "min": 0.2465,
   "max": 0.2496,
   "margin_in_seed_sigma": 0.65
  },
  "t1": {
   "pass_rate": 1.0,
   "mean": 2.5472,
   "sd": 0.0052,
   "min": 2.5357,
   "max": 2.5616,
   "margin_in_seed_sigma": 3.97
  }
 },
 "winners": {
  "gate": "W-1 Brier < 0.2488",
  "passed": false,
  "passed_single_fit": true,
  "verdict": "DOES NOT CLEAR reliably - only 73% of refits; the single-fit result is seed-dependent, and is NOT statistically distinguishable from climatology (gap CI [-0.0060, +0.0011])",
  "basis": {
   "gate_value": 0.2488,
   "gate_basis": "pooled in-sample over all games (0.5350 * 0.4650)",
   "holdout_scored_climatology": 0.251,
   "holdout_oracle_climatology": 0.2499,
   "home_rate_train": 0.5396,
   "home_rate_holdout": 0.5072,
   "note": "home-field advantage collapsed in the holdout: 53.96% in train vs 50.72% over 6/1-7/16. Consequence: NO constant forecast can reach 0.2488 here -- even the oracle constant (p = the holdout rate) scores 0.2499. W-1 therefore demands genuine per-game discrimination, not a well-chosen base rate. Flagged for a dated BENCHMARKS clarification: W-1 and T-1 are stated on different bases (pooled in-sample vs holdout-scored)."
  },
  "model": {
   "brier": 0.2486,
   "ece": 0.0102,
   "logloss": 0.6904,
   "auc": 0.5393,
   "mean_p": 0.505,
   "p_std": 0.0254
  },
  "climatology": {
   "brier": 0.251,
   "ece": 0.0325,
   "logloss": 0.6952,
   "auc": 0.5,
   "mean_p": 0.5396,
   "p_std": 0.0
  },
  "persistence": {
   "brier": 0.2645,
   "ece": 0.1157,
   "logloss": 0.7234,
   "auc": 0.4701,
   "mean_p": 0.4974,
   "p_std": 0.0996
  },
  "null_is_home_only": {
   "brier": 0.25,
   "ece": 0.0042,
   "logloss": 0.6931,
   "auc": 0.5,
   "mean_p": 0.5114,
   "p_std": 0.0
  },
  "gap_vs_climatology": -0.0024,
  "gap_ci95": [
   -0.006,
   0.0011
  ]
 },
 "totals": {
  "gate": "T-1 mean CRPS < 2.5678",
  "passed": true,
  "passed_single_fit": true,
  "verdict": "CLEARS the pre-registered floor on 100% of refits, and IS distinguishable from climatology (gap CI [-0.0413, -0.0007])",
  "model": {
   "crps": 2.5466,
   "mean_total": 9.055,
   "lines": {
    "7.5": {
     "brier": 0.2403,
     "mean_p": 0.5914,
     "obs_rate": 0.5896
    },
    "8.5": {
     "brier": 0.2486,
     "mean_p": 0.5099,
     "obs_rate": 0.509
    },
    "9.5": {
     "brier": 0.2426,
     "mean_p": 0.4156,
     "obs_rate": 0.4265
    }
   }
  },
  "climatology": {
   "crps": 2.5678,
   "mean_total": 8.878,
   "lines": {
    "7.5": {
     "brier": 0.243,
     "mean_p": 0.5574,
     "obs_rate": 0.5896
    },
    "8.5": {
     "brier": 0.2504,
     "mean_p": 0.4878,
     "obs_rate": 0.509
    },
    "9.5": {
     "brier": 0.2458,
     "mean_p": 0.3916,
     "obs_rate": 0.4265
    }
   }
  },
  "persistence": {
   "crps": 2.6783,
   "mean_total": 9.17,
   "lines": {
    "7.5": {
     "brier": 0.2592,
     "mean_p": 0.6816,
     "obs_rate": 0.5896
    },
    "8.5": {
     "brier": 0.2643,
     "mean_p": 0.5734,
     "obs_rate": 0.509
    },
    "9.5": {
     "brier": 0.2548,
     "mean_p": 0.425,
     "obs_rate": 0.4265
    }
   }
  },
  "null_is_home_only": {
   "crps": 2.5604,
   "mean_total": 9.02,
   "lines": {
    "7.5": {
     "brier": 0.242,
     "mean_p": 0.5874,
     "obs_rate": 0.5896
    },
    "8.5": {
     "brier": 0.2499,
     "mean_p": 0.5056,
     "obs_rate": 0.509
    },
    "9.5": {
     "brier": 0.2448,
     "mean_p": 0.412,
     "obs_rate": 0.4265
    }
   }
  },
  "gate_is_a_floor": {
   "null_crps": 2.5604,
   "gate": 2.5678,
   "note": "a model with ONE binary feature and zero information scores this. The gate sits at essentially that level, so clearing T-1 means 'not worse than knowing nothing', NOT 'has skill'. Stated here so the launch narrative cannot imply otherwise."
  },
  "level_bias": {
   "model_mean_total": 9.0551,
   "observed_mean_total": 9.3459,
   "bias": -0.2908,
   "sigma": -1.46,
   "train_mean_total": 8.8776,
   "note": "the model tracks the TRAIN run environment almost exactly and the holdout ran hotter, so this is level drift, not a coding error. Same failure family as the HR summer drift behind HR-1/HR-2. Disclosed because a calibration product that hides a level miss is not a calibration product."
  },
  "weather": {
   "features": [
    "wx_air_density",
    "wx_temp_c",
    "wx_wind_out_cf",
    "wx_is_dome"
   ],
   "crps_with_wx": 2.5466,
   "crps_without_wx": 2.5495,
   "delta_crps": -0.0029,
   "share_of_edge_vs_climatology": null,
   "vintage_note": "BACKTEST weather is Open-Meteo ARCHIVE reanalysis, i.e. what actually happened. LIVE serving uses the FORECAST API at publish time. Live lift will be smaller than shown here and the size of that gap is NOT yet measured -- it needs a forecast/reanalysis paired sample at first-pitch hour, which we do not have. Same disclosure HR-2 carries."
  },
  "gap_vs_climatology": -0.0212,
  "gap_ci95": [
   -0.0413,
   -0.0007
  ]
 },
 "coherence": {
  "construction_guards": {
   "totals_sum_to_one": 4.440892098500626e-16,
   "totals_mean_matches_joint": 5.329070518200751e-15,
   "note": "algebraic identities; regression guards, not evidence"
  },
  "falsifiable": {
   "tie_mass_in_joint": 0.0,
   "tie_mass_removed_by_conditioning": 0.0984,
   "observed_tie_rate_in_store": 0.0,
   "p_even_model": 0.4461,
   "p_even_observed": 0.4391,
   "p_even_sigma": 0.33,
   "winner_vs_rundiff_sign_corr": 0.9781,
   "note": "parity is the check the old suite lacked; it is what detects impossible tie mass leaking into the strip"
  }
 },
 "censoring": {
  "team_runs_at_or_above_cap": 23,
  "team_runs_share": 0.02061,
  "totals_above_support": 2,
  "note": "class 14 is a '14 or more' lump, so the totals strip is slightly short in the right tail; CRPS charges for it"
 },
 "references_reproduced": true,
 "winners_diagnosis": {
  "auc": 0.5393,
  "p_std": 0.0254,
  "mean_p": 0.505,
  "home_rate_train": 0.5396,
  "structural_handicap": "Deriving P(win) from two runs-scored marginals is fighting the ninth-inning truncation measured in C2. The home team forfeits its last at-bat in 44.8% of games, and it does so PRECISELY WHEN WINNING -- an endogenous stopping rule. So home runs-scored understates home strength: the model's mean P(home) is 0.505 against a training home win rate of 0.540. The censoring also adds target noise for home rows, which is the leading hypothesis (not proof) for the weak AUC of 0.539 -- elite public winner models run ~0.60. Fixes worth costing in v1.1: model 9-inning-equivalent runs, or carry the truncation explicitly. A separate direct win model would score better and is REJECTED -- it breaks the one-engine coherence that is the whole point of this design."
 },
 "generated_at": "2026-07-20T00:49:54+00:00"
}