{
 "slug": "weekly-calibration-2026-09-28",
 "series": "weekly",
 "series_label": "WEEKLY · CALIBRATION",
 "issue": 9,
 "title": "Weekly Calibration #9",
 "language": "en",
 "window_start": "2026-09-28T00:00:00+00:00",
 "window_end": "2026-10-05T00:00:00+00:00",
 "window_text": "Sep 28-Oct 4, 2026 UTC",
 "cutoff": "2026-10-05T16:00:00+00:00",
 "generated_at": "2026-10-07T07:28:57+00:00",
 "methodology": "v1.1 (2026-08-10)",
 "methodology_hash": "e66c7e8c864a2233",
 "source": "weekly_metrics_2026-09-28.json",
 "hit_rule": "direction: exit tp1/tp2 -> hit, sl -> miss, expiry -> sign of gross pnl",
 "bibtex_key": "mm_calibration_2026w40",
 "market_state": {
  "panel": "market_state_6p2",
  "week": {
   "start": "2026-09-28",
   "end": "2026-10-05"
  },
  "prev_week": {
   "start": "2026-09-21",
   "end": "2026-09-28"
  },
  "current": {
   "btc_net_pct": 2.44,
   "realized_vol_ann_pct": 21.13,
   "volume_top5_usd_bn": 17.9648,
   "volume_top5_wow_pct": -19.9,
   "avg_pairwise_corr": 0.57
  },
  "previous_as_published_issue8": {
   "btc_net_pct": 4.06,
   "realized_vol_ann_pct": 52.25,
   "volume_top5_usd_bn": 22.4265,
   "avg_pairwise_corr": 0.71
  },
  "snapshot_line": "BTC net +2.44% (prior +4.06%) - ann vol 21.1% (was 52.2%) - TOP5 volume $17.96B, -20% w/w - pairwise corr 0.57 (was 0.71)",
  "exchange": "binance",
  "audit_filter_applied": true,
  "source": "single-exchange (binance) daily candle recompute, market_state_2026-09-28.json; prior-week values as published in issue #8",
  "audit_note": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #8 after the audit that found the raw table mixes two exchanges; issue #8 values are as published.",
  "prev_week_control": {
   "published_issue3": {
    "btc_net_pct": 4.06,
    "realized_vol_ann_pct": 52.25,
    "volume_top5_usd_bn": 22.4265,
    "avg_pairwise_corr": 0.71
   },
   "published_issue8": {
    "btc_net_pct": 4.06,
    "realized_vol_ann_pct": 52.25,
    "volume_top5_usd_bn": 22.4265,
    "avg_pairwise_corr": 0.71
   },
   "got": {
    "btc_net_pct": 4.06,
    "realized_vol_ann_pct": 52.25,
    "volume_top5_usd_bn": 22.4265,
    "avg_pairwise_corr": 0.71
   },
   "match": {
    "btc_net_pct": true,
    "realized_vol_ann_pct": true,
    "volume_top5_usd_bn": true,
    "avg_pairwise_corr": true
   },
   "tolerance": {
    "btc_net_pct": 0.05,
    "realized_vol_ann_pct": 0.5,
    "volume_top5_usd_bn": 0.05,
    "avg_pairwise_corr": 0.02
   },
   "all_match": true,
   "note": "previous week of this run == published issue-#8 week; the four Snapshot pins must reproduce (key published_issue3 kept for the builder schema, published_issue8 is the add-only alias)"
  },
  "regime_days": {
   "2026-09-28": {
    "move_pct": -1.19,
    "regime": "flat"
   },
   "2026-09-29": {
    "move_pct": 0.22,
    "regime": "flat"
   },
   "2026-09-30": {
    "move_pct": -0.22,
    "regime": "flat"
   },
   "2026-10-01": {
    "move_pct": 1.54,
    "regime": "trend"
   },
   "2026-10-02": {
    "move_pct": -0.39,
    "regime": "flat"
   },
   "2026-10-03": {
    "move_pct": 0.4,
    "regime": "flat"
   },
   "2026-10-04": {
    "move_pct": 1.81,
    "regime": "trend"
   }
  },
  "alive_slice_context": {
   "abs_move_1d_pct": {
    "A": 1.835,
    "B": 0.873
   },
   "btc_realized_vol_ann_pct_hourly": {
    "A": 35.8,
    "B": 32.24
   }
  }
 },
 "tables": {
  "calibration_main": [
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "legacy": false,
    "is_field": false,
    "n": 431,
    "coverage": 0.326,
    "hit_rate": 0.5267,
    "mean_conf": 60.4,
    "gap_pp": 7.7,
    "brier": 0.2566,
    "wilson_95": {
     "lo": 0.4795,
     "hi": 0.5734
    }
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "legacy": false,
    "is_field": false,
    "n": 687,
    "coverage": 0.5165,
    "hit_rate": 0.5095,
    "mean_conf": 60.3,
    "gap_pp": 9.3,
    "brier": 0.2594,
    "wilson_95": {
     "lo": 0.4721,
     "hi": 0.5467
    }
   },
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "legacy": false,
    "is_field": false,
    "n": 711,
    "coverage": 0.535,
    "hit_rate": 0.4895,
    "mean_conf": 59.4,
    "gap_pp": 10.5,
    "brier": 0.2608,
    "wilson_95": {
     "lo": 0.4529,
     "hi": 0.5262
    }
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "legacy": false,
    "is_field": false,
    "n": 534,
    "coverage": 0.4015,
    "hit_rate": 0.5075,
    "mean_conf": 61.1,
    "gap_pp": 10.3,
    "brier": 0.2617,
    "wilson_95": {
     "lo": 0.4652,
     "hi": 0.5497
    }
   },
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "legacy": false,
    "is_field": false,
    "n": 623,
    "coverage": 0.4684,
    "hit_rate": 0.4992,
    "mean_conf": 61.0,
    "gap_pp": 11.1,
    "brier": 0.263,
    "wilson_95": {
     "lo": 0.4601,
     "hi": 0.5383
    }
   },
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "legacy": false,
    "is_field": false,
    "n": 694,
    "coverage": 0.5218,
    "hit_rate": 0.4899,
    "mean_conf": 66.6,
    "gap_pp": 17.7,
    "brier": 0.2858,
    "wilson_95": {
     "lo": 0.4529,
     "hi": 0.5271
    }
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "legacy": false,
    "is_field": false,
    "n": 651,
    "coverage": 0.4951,
    "hit_rate": 0.4654,
    "mean_conf": 67.9,
    "gap_pp": 21.3,
    "brier": 0.2951,
    "wilson_95": {
     "lo": 0.4274,
     "hi": 0.5038
    }
   },
   {
    "model": "Field (all models)",
    "legacy": false,
    "is_field": true,
    "n": 4331,
    "coverage": null,
    "hit_rate": 0.4964,
    "mean_conf": 62.5,
    "gap_pp": 12.9,
    "brier": 0.2698,
    "wilson_95": {
     "lo": null,
     "hi": null
    }
   }
  ],
  "buckets": [
   {
    "model": "grok-4.6",
    "bucket": "50-60",
    "n": 138,
    "hit_rate": 0.5072,
    "mean_conf": 57.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "60-70",
    "n": 291,
    "hit_rate": 0.5395,
    "mean_conf": 61.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "70-80",
    "n": 2,
    "hit_rate": 0.0,
    "mean_conf": 71,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "50-60",
    "n": 270,
    "hit_rate": 0.4926,
    "mean_conf": 56.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "60-70",
    "n": 401,
    "hit_rate": 0.5187,
    "mean_conf": 62.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "70-80",
    "n": 16,
    "hit_rate": 0.5625,
    "mean_conf": 70.6,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "50-60",
    "n": 345,
    "hit_rate": 0.4638,
    "mean_conf": 56.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "60-70",
    "n": 352,
    "hit_rate": 0.5085,
    "mean_conf": 62.3,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "70-80",
    "n": 4,
    "hit_rate": 0.75,
    "mean_conf": 70.5,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "50-60",
    "n": 103,
    "hit_rate": 0.534,
    "mean_conf": 57.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "60-70",
    "n": 431,
    "hit_rate": 0.5012,
    "mean_conf": 61.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-fable-5",
    "bucket": "50-60",
    "n": 201,
    "hit_rate": 0.4726,
    "mean_conf": 56.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "60-70",
    "n": 421,
    "hit_rate": 0.5131,
    "mean_conf": 63.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "70-80",
    "n": 1,
    "hit_rate": 0.0,
    "mean_conf": 70,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "50-60",
    "n": 5,
    "hit_rate": 0.6,
    "mean_conf": 55,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "60-70",
    "n": 458,
    "hit_rate": 0.4978,
    "mean_conf": 63.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "70-80",
    "n": 209,
    "hit_rate": 0.4641,
    "mean_conf": 73.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "80-100",
    "n": 22,
    "hit_rate": 0.5455,
    "mean_conf": 82.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "50-60",
    "n": 6,
    "hit_rate": 0.5,
    "mean_conf": 58.2,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "60-70",
    "n": 429,
    "hit_rate": 0.4615,
    "mean_conf": 65.3,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "70-80",
    "n": 216,
    "hit_rate": 0.4722,
    "mean_conf": 73.3,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   }
  ],
  "sub50": [
   {
    "model": "grok-4.6",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "deepseek-v4-pro",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "qwen-3.8-max",
    "n": 10,
    "hit_rate": 0.6
   },
   {
    "model": "claude-opus-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "claude-fable-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gemini-3.1-pro",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gpt-5.6-sol",
    "n": 0,
    "hit_rate": null
   }
  ],
  "by_fh": [
   {
    "model": "grok-4.6",
    "fh": "1h",
    "n": 301,
    "hit_rate": 0.5282,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "4h",
    "n": 123,
    "hit_rate": 0.5203,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "1d",
    "n": 7,
    "hit_rate": 0.5714,
    "flag": "N<10"
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1h",
    "n": 459,
    "hit_rate": 0.5033,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "4h",
    "n": 192,
    "hit_rate": 0.4948,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1d",
    "n": 36,
    "hit_rate": 0.6667,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1h",
    "n": 469,
    "hit_rate": 0.4861,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "4h",
    "n": 213,
    "hit_rate": 0.4836,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1d",
    "n": 29,
    "hit_rate": 0.5862,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1h",
    "n": 375,
    "hit_rate": 0.5147,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "4h",
    "n": 150,
    "hit_rate": 0.4933,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1d",
    "n": 9,
    "hit_rate": 0.4444,
    "flag": "N<10"
   },
   {
    "model": "claude-fable-5",
    "fh": "1h",
    "n": 434,
    "hit_rate": 0.5023,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "4h",
    "n": 177,
    "hit_rate": 0.4859,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "1d",
    "n": 12,
    "hit_rate": 0.5833,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1h",
    "n": 477,
    "hit_rate": 0.4969,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "4h",
    "n": 192,
    "hit_rate": 0.474,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1d",
    "n": 25,
    "hit_rate": 0.48,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1h",
    "n": 429,
    "hit_rate": 0.4615,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "4h",
    "n": 199,
    "hit_rate": 0.4623,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1d",
    "n": 23,
    "hit_rate": 0.5652,
    "flag": null
   }
  ],
  "trading": [
   {
    "model": "grok-4.6",
    "n_trades": 431,
    "wr": 0.4548,
    "pnl_net_usd": -30.94,
    "pnl_gross_usd": 12.16,
    "max_dd_usd": -55.18
   },
   {
    "model": "deepseek-v4-pro",
    "n_trades": 687,
    "wr": 0.4338,
    "pnl_net_usd": -38.0,
    "pnl_gross_usd": 30.7,
    "max_dd_usd": -63.69
   },
   {
    "model": "qwen-3.8-max",
    "n_trades": 711,
    "wr": 0.4163,
    "pnl_net_usd": -49.8,
    "pnl_gross_usd": 21.3,
    "max_dd_usd": -77.4
   },
   {
    "model": "claude-opus-5",
    "n_trades": 534,
    "wr": 0.4064,
    "pnl_net_usd": -56.43,
    "pnl_gross_usd": -3.03,
    "max_dd_usd": -71.19
   },
   {
    "model": "claude-fable-5",
    "n_trades": 623,
    "wr": 0.4157,
    "pnl_net_usd": -60.76,
    "pnl_gross_usd": 1.54,
    "max_dd_usd": -71.65
   },
   {
    "model": "gpt-5.6-sol",
    "n_trades": 651,
    "wr": 0.3763,
    "pnl_net_usd": -80.82,
    "pnl_gross_usd": -15.72,
    "max_dd_usd": -100.2
   },
   {
    "model": "gemini-3.1-pro",
    "n_trades": 694,
    "wr": 0.3833,
    "pnl_net_usd": -91.49,
    "pnl_gross_usd": -22.09,
    "max_dd_usd": -109.15
   }
  ],
  "gap_week_over_week": [
   {
    "model": "grok-4.6",
    "gap_pp_issue8": 13.5,
    "gap_pp_issue9": 7.7,
    "delta_pp": -5.8
   },
   {
    "model": "deepseek-v4-pro",
    "gap_pp_issue8": 17.0,
    "gap_pp_issue9": 9.3,
    "delta_pp": -7.7
   },
   {
    "model": "claude-opus-5",
    "gap_pp_issue8": 14.0,
    "gap_pp_issue9": 10.3,
    "delta_pp": -3.7
   },
   {
    "model": "qwen-3.8-max",
    "gap_pp_issue8": 15.6,
    "gap_pp_issue9": 10.5,
    "delta_pp": -5.1
   },
   {
    "model": "claude-fable-5",
    "gap_pp_issue8": 14.7,
    "gap_pp_issue9": 11.1,
    "delta_pp": -3.6
   },
   {
    "model": "gemini-3.1-pro",
    "gap_pp_issue8": 16.2,
    "gap_pp_issue9": 17.7,
    "delta_pp": 1.5
   },
   {
    "model": "gpt-5.6-sol",
    "gap_pp_issue8": 21.0,
    "gap_pp_issue9": 21.3,
    "delta_pp": 0.3
   },
   {
    "model": "Field",
    "gap_pp_issue8": 16.1,
    "gap_pp_issue9": 12.9,
    "delta_pp": -3.2
   }
  ],
  "ok_rate_by_model": {
   "grok-4.6": 0.9946,
   "deepseek-v4-pro": 1.0,
   "qwen-3.8-max": 0.9966,
   "claude-opus-5": 1.0,
   "claude-fable-5": 1.0,
   "gemini-3.1-pro": 1.0,
   "gpt-5.6-sol": 0.9898
  },
  "counters": {
   "forecasts_total": 10290,
   "ok_in_gate": 9286,
   "out_of_gate": 976,
   "out_of_gate_by_fh": {
    "1w": 490,
    "1m": 486
   },
   "invalid": 28,
   "mature": 9286,
   "mature_directional": 4331,
   "mature_sideways": 4955,
   "pending_next_issue": 0,
   "pending_by_fh": {},
   "late_closes": 0,
   "uptime": [
    {
     "fh": "1h",
     "tf": "1h",
     "slots_seen": 168,
     "slots_expected": 168
    },
    {
     "fh": "4h",
     "tf": "4h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "4h",
     "tf": "1h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "1d",
     "tf": "1d",
     "slots_seen": 7,
     "slots_expected": 7
    },
    {
     "fh": "1d",
     "tf": "4h",
     "slots_seen": 7,
     "slots_expected": 7
    }
   ],
   "regime_days": {
    "2026-09-28": {
     "move_pct": -1.19,
     "regime": "flat"
    },
    "2026-09-29": {
     "move_pct": 0.22,
     "regime": "flat"
    },
    "2026-09-30": {
     "move_pct": -0.22,
     "regime": "flat"
    },
    "2026-10-01": {
     "move_pct": 1.54,
     "regime": "trend"
    },
    "2026-10-02": {
     "move_pct": -0.39,
     "regime": "flat"
    },
    "2026-10-03": {
     "move_pct": 0.4,
     "regime": "flat"
    },
    "2026-10-04": {
     "move_pct": 1.81,
     "regime": "trend"
    }
   },
   "grok_era": {
    "ids_in_window": [
     "grok-4.6"
    ],
    "n_grok45": 0,
    "n_grok46": 1470,
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470,
    "first_grok46_slot": "2026-09-28T00:01:00+00:00",
    "last_grok45_slot": null,
    "first_grok45_slot": null,
    "flip_at": "2026-08-24T09:22:00+00:00",
    "flip_inside_window": false,
    "prev_window": {
     "n_grok45_rows": 0,
     "n_grok46_rows": 1470
    },
    "merged_model_id": "grok-4.6",
    "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "lineage_ids": [
     "grok-4.5",
     "grok-4.6"
    ],
    "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
    "lineage_warnings": [],
    "rows_remapped_a_plus_b": 0,
    "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- five windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #8) and week B (this window) are pure grok-4.6.",
    "cells_with_two_grok_rows": 0
   },
   "series_density": {
    "1d": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-28",
      "2026-09-29",
      "2026-09-30",
      "2026-10-01",
      "2026-10-02",
      "2026-10-03",
      "2026-10-04"
     ],
     "missing_days": []
    },
    "1h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 168,
     "n_rows": 5880,
     "days": [
      "2026-09-28",
      "2026-09-29",
      "2026-09-30",
      "2026-10-01",
      "2026-10-02",
      "2026-10-03",
      "2026-10-04"
     ],
     "missing_days": []
    },
    "1m": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-28",
      "2026-09-29",
      "2026-09-30",
      "2026-10-01",
      "2026-10-02",
      "2026-10-03",
      "2026-10-04"
     ],
     "missing_days": []
    },
    "1w": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-28",
      "2026-09-29",
      "2026-09-30",
      "2026-10-01",
      "2026-10-02",
      "2026-10-03",
      "2026-10-04"
     ],
     "missing_days": []
    },
    "4h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 42,
     "n_rows": 2940,
     "days": [
      "2026-09-28",
      "2026-09-29",
      "2026-09-30",
      "2026-10-01",
      "2026-10-02",
      "2026-10-03",
      "2026-10-04"
     ],
     "missing_days": []
    }
   },
   "model_roster": {
    "current": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "current_count": 7,
    "current_raw_ids_in_window": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "archived_legacy": [
     "grok-4.5",
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "archived_legacy_count": 3,
    "archived_legacy_detail": [
     {
      "model": "grok-4.5",
      "n_rows_since": 11128,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-24T09:01:00+00:00"
     },
     {
      "model": "qwen-3.7-max",
      "n_rows_since": 7479,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-05T09:01:00+00:00"
     },
     {
      "model": "claude-opus-4.8",
      "n_rows_since": 6315,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-07-30T08:01:00+00:00"
     }
    ],
    "archived_excluding_lineage_merged": [
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "tracked_total": 10,
    "canon_expected": {
     "current": 7,
     "archived_legacy": 4,
     "tracked_total": 11
    },
    "matches_canon_11_7_4": false,
    "roster_since": "2026-07-11T00:00:00+00:00",
    "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
   }
  }
 },
 "key_finding": {
  "label": "KEY FINDING · OBSERVATION (one weekly window)",
  "text": "The field's overconfidence gap moved +16.1pp -> +12.9pp week over week, 5 of 7 model lines narrowed, and field Brier went 0.2751 -> 0.2698.",
  "evidence_level": "observation"
 },
 "key_findings": [
  "**OBSERVATION -- The gap read +12.9pp this week.** Directional calls hit 49.6% against 62.5 stated mean confidence (issue #8: +16.1pp on 46.8% vs 63.0). Field Brier is 0.2698 (was 0.2751) and 0 of 7 model lines scored below 0.25, the score of an uninformative always-50% predictor. One window, descriptive.",
  "**5 of 7 models narrowed their gap.** Gap moves ran from deepseek-v4-pro's -7.7pp (+17.0pp -> +9.3pp) to gemini-3.1-pro's +1.5pp (+16.2pp -> +17.7pp); 5 of 7 model lines narrowed week over week. grok-4.6 is both the lowest-Brier line (Brier 0.2566) and the hit-rate leader (52.7%).",
  "**The high-confidence buckets, N>=10 only.** deepseek-v4-pro 70-80 56.2% (n=16), gemini-3.1-pro 70-80 46.4% (n=209), gemini-3.1-pro 80-100 54.5% (n=22), gpt-5.6-sol 70-80 47.2% (n=216). Cells below N=10 are marked insufficient and rank nothing.",
  "**Trading, kept separate from every calibration table.** 0 of 7 lines finished net-positive; net PnL ran -$30.94 (grok-4.6, best) to -$91.49 (gemini-3.1-pro, worst), field -$408.24 against issue #8's -$315.68."
 ],
 "market_check": {
  "label": "MARKET CHECK · OBSERVATION (two adjacent calendar weeks)",
  "text": "OBSERVATION — mean |1d move| 1.83% -> 0.87% (-52% rel), BTC realized vol 35.8% -> 32.2% (ann., hourly); field directional accuracy 46.8% -> 49.6% (+2.8pp), 5 of 7 models improved, sim win-rate up for 5 of 7, field sim PnL -$315.68 -> -$408.24 (adjacent calendar weeks Sep 21-27 vs Sep 28-Oct 4; descriptive, one pair of weeks, not a claim).",
  "robustness": "Robustness: raw price-sign accuracy 49.2% -> 49.0% — the opposite direction to the trade-based hit rule.",
  "weeks": {
   "A": {
    "days": "2026-09-21..2026-09-27",
    "slots": [
     "2026-09-21T00:00:00+00:00",
     "2026-09-28T00:00:00+00:00"
    ],
    "cutoff": "2026-09-28T16:00:00+00:00"
   },
   "B": {
    "days": "2026-09-28..2026-10-04",
    "slots": [
     "2026-09-28T00:00:00+00:00",
     "2026-10-05T00:00:00+00:00"
    ],
    "cutoff": "2026-10-05T16:00:00+00:00"
   }
  },
  "price_source": "crypto_spot/1h@:00",
  "accuracy_field": {
   "A": {
    "n": 4958,
    "hit_rate": 0.4683,
    "wilson_95": [
     0.4545,
     0.4822
    ]
   },
   "B": {
    "n": 4331,
    "hit_rate": 0.4964,
    "wilson_95": [
     0.4815,
     0.5113
    ]
   },
   "delta_pp": 2.8
  },
  "accuracy_per_model": [
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "A": {
     "n": 777,
     "hit_rate": 0.4389,
     "wilson_95": [
      0.4044,
      0.474
     ]
    },
    "B": {
     "n": 687,
     "hit_rate": 0.5095,
     "wilson_95": [
      0.4721,
      0.5467
     ]
    },
    "delta_pp": 7.1
   },
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "A": {
     "n": 537,
     "hit_rate": 0.4693,
     "wilson_95": [
      0.4274,
      0.5116
     ]
    },
    "B": {
     "n": 431,
     "hit_rate": 0.5267,
     "wilson_95": [
      0.4795,
      0.5734
     ]
    },
    "delta_pp": 5.7
   },
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "A": {
     "n": 750,
     "hit_rate": 0.4453,
     "wilson_95": [
      0.4101,
      0.4811
     ]
    },
    "B": {
     "n": 711,
     "hit_rate": 0.4895,
     "wilson_95": [
      0.4529,
      0.5262
     ]
    },
    "delta_pp": 4.4
   },
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "A": {
     "n": 736,
     "hit_rate": 0.4674,
     "wilson_95": [
      0.4316,
      0.5035
     ]
    },
    "B": {
     "n": 623,
     "hit_rate": 0.4992,
     "wilson_95": [
      0.4601,
      0.5383
     ]
    },
    "delta_pp": 3.2
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "A": {
     "n": 620,
     "hit_rate": 0.4758,
     "wilson_95": [
      0.4368,
      0.5151
     ]
    },
    "B": {
     "n": 534,
     "hit_rate": 0.5075,
     "wilson_95": [
      0.4652,
      0.5497
     ]
    },
    "delta_pp": 3.2
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "A": {
     "n": 744,
     "hit_rate": 0.4677,
     "wilson_95": [
      0.4321,
      0.5037
     ]
    },
    "B": {
     "n": 651,
     "hit_rate": 0.4654,
     "wilson_95": [
      0.4274,
      0.5038
     ]
    },
    "delta_pp": -0.2
   },
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "A": {
     "n": 794,
     "hit_rate": 0.5139,
     "wilson_95": [
      0.4791,
      0.5485
     ]
    },
    "B": {
     "n": 694,
     "hit_rate": 0.4899,
     "wilson_95": [
      0.4529,
      0.5271
     ]
    },
    "delta_pp": -2.4
   }
  ],
  "raw_sign_field": {
   "A": {
    "n_scored": 4954,
    "hit_rate": 0.4923
   },
   "B": {
    "n_scored": 4311,
    "hit_rate": 0.4901
   }
  },
  "abs_move_1d": {
   "A": 1.835,
   "B": 0.873,
   "delta_pct_rel": -52.4
  },
  "btc_realized_vol_ann_pct_hourly": {
   "A": 35.8,
   "B": 32.24
  },
  "trading_field": {
   "A": {
    "n_trades": 4958,
    "wr": 0.3998,
    "pnl_net_usd": -315.68,
    "pnl_gross_usd": 180.12
   },
   "B": {
    "n_trades": 4331,
    "wr": 0.4103,
    "pnl_net_usd": -408.24,
    "pnl_gross_usd": 24.86
   }
  },
  "verdict": {
   "models_with_hit_improved": "5 of 7",
   "models_with_rawsign_improved": "3 of 7",
   "models_with_wr_improved": "5 of 7",
   "field_hit_delta_pp": 2.8,
   "field_pnl_net": {
    "A": -315.68,
    "B": -408.24
   },
   "abs_move_delta": {
    "1h": {
     "A": 0.358,
     "B": 0.314,
     "delta_pct_rel": -12.3
    },
    "4h": {
     "A": 0.766,
     "B": 0.706,
     "delta_pct_rel": -7.8
    },
    "1d": {
     "A": 1.835,
     "B": 0.873,
     "delta_pct_rel": -52.4
    }
   },
   "btc_vol_ann_pct": {
    "A": 35.8,
    "B": 32.24
   },
   "note": "descriptive, two adjacent calendar weeks; hit rule = methodology v1.1 trade-based; raw_sign = price-sign robustness check (closes at :00 vs slots at :01)"
  },
  "lineage_note": "lineage splice ACTIVE: grok-4.5 rows are aggregated into grok-4.6 (flip 2026-08-24T09:22:00+00:00, five windows back, inside issue #4's window). Week A (2026-09-21..2026-09-27) is pure grok-4.6 (it reproduces the published issue #8); week B is pure grok-4.6."
 },
 "why_it_matters": "Every MarketMania forecast carries a model-stated confidence from 0 to 100 alongside its direction call. This report checks whether that number tracks reality: on a well-calibrated forecaster, calls made at 70% confidence should hit their direction about 70% of the time. With issue #9 the series has nine points on every model's gap -- enough to see movement, not enough to claim a trend: calibration and market regime move together, and durability claims belong in the Monthly series, not here.",
 "how_to_read_this": "Gap pp = mean stated confidence minus hit-rate, in percentage points; positive means overconfident. Brier is the mean squared error of the stated probability against the realised outcome, so lower is better and 0.25 is what an uninformative always-50% predictor scores. Coverage = calls scored divided by the mature calls available to that model. Confidence buckets are the models' own natural breakpoints, not an arbitrary binning; cells below N=10 are marked insufficient and never used to rank anything. Prediction and trading metrics live in separate sections and are never combined. The grok 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits five windows back (issue #4): both weeks of this issue are pure grok-4.6, with no 4.5-era rows in either window, so the grok week-over-week row is the fourth pure-4.6 against pure-4.6 comparison of the series.",
 "practical_implications": [
  "Stated confidence is still a ranking hint at best, not a probability: the field's gap is +12.9pp this week against +16.1pp in issue #8.",
  "Gap and regime move together. The field hit 49.6% against 46.8% in issue #8 and stated mean confidence read 62.5 against 63.0, so the gap came out at +12.9pp: a field that hits 49.6% instead of 46.8% closes or opens a confidence gap without a single stated number changing.",
  "Trading direction this week (0 of 7 lines net-positive) is a regime read, not a strategy result; the same lines printed the opposite sign inside seven weeks."
 ],
 "limitations": [
  "Prediction metrics (this report) and trading metrics are kept in separate sections per methodology -- they are never combined into a single score.",
  "95% Wilson CIs shown are descriptive, not inferential: observations inside one window are dependent, so read them as a range, not a formal coverage guarantee.",
  "All 4,331 scored calls sit inside one market regime. This window ran 2 trend days of 7 (issue #8: 2 of 7); every comparison with issue #8 is a comparison across regimes as well as across weeks. Nine week-over-week points cannot separate drift from regime; no durability claim is made.",
  "Models report confidence at discrete levels, not a continuous scale; the buckets reflect those natural breakpoints. Cells below N=10 are marked insufficient.",
  "Underlying price series are reconstructed from trade entry prices (median per symbol-slot), not an independent tick feed. Series density and lineage notes (wave 9): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 28-Oct 4) daily coverage is FULL, as it was in issues #4, #5, #6, #7 and #8: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 490, 1M 486). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, five windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#8 window), so every grok number in this issue and in issue #8 is a pure grok-4.6 line -- the grok week-over-week row is the fourth pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #8 after the audit that found the raw table mixes two exchanges; issue #8 values are as published.",
  "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Oct 5 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6, #7 and #8, which printed 57,929 at the Sep 28 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 62,260 is an additive step under one definition; counter deltas against issue #7 and earlier remain definitional."
 ],
 "notes": {
  "series_density_and_lineage_wave9": "Series density and lineage notes (wave 9): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 28-Oct 4) daily coverage is FULL, as it was in issues #4, #5, #6, #7 and #8: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 490, 1M 486). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, five windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#8 window), so every grok number in this issue and in issue #8 is a pure grok-4.6 line -- the grok week-over-week row is the fourth pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "grok_era": {
   "ids_in_window": [
    "grok-4.6"
   ],
   "n_grok45": 0,
   "n_grok46": 1470,
   "n_grok45_rows": 0,
   "n_grok46_rows": 1470,
   "first_grok46_slot": "2026-09-28T00:01:00+00:00",
   "last_grok45_slot": null,
   "first_grok45_slot": null,
   "flip_at": "2026-08-24T09:22:00+00:00",
   "flip_inside_window": false,
   "prev_window": {
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470
   },
   "merged_model_id": "grok-4.6",
   "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
   "lineage_ids": [
    "grok-4.5",
    "grok-4.6"
   ],
   "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
   "lineage_warnings": [],
   "rows_remapped_a_plus_b": 0,
   "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- five windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #8) and week B (this window) are pure grok-4.6.",
   "cells_with_two_grok_rows": 0
  },
  "model_ids_line": [
   "claude-fable-5",
   "claude-opus-5",
   "deepseek-v4-pro",
   "gemini-3.1-pro",
   "gpt-5.6-sol",
   "grok-4.6",
   "qwen-3.8-max"
  ],
  "model_roster": {
   "current": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "current_count": 7,
   "current_raw_ids_in_window": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "archived_legacy": [
    "grok-4.5",
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "archived_legacy_count": 3,
   "archived_legacy_detail": [
    {
     "model": "grok-4.5",
     "n_rows_since": 11128,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-24T09:01:00+00:00"
    },
    {
     "model": "qwen-3.7-max",
     "n_rows_since": 7479,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-05T09:01:00+00:00"
    },
    {
     "model": "claude-opus-4.8",
     "n_rows_since": 6315,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-07-30T08:01:00+00:00"
    }
   ],
   "archived_excluding_lineage_merged": [
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "tracked_total": 10,
   "canon_expected": {
    "current": 7,
    "archived_legacy": 4,
    "tracked_total": 11
   },
   "matches_canon_11_7_4": false,
   "roster_since": "2026-07-11T00:00:00+00:00",
   "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
  },
  "series_density": {
   "1d": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-28",
     "2026-09-29",
     "2026-09-30",
     "2026-10-01",
     "2026-10-02",
     "2026-10-03",
     "2026-10-04"
    ],
    "missing_days": []
   },
   "1h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 168,
    "n_rows": 5880,
    "days": [
     "2026-09-28",
     "2026-09-29",
     "2026-09-30",
     "2026-10-01",
     "2026-10-02",
     "2026-10-03",
     "2026-10-04"
    ],
    "missing_days": []
   },
   "1m": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-28",
     "2026-09-29",
     "2026-09-30",
     "2026-10-01",
     "2026-10-02",
     "2026-10-03",
     "2026-10-04"
    ],
    "missing_days": []
   },
   "1w": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-28",
     "2026-09-29",
     "2026-09-30",
     "2026-10-01",
     "2026-10-02",
     "2026-10-03",
     "2026-10-04"
    ],
    "missing_days": []
   },
   "4h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 42,
    "n_rows": 2940,
    "days": [
     "2026-09-28",
     "2026-09-29",
     "2026-09-30",
     "2026-10-01",
     "2026-10-02",
     "2026-10-03",
     "2026-10-04"
    ],
    "missing_days": []
   }
  },
  "week_over_week_note": "Week-over-week columns compare back-to-back windows: Sep 21-27 (issue #8, and week A of this issue's alive slice) vs Sep 28-Oct 4 (this issue). \"Was\" values are the numbers published in issue #8; deltas are computed on unrounded rates.",
  "research_to_date_pin": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Oct 5 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6, #7 and #8, which printed 57,929 at the Sep 28 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 62,260 is an additive step under one definition; counter deltas against issue #7 and earlier remain definitional.",
  "market_state_audit": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #8 after the audit that found the raw table mixes two exchanges; issue #8 values are as published.",
  "engine": "Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
  "report_template": "PDF built with the canonical mm_report.py template module; preview card with report_preview.render_preview"
 },
 "platform_note_tpsl": "**Context, not a finding.** Since Aug 13 the public sandbox can trade with model-specific TP/SL multipliers learned from this same weekly history (v1); since Aug 19, v2 adds per-ticker and confidence-bucket (50-70 / 70-100) resolution. That feature consumes calibration history; it does not feed back into any table in this report, which measures stated confidence vs. direction-hit only. No effect size is claimed for v2. The toggle lives at marketmania.ai/indices.",
 "living_series_note": "**Issue #9.** Weekly Calibration is a living series. Issue #8's open questions -- \"the gap moved +4.0pp while the field base moved -3.9pp -- the third issue running in which the two move nearly one-for-one in opposite directions; does that hold for a fourth?\" and \"The best-Brier line changed hands again, to qwen-3.8-max -- does it hold for a second issue, or change hands again?\" -- read this issue as: field gap +16.1pp -> +12.9pp (5 of 7 lines narrowed) on a field base that moved +2.8pp, a fourth issue running in opposite directions, and the best-Brier line changed hands again: grok-4.6 at 0.2566 (issue #8: qwen-3.8-max at 0.2689). Also on the record: gpt-5.6-sol's 70-80 bucket read 47.2% (n=216) against its own 60-70 at 46.2% (n=429). The grok row is pure grok-4.6 this issue and was pure grok-4.6 in issue #8; the 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits five windows back. Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
 "testing_next": [
  "Next issue: the gap moved -3.2pp while the field base moved +2.8pp -- the fourth issue running in which the two move nearly one-for-one in opposite directions; does that hold for a fifth?",
  "The best-Brier line changed hands again, to grok-4.6 -- does it hold for a second issue, or change hands again?",
  "Monthly series: is the overconfidence gap stable across market regimes (trend vs flat) at monthly n?"
 ],
 "related_research": [
  {
   "title": "Consensus Watch #9",
   "url": "https://marketmania.ai/research/reports/consensus-watch-2026-09-28.pdf"
  },
  {
   "title": "Weekly Model Watch #9",
   "url": "https://marketmania.ai/research/reports/model-watch-2026-09-28.pdf"
  },
  {
   "title": "Weekly Calibration #8",
   "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.pdf"
  }
 ],
 "related_research_note": "The three weekly reports publish together as one issue each week; each links straight to the others' PDF and to its own previous issue. Direct links are the posting rule from wave 2 on.",
 "citation": {
  "bibtex_key": "mm_calibration_2026w40",
  "title": "Weekly Calibration #9: confidence vs. direction-hit, Sep 28-Oct 4 2026",
  "author": "MarketMania Research",
  "year": 2026,
  "month": "October",
  "day": 7,
  "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-28.pdf",
  "note": "Methodology v1.1; window Sep 28-Oct 4, 2026 UTC; source weekly_metrics_2026-09-28.json"
 },
 "research_to_date": {
  "this_report": {
   "scored_observations": 4331
  },
  "platform": {
   "as_of_cutoff": "2026-10-05",
   "resolved_forecasts": 62260,
   "resolved_forecasts_definition": "directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, at the issue cutoff",
   "resolved_forecasts_since": "2026-07-11",
   "models_tracked": 10,
   "models_tracked_detail": "7 models tracked (current line-up; earlier versions folded into their successors' lineage)",
   "assets": 5,
   "forecast_horizons": 5,
   "cadence": "hourly",
   "published_research_reports": 44
  },
  "source": "weekly_metrics_2026-09-28.json",
  "control_reproduces_wave3_pin": true,
  "pin_note": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Oct 5 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6, #7 and #8, which printed 57,929 at the Sep 28 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 62,260 is an additive step under one definition; counter deltas against issue #7 and earlier remain definitional."
 },
 "site_entry": {
  "slug": "weekly-calibration-2026-09-28",
  "series": "weekly",
  "seriesLabel": "Weekly · Calibration",
  "date": "October 7, 2026",
  "title": "Weekly Calibration #9",
  "summary": "4,331 directional calls asked whether stated confidence tracks reality. The field hit 49.6% against 62.5 stated mean confidence, an overconfidence gap of +12.9pp against +16.1pp in issue #8, and 5 of 7 model lines narrowed week over week. Field Brier moved 0.2751 -> 0.2698; grok-4.6 is the lowest-Brier line at 0.2566 and grok-4.6 leads on hit-rate at 52.7%. Trading, kept in its own section, finished with 0 of 7 lines net-positive as field accuracy moved 46.8% -> 49.6% (+2.8pp).",
  "stats": [
   "4,331 directional calls",
   "gap +12.9pp (was +16.1pp)",
   "0 of 7 net-positive"
  ],
  "manifest_key_finding": "The overconfidence gap read +12.9pp against +16.1pp in issue #8, with 5 of 7 model lines narrowing and field Brier moving 0.2751 -> 0.2698; trading, kept separate, finished 0 of 7 lines net-positive.",
  "files": {
   "pdf": "/research/reports/weekly-calibration-2026-09-28.pdf",
   "md": "/research/reports/weekly-calibration-2026-09-28.md",
   "json": "/research/reports/weekly-calibration-2026-09-28.json"
  }
 },
 "social": {
  "telegram": {
   "photo_caption_html": "🎯 <b>Weekly Calibration #9</b> — weekly series (window Sep 28–Oct 4)\n\nResearch question: when a model says 70, does it hit 70% of the time?\n\n4,331 directional calls across 7 model lines:\n• Field hit <b>49.6%</b> vs <b>62.5</b> stated — an overconfidence gap of <b>+12.9pp</b> (issue #8: +16.1pp)\n• Gap week over week: <b>5 of 7 lines narrowed</b>; field Brier 0.2751 -> 0.2698\n• Lowest-Brier line: <b>grok-4.6</b> (Brier 0.2566); hit-rate leader grok-4.6 at 52.7%\n• Market check: field accuracy 46.8% -> 49.6% (+2.8pp); trading kept separate — <b>0 of 7 net-positive</b>\n\nNote: grok is a pure 4.6 line in both weeks of this issue; the 4.5 -> 4.6 flip (Aug 24 09:22 UTC) sits five windows back\n\n📄 <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-28.pdf\">Full report (PDF)</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "document_caption_html": "📄 Weekly Calibration #9 — full report (PDF).\nWeb copy: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-28.pdf\">weekly-calibration-2026-09-28.pdf</a> · machine-readable: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-28.json\">JSON</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "caption_len": 927,
   "document_caption_len": 416,
   "limit": 1024
  },
  "x": {
   "main": "Weekly Calibration #9 is out.\n\n4,331 LLM calls: 49.6% hit vs 62.5 stated confidence — an overconfidence gap of +12.9pp against +16.1pp a week earlier, with 5 of 7 lines narrowing.\n\nFull report: https://marketmania.ai/research/reports/weekly-calibration-2026-09-28.pdf\n\n#AI #Crypto #Research #Calibration",
   "main_tco_len": 253,
   "reply": "Overconfidence gap, w/w (pp):\ngrok-4.6: 13.5 -> 7.7\ndeepseek: 17.0 -> 9.3\nopus-5: 14.0 -> 10.3\nqwen-3.8: 15.6 -> 10.5\nfable-5: 14.7 -> 11.1\ngemini-3.1: 16.2 -> 17.7\ngpt-5.6: 21.0 -> 21.3\nField: +16.1pp -> +12.9pp. Methodology v1.1.\nData: https://marketmania.ai/research/reports/weekly-calibration-2026-09-28.json",
   "reply_tco_len": 261,
   "limit_tco": 280,
   "replies_max": 1
  }
 }
}
