{
 "slug": "weekly-calibration-2026-09-21",
 "series": "weekly",
 "series_label": "WEEKLY · CALIBRATION",
 "issue": 8,
 "title": "Weekly Calibration #8",
 "language": "en",
 "window_start": "2026-09-21T00:00:00+00:00",
 "window_end": "2026-09-28T00:00:00+00:00",
 "window_text": "Sep 21-27, 2026 UTC",
 "cutoff": "2026-09-28T16:00:00+00:00",
 "generated_at": "2026-09-30T05:45:53+00:00",
 "methodology": "v1.1 (2026-08-10)",
 "methodology_hash": "e66c7e8c864a2233",
 "source": "weekly_metrics_2026-09-21.json",
 "hit_rule": "direction: exit tp1/tp2 -> hit, sl -> miss, expiry -> sign of gross pnl",
 "bibtex_key": "mm_calibration_2026w39",
 "market_state": {
  "panel": "market_state_6p2",
  "week": {
   "start": "2026-09-21",
   "end": "2026-09-28"
  },
  "prev_week": {
   "start": "2026-09-14",
   "end": "2026-09-21"
  },
  "current": {
   "btc_net_pct": 4.06,
   "realized_vol_ann_pct": 52.25,
   "volume_top5_usd_bn": 22.4265,
   "volume_top5_wow_pct": 19.7,
   "avg_pairwise_corr": 0.71
  },
  "previous_as_published_issue7": {
   "btc_net_pct": 5.64,
   "realized_vol_ann_pct": 51.01,
   "volume_top5_usd_bn": 18.7386,
   "avg_pairwise_corr": 0.93
  },
  "snapshot_line": "BTC net +4.06% (prior +5.64%) - ann vol 52.2% (was 51.0%) - TOP5 volume $22.43B, +20% w/w - pairwise corr 0.71 (was 0.93)",
  "exchange": "binance",
  "audit_filter_applied": true,
  "source": "single-exchange (binance) daily candle recompute, market_state_2026-09-21.json; prior-week values as published in issue #7",
  "audit_note": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #7 after the audit that found the raw table mixes two exchanges; issue #7 values are as published.",
  "prev_week_control": {
   "published_issue3": {
    "btc_net_pct": 5.64,
    "realized_vol_ann_pct": 51.01,
    "volume_top5_usd_bn": 18.7386,
    "avg_pairwise_corr": 0.93
   },
   "published_issue7": {
    "btc_net_pct": 5.64,
    "realized_vol_ann_pct": 51.01,
    "volume_top5_usd_bn": 18.7386,
    "avg_pairwise_corr": 0.93
   },
   "got": {
    "btc_net_pct": 5.64,
    "realized_vol_ann_pct": 51.01,
    "volume_top5_usd_bn": 18.7386,
    "avg_pairwise_corr": 0.93
   },
   "match": {
    "btc_net_pct": true,
    "realized_vol_ann_pct": true,
    "volume_top5_usd_bn": true,
    "avg_pairwise_corr": true
   },
   "tolerance": {
    "btc_net_pct": 0.05,
    "realized_vol_ann_pct": 0.5,
    "volume_top5_usd_bn": 0.05,
    "avg_pairwise_corr": 0.02
   },
   "all_match": true,
   "note": "previous week of this run == published issue-#7 week; the four Snapshot pins must reproduce (key published_issue3 kept for the builder schema, published_issue7 is the add-only alias)"
  },
  "regime_days": {
   "2026-09-21": {
    "move_pct": 6.39,
    "regime": "trend"
   },
   "2026-09-22": {
    "move_pct": -0.4,
    "regime": "flat"
   },
   "2026-09-23": {
    "move_pct": -2.04,
    "regime": "trend"
   },
   "2026-09-24": {
    "move_pct": 0.02,
    "regime": "flat"
   },
   "2026-09-25": {
    "move_pct": -0.46,
    "regime": "flat"
   },
   "2026-09-26": {
    "move_pct": 0.36,
    "regime": "flat"
   },
   "2026-09-27": {
    "move_pct": 0.2,
    "regime": "flat"
   }
  },
  "alive_slice_context": {
   "abs_move_1d_pct": {
    "A": 2.693,
    "B": 1.835
   },
   "btc_realized_vol_ann_pct_hourly": {
    "A": 35.07,
    "B": 35.8
   }
  }
 },
 "tables": {
  "calibration_main": [
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "legacy": false,
    "is_field": false,
    "n": 750,
    "coverage": 0.5682,
    "hit_rate": 0.4453,
    "mean_conf": 60.1,
    "gap_pp": 15.6,
    "brier": 0.2689,
    "wilson_95": {
     "lo": 0.4101,
     "hi": 0.4811
    }
   },
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "legacy": false,
    "is_field": false,
    "n": 736,
    "coverage": 0.5534,
    "hit_rate": 0.4674,
    "mean_conf": 61.4,
    "gap_pp": 14.7,
    "brier": 0.2692,
    "wilson_95": {
     "lo": 0.4316,
     "hi": 0.5035
    }
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "legacy": false,
    "is_field": false,
    "n": 620,
    "coverage": 0.4662,
    "hit_rate": 0.4758,
    "mean_conf": 61.5,
    "gap_pp": 14.0,
    "brier": 0.2693,
    "wilson_95": {
     "lo": 0.4368,
     "hi": 0.5151
    }
   },
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "legacy": false,
    "is_field": false,
    "n": 537,
    "coverage": 0.4059,
    "hit_rate": 0.4693,
    "mean_conf": 60.4,
    "gap_pp": 13.5,
    "brier": 0.2704,
    "wilson_95": {
     "lo": 0.4274,
     "hi": 0.5116
    }
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "legacy": false,
    "is_field": false,
    "n": 777,
    "coverage": 0.5931,
    "hit_rate": 0.4389,
    "mean_conf": 60.9,
    "gap_pp": 17.0,
    "brier": 0.274,
    "wilson_95": {
     "lo": 0.4044,
     "hi": 0.474
    }
   },
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "legacy": false,
    "is_field": false,
    "n": 794,
    "coverage": 0.597,
    "hit_rate": 0.5139,
    "mean_conf": 67.6,
    "gap_pp": 16.2,
    "brier": 0.2764,
    "wilson_95": {
     "lo": 0.4791,
     "hi": 0.5485
    }
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "legacy": false,
    "is_field": false,
    "n": 744,
    "coverage": 0.5594,
    "hit_rate": 0.4677,
    "mean_conf": 67.7,
    "gap_pp": 21.0,
    "brier": 0.2951,
    "wilson_95": {
     "lo": 0.4321,
     "hi": 0.5037
    }
   },
   {
    "model": "Field (all models)",
    "legacy": false,
    "is_field": true,
    "n": 4958,
    "coverage": null,
    "hit_rate": 0.4683,
    "mean_conf": 63.0,
    "gap_pp": 16.1,
    "brier": 0.2751,
    "wilson_95": {
     "lo": null,
     "hi": null
    }
   }
  ],
  "buckets": [
   {
    "model": "qwen-3.8-max",
    "bucket": "50-60",
    "n": 299,
    "hit_rate": 0.388,
    "mean_conf": 56.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "60-70",
    "n": 437,
    "hit_rate": 0.492,
    "mean_conf": 62.5,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "70-80",
    "n": 5,
    "hit_rate": 0.2,
    "mean_conf": 70.8,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-fable-5",
    "bucket": "50-60",
    "n": 199,
    "hit_rate": 0.3769,
    "mean_conf": 56.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "60-70",
    "n": 536,
    "hit_rate": 0.5019,
    "mean_conf": 63.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "70-80",
    "n": 1,
    "hit_rate": 0.0,
    "mean_conf": 70,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "50-60",
    "n": 85,
    "hit_rate": 0.4588,
    "mean_conf": 57.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "60-70",
    "n": 535,
    "hit_rate": 0.4785,
    "mean_conf": 62.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "grok-4.6",
    "bucket": "50-60",
    "n": 182,
    "hit_rate": 0.544,
    "mean_conf": 57.3,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "60-70",
    "n": 355,
    "hit_rate": 0.431,
    "mean_conf": 62.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "grok-4.6",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "50-60",
    "n": 264,
    "hit_rate": 0.4129,
    "mean_conf": 56.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "60-70",
    "n": 483,
    "hit_rate": 0.4472,
    "mean_conf": 62.6,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "70-80",
    "n": 29,
    "hit_rate": 0.5517,
    "mean_conf": 71.4,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "50-60",
    "n": 3,
    "hit_rate": 0.3333,
    "mean_conf": 55,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "60-70",
    "n": 456,
    "hit_rate": 0.5,
    "mean_conf": 63.4,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "70-80",
    "n": 320,
    "hit_rate": 0.5344,
    "mean_conf": 73.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "80-100",
    "n": 15,
    "hit_rate": 0.5333,
    "mean_conf": 80,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "50-60",
    "n": 5,
    "hit_rate": 1.0,
    "mean_conf": 58.2,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "60-70",
    "n": 517,
    "hit_rate": 0.4565,
    "mean_conf": 65.5,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "70-80",
    "n": 220,
    "hit_rate": 0.4864,
    "mean_conf": 73.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "80-100",
    "n": 2,
    "hit_rate": 0.0,
    "mean_conf": 80,
    "insufficient": true,
    "no_data": false
   }
  ],
  "sub50": [
   {
    "model": "qwen-3.8-max",
    "n": 9,
    "hit_rate": 0.2222
   },
   {
    "model": "claude-fable-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "claude-opus-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "grok-4.6",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "deepseek-v4-pro",
    "n": 1,
    "hit_rate": 0.0
   },
   {
    "model": "gemini-3.1-pro",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gpt-5.6-sol",
    "n": 0,
    "hit_rate": null
   }
  ],
  "by_fh": [
   {
    "model": "qwen-3.8-max",
    "fh": "1h",
    "n": 463,
    "hit_rate": 0.4514,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "4h",
    "n": 234,
    "hit_rate": 0.4274,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1d",
    "n": 53,
    "hit_rate": 0.4717,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "1h",
    "n": 464,
    "hit_rate": 0.4806,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "4h",
    "n": 220,
    "hit_rate": 0.4409,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "1d",
    "n": 52,
    "hit_rate": 0.4615,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1h",
    "n": 378,
    "hit_rate": 0.4921,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "4h",
    "n": 201,
    "hit_rate": 0.4478,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1d",
    "n": 41,
    "hit_rate": 0.4634,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "1h",
    "n": 323,
    "hit_rate": 0.4892,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "4h",
    "n": 170,
    "hit_rate": 0.4353,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "1d",
    "n": 44,
    "hit_rate": 0.4545,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1h",
    "n": 470,
    "hit_rate": 0.4532,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "4h",
    "n": 252,
    "hit_rate": 0.4008,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1d",
    "n": 55,
    "hit_rate": 0.4909,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1h",
    "n": 500,
    "hit_rate": 0.516,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "4h",
    "n": 237,
    "hit_rate": 0.5232,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1d",
    "n": 57,
    "hit_rate": 0.4561,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1h",
    "n": 448,
    "hit_rate": 0.471,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "4h",
    "n": 241,
    "hit_rate": 0.4647,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1d",
    "n": 55,
    "hit_rate": 0.4545,
    "flag": null
   }
  ],
  "trading": [
   {
    "model": "gemini-3.1-pro",
    "n_trades": 794,
    "wr": 0.4295,
    "pnl_net_usd": -5.94,
    "pnl_gross_usd": 73.46,
    "max_dd_usd": -99.6
   },
   {
    "model": "grok-4.6",
    "n_trades": 537,
    "wr": 0.3985,
    "pnl_net_usd": -29.75,
    "pnl_gross_usd": 23.95,
    "max_dd_usd": -87.02
   },
   {
    "model": "claude-opus-5",
    "n_trades": 620,
    "wr": 0.4016,
    "pnl_net_usd": -43.83,
    "pnl_gross_usd": 18.17,
    "max_dd_usd": -121.44
   },
   {
    "model": "claude-fable-5",
    "n_trades": 736,
    "wr": 0.394,
    "pnl_net_usd": -45.6,
    "pnl_gross_usd": 28.0,
    "max_dd_usd": -127.29
   },
   {
    "model": "gpt-5.6-sol",
    "n_trades": 744,
    "wr": 0.4046,
    "pnl_net_usd": -48.68,
    "pnl_gross_usd": 25.72,
    "max_dd_usd": -129.1
   },
   {
    "model": "qwen-3.8-max",
    "n_trades": 750,
    "wr": 0.3827,
    "pnl_net_usd": -55.28,
    "pnl_gross_usd": 19.72,
    "max_dd_usd": -125.77
   },
   {
    "model": "deepseek-v4-pro",
    "n_trades": 777,
    "wr": 0.3861,
    "pnl_net_usd": -86.6,
    "pnl_gross_usd": -8.9,
    "max_dd_usd": -154.88
   }
  ],
  "gap_week_over_week": [
   {
    "model": "grok-4.6",
    "gap_pp_issue7": 10.7,
    "gap_pp_issue8": 13.5,
    "delta_pp": 2.8
   },
   {
    "model": "claude-opus-5",
    "gap_pp_issue7": 9.9,
    "gap_pp_issue8": 14.0,
    "delta_pp": 4.1
   },
   {
    "model": "claude-fable-5",
    "gap_pp_issue7": 9.1,
    "gap_pp_issue8": 14.7,
    "delta_pp": 5.6
   },
   {
    "model": "qwen-3.8-max",
    "gap_pp_issue7": 9.2,
    "gap_pp_issue8": 15.6,
    "delta_pp": 6.4
   },
   {
    "model": "gemini-3.1-pro",
    "gap_pp_issue7": 14.0,
    "gap_pp_issue8": 16.2,
    "delta_pp": 2.2
   },
   {
    "model": "deepseek-v4-pro",
    "gap_pp_issue7": 12.4,
    "gap_pp_issue8": 17.0,
    "delta_pp": 4.6
   },
   {
    "model": "gpt-5.6-sol",
    "gap_pp_issue7": 18.4,
    "gap_pp_issue8": 21.0,
    "delta_pp": 2.6
   },
   {
    "model": "Field",
    "gap_pp_issue7": 12.1,
    "gap_pp_issue8": 16.1,
    "delta_pp": 4.0
   }
  ],
  "ok_rate_by_model": {
   "qwen-3.8-max": 0.9891,
   "claude-fable-5": 0.9993,
   "claude-opus-5": 1.0,
   "grok-4.6": 0.9952,
   "deepseek-v4-pro": 0.9864,
   "gemini-3.1-pro": 1.0,
   "gpt-5.6-sol": 1.0
  },
  "counters": {
   "forecasts_total": 10290,
   "ok_in_gate": 9273,
   "out_of_gate": 973,
   "out_of_gate_by_fh": {
    "1w": 488,
    "1m": 485
   },
   "invalid": 44,
   "mature": 9273,
   "mature_directional": 4958,
   "mature_sideways": 4315,
   "pending_next_issue": 0,
   "pending_by_fh": {},
   "late_closes": 0,
   "uptime": [
    {
     "fh": "1h",
     "tf": "1h",
     "slots_seen": 168,
     "slots_expected": 168
    },
    {
     "fh": "4h",
     "tf": "4h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "4h",
     "tf": "1h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "1d",
     "tf": "1d",
     "slots_seen": 7,
     "slots_expected": 7
    },
    {
     "fh": "1d",
     "tf": "4h",
     "slots_seen": 7,
     "slots_expected": 7
    }
   ],
   "regime_days": {
    "2026-09-21": {
     "move_pct": 6.39,
     "regime": "trend"
    },
    "2026-09-22": {
     "move_pct": -0.4,
     "regime": "flat"
    },
    "2026-09-23": {
     "move_pct": -2.04,
     "regime": "trend"
    },
    "2026-09-24": {
     "move_pct": 0.02,
     "regime": "flat"
    },
    "2026-09-25": {
     "move_pct": -0.46,
     "regime": "flat"
    },
    "2026-09-26": {
     "move_pct": 0.36,
     "regime": "flat"
    },
    "2026-09-27": {
     "move_pct": 0.2,
     "regime": "flat"
    }
   },
   "grok_era": {
    "ids_in_window": [
     "grok-4.6"
    ],
    "n_grok45": 0,
    "n_grok46": 1470,
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470,
    "first_grok46_slot": "2026-09-21T00:01:00+00:00",
    "last_grok45_slot": null,
    "first_grok45_slot": null,
    "flip_at": "2026-08-24T09:22:00+00:00",
    "flip_inside_window": false,
    "prev_window": {
     "n_grok45_rows": 0,
     "n_grok46_rows": 1470
    },
    "merged_model_id": "grok-4.6",
    "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "lineage_ids": [
     "grok-4.5",
     "grok-4.6"
    ],
    "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
    "lineage_warnings": [],
    "rows_remapped_a_plus_b": 0,
    "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- four windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #7) and week B (this window) are pure grok-4.6.",
    "cells_with_two_grok_rows": 0
   },
   "series_density": {
    "1d": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "1h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 168,
     "n_rows": 5880,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "1m": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "1w": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "4h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 42,
     "n_rows": 2940,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    }
   },
   "model_roster": {
    "current": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "current_count": 7,
    "current_raw_ids_in_window": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "archived_legacy": [
     "grok-4.5",
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "archived_legacy_count": 3,
    "archived_legacy_detail": [
     {
      "model": "grok-4.5",
      "n_rows_since": 11128,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-24T09:01:00+00:00"
     },
     {
      "model": "qwen-3.7-max",
      "n_rows_since": 7479,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-05T09:01:00+00:00"
     },
     {
      "model": "claude-opus-4.8",
      "n_rows_since": 6315,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-07-30T08:01:00+00:00"
     }
    ],
    "archived_excluding_lineage_merged": [
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "tracked_total": 10,
    "canon_expected": {
     "current": 7,
     "archived_legacy": 4,
     "tracked_total": 11
    },
    "matches_canon_11_7_4": false,
    "roster_since": "2026-07-11T00:00:00+00:00",
    "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
   }
  }
 },
 "key_finding": {
  "label": "KEY FINDING · OBSERVATION (one weekly window)",
  "text": "The field's overconfidence gap moved +12.1pp -> +16.1pp week over week, 0 of 7 model lines narrowed, and field Brier went 0.2669 -> 0.2751.",
  "evidence_level": "observation"
 },
 "key_findings": [
  "**OBSERVATION -- The gap read +16.1pp this week.** Directional calls hit 46.8% against 63.0 stated mean confidence (issue #7: +12.1pp on 50.7% vs 62.8). Field Brier is 0.2751 (was 0.2669) and 0 of 7 model lines scored below 0.25, the score of an uninformative always-50% predictor. One window, descriptive.",
  "**No model narrowed its gap.** Gap moves ran from gemini-3.1-pro's +2.2pp (+14.0pp -> +16.2pp) to qwen-3.8-max's +6.4pp (+9.2pp -> +15.6pp); 0 of 7 model lines narrowed week over week. qwen-3.8-max is the lowest-Brier line (Brier 0.2689); gemini-3.1-pro leads on hit-rate (51.4%).",
  "**The high-confidence buckets, N>=10 only.** deepseek-v4-pro 70-80 55.2% (n=29), gemini-3.1-pro 70-80 53.4% (n=320), gemini-3.1-pro 80-100 53.3% (n=15), gpt-5.6-sol 70-80 48.6% (n=220). Cells below N=10 are marked insufficient and rank nothing.",
  "**Trading, kept separate from every calibration table.** 0 of 7 lines finished net-positive; net PnL ran -$5.94 (gemini-3.1-pro, best) to -$86.60 (deepseek-v4-pro, worst), field -$315.68 against issue #7's -$337.72."
 ],
 "market_check": {
  "label": "MARKET CHECK · OBSERVATION (two adjacent calendar weeks)",
  "text": "OBSERVATION — mean |1d move| 2.69% -> 1.83% (-32% rel), BTC realized vol 35.1% -> 35.8% (ann., hourly); field directional accuracy 50.7% -> 46.8% (-3.9pp), 0 of 7 models improved, sim win-rate up for 0 of 7, field sim PnL -$337.72 -> -$315.68 (adjacent calendar weeks Sep 14-20 vs Sep 21-27; descriptive, one pair of weeks, not a claim).",
  "robustness": "Robustness: raw price-sign accuracy 54.2% -> 49.2% — the same direction as the trade-based hit rule.",
  "weeks": {
   "A": {
    "days": "2026-09-14..2026-09-20",
    "slots": [
     "2026-09-14T00:00:00+00:00",
     "2026-09-21T00:00:00+00:00"
    ],
    "cutoff": "2026-09-21T16:00:00+00:00"
   },
   "B": {
    "days": "2026-09-21..2026-09-27",
    "slots": [
     "2026-09-21T00:00:00+00:00",
     "2026-09-28T00:00:00+00:00"
    ],
    "cutoff": "2026-09-28T16:00:00+00:00"
   }
  },
  "price_source": "crypto_spot/1h@:00",
  "accuracy_field": {
   "A": {
    "n": 5598,
    "hit_rate": 0.5075,
    "wilson_95": [
     0.4944,
     0.5206
    ]
   },
   "B": {
    "n": 4958,
    "hit_rate": 0.4683,
    "wilson_95": [
     0.4545,
     0.4822
    ]
   },
   "delta_pp": -3.9
  },
  "accuracy_per_model": [
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "A": {
     "n": 883,
     "hit_rate": 0.5289,
     "wilson_95": [
      0.4959,
      0.5616
     ]
    },
    "B": {
     "n": 794,
     "hit_rate": 0.5139,
     "wilson_95": [
      0.4791,
      0.5485
     ]
    },
    "delta_pp": -1.5
   },
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "A": {
     "n": 629,
     "hit_rate": 0.496,
     "wilson_95": [
      0.4571,
      0.535
     ]
    },
    "B": {
     "n": 537,
     "hit_rate": 0.4693,
     "wilson_95": [
      0.4274,
      0.5116
     ]
    },
    "delta_pp": -2.7
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "A": {
     "n": 821,
     "hit_rate": 0.497,
     "wilson_95": [
      0.4628,
      0.5311
     ]
    },
    "B": {
     "n": 744,
     "hit_rate": 0.4677,
     "wilson_95": [
      0.4321,
      0.5037
     ]
    },
    "delta_pp": -2.9
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "A": {
     "n": 711,
     "hit_rate": 0.5162,
     "wilson_95": [
      0.4795,
      0.5527
     ]
    },
    "B": {
     "n": 620,
     "hit_rate": 0.4758,
     "wilson_95": [
      0.4368,
      0.5151
     ]
    },
    "delta_pp": -4.0
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "A": {
     "n": 936,
     "hit_rate": 0.484,
     "wilson_95": [
      0.4521,
      0.516
     ]
    },
    "B": {
     "n": 777,
     "hit_rate": 0.4389,
     "wilson_95": [
      0.4044,
      0.474
     ]
    },
    "delta_pp": -4.5
   },
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "A": {
     "n": 799,
     "hit_rate": 0.5232,
     "wilson_95": [
      0.4885,
      0.5576
     ]
    },
    "B": {
     "n": 736,
     "hit_rate": 0.4674,
     "wilson_95": [
      0.4316,
      0.5035
     ]
    },
    "delta_pp": -5.6
   },
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "A": {
     "n": 819,
     "hit_rate": 0.5079,
     "wilson_95": [
      0.4737,
      0.5421
     ]
    },
    "B": {
     "n": 750,
     "hit_rate": 0.4453,
     "wilson_95": [
      0.4101,
      0.4811
     ]
    },
    "delta_pp": -6.3
   }
  ],
  "raw_sign_field": {
   "A": {
    "n_scored": 5583,
    "hit_rate": 0.5416
   },
   "B": {
    "n_scored": 4954,
    "hit_rate": 0.4923
   }
  },
  "abs_move_1d": {
   "A": 2.693,
   "B": 1.835,
   "delta_pct_rel": -31.9
  },
  "btc_realized_vol_ann_pct_hourly": {
   "A": 35.07,
   "B": 35.8
  },
  "trading_field": {
   "A": {
    "n_trades": 5598,
    "wr": 0.4337,
    "pnl_net_usd": -337.72,
    "pnl_gross_usd": 222.08
   },
   "B": {
    "n_trades": 4958,
    "wr": 0.3998,
    "pnl_net_usd": -315.68,
    "pnl_gross_usd": 180.12
   }
  },
  "verdict": {
   "models_with_hit_improved": "0 of 7",
   "models_with_rawsign_improved": "0 of 7",
   "models_with_wr_improved": "0 of 7",
   "field_hit_delta_pp": -3.9,
   "field_pnl_net": {
    "A": -337.72,
    "B": -315.68
   },
   "abs_move_delta": {
    "1h": {
     "A": 0.369,
     "B": 0.358,
     "delta_pct_rel": -3.0
    },
    "4h": {
     "A": 0.835,
     "B": 0.766,
     "delta_pct_rel": -8.3
    },
    "1d": {
     "A": 2.693,
     "B": 1.835,
     "delta_pct_rel": -31.9
    }
   },
   "btc_vol_ann_pct": {
    "A": 35.07,
    "B": 35.8
   },
   "note": "descriptive, two adjacent calendar weeks; hit rule = methodology v1.1 trade-based; raw_sign = price-sign robustness check (closes at :00 vs slots at :01)"
  },
  "lineage_note": "lineage splice ACTIVE: grok-4.5 rows are aggregated into grok-4.6 (flip 2026-08-24T09:22:00+00:00, four windows back, inside issue #4's window). Week A (2026-09-14..2026-09-20) is pure grok-4.6 (it reproduces the published issue #7); week B is pure grok-4.6."
 },
 "why_it_matters": "Every MarketMania forecast carries a model-stated confidence from 0 to 100 alongside its direction call. This report checks whether that number tracks reality: on a well-calibrated forecaster, calls made at 70% confidence should hit their direction about 70% of the time. With issue #8 the series has eight points on every model's gap -- enough to see movement, not enough to claim a trend: calibration and market regime move together, and durability claims belong in the Monthly series, not here.",
 "how_to_read_this": "Gap pp = mean stated confidence minus hit-rate, in percentage points; positive means overconfident. Brier is the mean squared error of the stated probability against the realised outcome, so lower is better and 0.25 is what an uninformative always-50% predictor scores. Coverage = calls scored divided by the mature calls available to that model. Confidence buckets are the models' own natural breakpoints, not an arbitrary binning; cells below N=10 are marked insufficient and never used to rank anything. Prediction and trading metrics live in separate sections and are never combined. The grok 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits four windows back (issue #4): both weeks of this issue are pure grok-4.6, with no 4.5-era rows in either window, so the grok week-over-week row is the third pure-4.6 against pure-4.6 comparison of the series.",
 "practical_implications": [
  "Stated confidence is still a ranking hint at best, not a probability: the field's gap is +16.1pp this week against +12.1pp in issue #7.",
  "Gap and regime move together. The field hit 46.8% against 50.7% in issue #7 and stated mean confidence read 63.0 against 62.8, so the gap came out at +16.1pp: a field that hits 46.8% instead of 50.7% closes or opens a confidence gap without a single stated number changing.",
  "Trading direction this week (0 of 7 lines net-positive) is a regime read, not a strategy result; the same lines printed the opposite sign inside six weeks."
 ],
 "limitations": [
  "Prediction metrics (this report) and trading metrics are kept in separate sections per methodology -- they are never combined into a single score.",
  "95% Wilson CIs shown are descriptive, not inferential: observations inside one window are dependent, so read them as a range, not a formal coverage guarantee.",
  "All 4,958 scored calls sit inside one market regime. This window ran 2 trend days of 7 (issue #7: 3 of 7); every comparison with issue #7 is a comparison across regimes as well as across weeks. Eight week-over-week points cannot separate drift from regime; no durability claim is made.",
  "Models report confidence at discrete levels, not a continuous scale; the buckets reflect those natural breakpoints. Cells below N=10 are marked insufficient.",
  "Underlying price series are reconstructed from trade entry prices (median per symbol-slot), not an independent tick feed. Series density and lineage notes (wave 8): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 21-27) daily coverage is FULL, as it was in issues #4, #5, #6 and #7: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 488, 1M 485). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, four windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#7 window), so every grok number in this issue and in issue #7 is a pure grok-4.6 line -- the grok week-over-week row is the third pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #7 after the audit that found the raw table mixes two exchanges; issue #7 values are as published.",
  "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 28 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6 and #7, which printed 52,971 at the Sep 21 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 57,929 is an additive step under one definition; counter deltas against issue #6 and earlier remain definitional."
 ],
 "notes": {
  "series_density_and_lineage_wave8": "Series density and lineage notes (wave 8): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 21-27) daily coverage is FULL, as it was in issues #4, #5, #6 and #7: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 488, 1M 485). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, four windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#7 window), so every grok number in this issue and in issue #7 is a pure grok-4.6 line -- the grok week-over-week row is the third pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "grok_era": {
   "ids_in_window": [
    "grok-4.6"
   ],
   "n_grok45": 0,
   "n_grok46": 1470,
   "n_grok45_rows": 0,
   "n_grok46_rows": 1470,
   "first_grok46_slot": "2026-09-21T00:01:00+00:00",
   "last_grok45_slot": null,
   "first_grok45_slot": null,
   "flip_at": "2026-08-24T09:22:00+00:00",
   "flip_inside_window": false,
   "prev_window": {
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470
   },
   "merged_model_id": "grok-4.6",
   "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
   "lineage_ids": [
    "grok-4.5",
    "grok-4.6"
   ],
   "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
   "lineage_warnings": [],
   "rows_remapped_a_plus_b": 0,
   "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- four windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #7) and week B (this window) are pure grok-4.6.",
   "cells_with_two_grok_rows": 0
  },
  "model_ids_line": [
   "claude-fable-5",
   "claude-opus-5",
   "deepseek-v4-pro",
   "gemini-3.1-pro",
   "gpt-5.6-sol",
   "grok-4.6",
   "qwen-3.8-max"
  ],
  "model_roster": {
   "current": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "current_count": 7,
   "current_raw_ids_in_window": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "archived_legacy": [
    "grok-4.5",
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "archived_legacy_count": 3,
   "archived_legacy_detail": [
    {
     "model": "grok-4.5",
     "n_rows_since": 11128,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-24T09:01:00+00:00"
    },
    {
     "model": "qwen-3.7-max",
     "n_rows_since": 7479,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-05T09:01:00+00:00"
    },
    {
     "model": "claude-opus-4.8",
     "n_rows_since": 6315,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-07-30T08:01:00+00:00"
    }
   ],
   "archived_excluding_lineage_merged": [
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "tracked_total": 10,
   "canon_expected": {
    "current": 7,
    "archived_legacy": 4,
    "tracked_total": 11
   },
   "matches_canon_11_7_4": false,
   "roster_since": "2026-07-11T00:00:00+00:00",
   "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
  },
  "series_density": {
   "1d": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "1h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 168,
    "n_rows": 5880,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "1m": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "1w": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "4h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 42,
    "n_rows": 2940,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   }
  },
  "week_over_week_note": "Week-over-week columns compare back-to-back windows: Sep 14-20 (issue #7, and week A of this issue's alive slice) vs Sep 21-27 (this issue). \"Was\" values are the numbers published in issue #7; deltas are computed on unrounded rates.",
  "research_to_date_pin": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 28 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6 and #7, which printed 52,971 at the Sep 21 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 57,929 is an additive step under one definition; counter deltas against issue #6 and earlier remain definitional.",
  "market_state_audit": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #7 after the audit that found the raw table mixes two exchanges; issue #7 values are as published.",
  "engine": "Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
  "report_template": "PDF built with the canonical mm_report.py template module; preview card with report_preview.render_preview"
 },
 "platform_note_tpsl": "**Context, not a finding.** Since Aug 13 the public sandbox can trade with model-specific TP/SL multipliers learned from this same weekly history (v1); since Aug 19, v2 adds per-ticker and confidence-bucket (50-70 / 70-100) resolution. That feature consumes calibration history; it does not feed back into any table in this report, which measures stated confidence vs. direction-hit only. No effect size is claimed for v2. The toggle lives at marketmania.ai/indices.",
 "living_series_note": "**Issue #8.** Weekly Calibration is a living series. Issue #7's open questions -- \"the gap moved -10.4pp while the field base moved +11.2pp -- the second issue running in which the two move nearly one-for-one in opposite directions; does that hold for a third?\" and \"The best-Brier line changed hands to claude-fable-5 -- does it hold for a second issue, or change hands again?\" -- read this issue as: field gap +12.1pp -> +16.1pp (0 of 7 lines narrowed) on a field base that moved -3.9pp, a third issue running in opposite directions, and the best-Brier line changed hands again: qwen-3.8-max at 0.2689 (issue #7: claude-fable-5 at 0.2593). Also on the record: gpt-5.6-sol's 70-80 bucket read 48.6% (n=220) against its own 60-70 at 45.6% (n=517). The grok row is pure grok-4.6 this issue and was pure grok-4.6 in issue #7; the 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits four windows back. Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
 "testing_next": [
  "Next issue: the gap moved +4.0pp while the field base moved -3.9pp -- the third issue running in which the two move nearly one-for-one in opposite directions; does that hold for a fourth?",
  "The best-Brier line changed hands again, to qwen-3.8-max -- does it hold for a second issue, or change hands again?",
  "Monthly series: is the overconfidence gap stable across market regimes (trend vs flat) at monthly n?"
 ],
 "related_research": [
  {
   "title": "Consensus Watch #8",
   "url": "https://marketmania.ai/research/reports/consensus-watch-2026-09-21.pdf"
  },
  {
   "title": "Weekly Model Watch #8",
   "url": "https://marketmania.ai/research/reports/model-watch-2026-09-21.pdf"
  },
  {
   "title": "Weekly Calibration #7",
   "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-14.pdf"
  }
 ],
 "related_research_note": "The three weekly reports publish together as one issue each week; each links straight to the others' PDF and to its own previous issue. Direct links are the posting rule from wave 2 on.",
 "citation": {
  "bibtex_key": "mm_calibration_2026w39",
  "title": "Weekly Calibration #8: confidence vs. direction-hit, Sep 21-27 2026",
  "author": "MarketMania Research",
  "year": 2026,
  "month": "September",
  "day": 30,
  "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.pdf",
  "note": "Methodology v1.1; window Sep 21-27, 2026 UTC; source weekly_metrics_2026-09-21.json"
 },
 "research_to_date": {
  "this_report": {
   "scored_observations": 4958
  },
  "platform": {
   "as_of_cutoff": "2026-09-28",
   "resolved_forecasts": 57929,
   "resolved_forecasts_definition": "directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, at the issue cutoff",
   "resolved_forecasts_since": "2026-07-11",
   "models_tracked": 10,
   "models_tracked_detail": "7 models tracked (current line-up; earlier versions folded into their successors' lineage)",
   "assets": 5,
   "forecast_horizons": 5,
   "cadence": "hourly",
   "published_research_reports": 36
  },
  "source": "weekly_metrics_2026-09-21.json",
  "control_reproduces_wave3_pin": true,
  "pin_note": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 28 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6 and #7, which printed 52,971 at the Sep 21 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 57,929 is an additive step under one definition; counter deltas against issue #6 and earlier remain definitional."
 },
 "site_entry": {
  "slug": "weekly-calibration-2026-09-21",
  "series": "weekly",
  "seriesLabel": "Weekly · Calibration",
  "date": "September 30, 2026",
  "title": "Weekly Calibration #8",
  "summary": "4,958 directional calls asked whether stated confidence tracks reality. The field hit 46.8% against 63.0 stated mean confidence, an overconfidence gap of +16.1pp against +12.1pp in issue #7, and 0 of 7 model lines narrowed week over week. Field Brier moved 0.2669 -> 0.2751; qwen-3.8-max is the lowest-Brier line at 0.2689 and gemini-3.1-pro leads on hit-rate at 51.4%. Trading, kept in its own section, finished with 0 of 7 lines net-positive as field accuracy moved 50.7% -> 46.8% (-3.9pp).",
  "stats": [
   "4,958 directional calls",
   "gap +16.1pp (was +12.1pp)",
   "0 of 7 net-positive"
  ],
  "manifest_key_finding": "The overconfidence gap read +16.1pp against +12.1pp in issue #7, with 0 of 7 model lines narrowing and field Brier moving 0.2669 -> 0.2751; trading, kept separate, finished 0 of 7 lines net-positive.",
  "files": {
   "pdf": "/research/reports/weekly-calibration-2026-09-21.pdf",
   "md": "/research/reports/weekly-calibration-2026-09-21.md",
   "json": "/research/reports/weekly-calibration-2026-09-21.json"
  }
 },
 "social": {
  "telegram": {
   "photo_caption_html": "🎯 <b>Weekly Calibration #8</b> — weekly series (window Sep 21–27)\n\nResearch question: when a model says 70, does it hit 70% of the time?\n\n4,958 directional calls across 7 model lines:\n• Field hit <b>46.8%</b> vs <b>63.0</b> stated — an overconfidence gap of <b>+16.1pp</b> (issue #7: +12.1pp)\n• Gap week over week: <b>0 of 7 lines narrowed</b>; field Brier 0.2669 -> 0.2751\n• Lowest-Brier line: <b>qwen-3.8-max</b> (Brier 0.2689); hit-rate leader gemini-3.1-pro at 51.4%\n• Market check: field accuracy 50.7% -> 46.8% (-3.9pp); trading kept separate — <b>0 of 7 net-positive</b>\n\nNote: grok is a pure 4.6 line in both weeks of this issue; the 4.5 -> 4.6 flip (Aug 24 09:22 UTC) sits four windows back\n\n📄 <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.pdf\">Full report (PDF)</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "document_caption_html": "📄 Weekly Calibration #8 — full report (PDF).\nWeb copy: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.pdf\">weekly-calibration-2026-09-21.pdf</a> · machine-readable: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.json\">JSON</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "caption_len": 934,
   "document_caption_len": 416,
   "limit": 1024
  },
  "x": {
   "main": "Weekly Calibration #8 is out.\n\n4,958 LLM calls: 46.8% hit vs 63.0 stated confidence — an overconfidence gap of +16.1pp against +12.1pp a week earlier, with 0 of 7 lines narrowing.\n\nFull report: https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.pdf\n\n#AI #Crypto #Research #Calibration",
   "main_tco_len": 253,
   "reply": "Overconfidence gap, w/w (pp):\ngrok-4.6: 10.7 -> 13.5\nopus-5: 9.9 -> 14.0\nfable-5: 9.1 -> 14.7\nqwen-3.8: 9.2 -> 15.6\ngemini-3.1: 14.0 -> 16.2\ndeepseek: 12.4 -> 17.0\ngpt-5.6: 18.4 -> 21.0\nField: +12.1pp -> +16.1pp. Methodology v1.1.\nData: https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.json",
   "reply_tco_len": 260,
   "limit_tco": 280,
   "replies_max": 1
  }
 }
}
