{
 "slug": "weekly-calibration-2026-09-07",
 "series": "weekly",
 "series_label": "WEEKLY · CALIBRATION",
 "issue": 6,
 "title": "Weekly Calibration #6",
 "language": "en",
 "window_start": "2026-09-07T00:00:00+00:00",
 "window_end": "2026-09-14T00:00:00+00:00",
 "window_text": "Sep 7-13, 2026 UTC",
 "cutoff": "2026-09-14T16:00:00+00:00",
 "generated_at": "2026-09-15T15:38:10+00:00",
 "methodology": "v1.1 (2026-08-10)",
 "methodology_hash": "e66c7e8c864a2233",
 "source": "weekly_metrics_2026-09-07.json",
 "hit_rule": "direction: exit tp1/tp2 -> hit, sl -> miss, expiry -> sign of gross pnl",
 "bibtex_key": "mm_calibration_2026w37",
 "market_state": {
  "panel": "market_state_6p2",
  "week": {
   "start": "2026-09-07",
   "end": "2026-09-14"
  },
  "prev_week": {
   "start": "2026-08-31",
   "end": "2026-09-07"
  },
  "current": {
   "btc_net_pct": -4.36,
   "realized_vol_ann_pct": 19.68,
   "volume_top5_usd_bn": 16.0492,
   "volume_top5_wow_pct": -1.2,
   "avg_pairwise_corr": 0.74
  },
  "previous_as_published_issue5": {
   "btc_net_pct": 3.42,
   "realized_vol_ann_pct": 43.35,
   "volume_top5_usd_bn": 16.2391,
   "avg_pairwise_corr": 0.78
  },
  "snapshot_line": "BTC net -4.36% (prior +3.42%) - ann vol 19.7% (was 43.4%) - TOP5 volume $16.05B, -1% w/w - pairwise corr 0.74 (was 0.78)",
  "exchange": "binance",
  "audit_filter_applied": true,
  "source": "single-exchange (binance) daily candle recompute, market_state_2026-09-07.json; prior-week values as published in issue #5",
  "audit_note": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #5 after the audit that found the raw table mixes two exchanges; issue #5 values are as published.",
  "prev_week_control": {
   "published_issue3": {
    "btc_net_pct": 3.42,
    "realized_vol_ann_pct": 43.35,
    "volume_top5_usd_bn": 16.2391,
    "avg_pairwise_corr": 0.78
   },
   "published_issue5": {
    "btc_net_pct": 3.42,
    "realized_vol_ann_pct": 43.35,
    "volume_top5_usd_bn": 16.2391,
    "avg_pairwise_corr": 0.78
   },
   "got": {
    "btc_net_pct": 3.42,
    "realized_vol_ann_pct": 43.35,
    "volume_top5_usd_bn": 16.2391,
    "avg_pairwise_corr": 0.78
   },
   "match": {
    "btc_net_pct": true,
    "realized_vol_ann_pct": true,
    "volume_top5_usd_bn": true,
    "avg_pairwise_corr": true
   },
   "tolerance": {
    "btc_net_pct": 0.05,
    "realized_vol_ann_pct": 0.5,
    "volume_top5_usd_bn": 0.05,
    "avg_pairwise_corr": 0.02
   },
   "all_match": true,
   "note": "previous week of this run == published issue-#5 week; the four Snapshot pins must reproduce (key published_issue3 kept for the builder schema, published_issue5 is the add-only alias)"
  },
  "regime_days": {
   "2026-09-07": {
    "move_pct": -1.52,
    "regime": "trend"
   },
   "2026-09-08": {
    "move_pct": -0.72,
    "regime": "flat"
   },
   "2026-09-09": {
    "move_pct": -0.25,
    "regime": "flat"
   },
   "2026-09-10": {
    "move_pct": -2.14,
    "regime": "trend"
   },
   "2026-09-11": {
    "move_pct": 0.85,
    "regime": "flat"
   },
   "2026-09-12": {
    "move_pct": 0.0,
    "regime": "flat"
   },
   "2026-09-13": {
    "move_pct": -0.57,
    "regime": "flat"
   }
  },
  "alive_slice_context": {
   "abs_move_1d_pct": {
    "A": 2.102,
    "B": 1.347
   },
   "btc_realized_vol_ann_pct_hourly": {
    "A": 35.22,
    "B": 31.4
   }
  }
 },
 "tables": {
  "calibration_main": [
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "legacy": false,
    "is_field": false,
    "n": 707,
    "coverage": 0.534,
    "hit_rate": 0.4003,
    "mean_conf": 58.8,
    "gap_pp": 18.8,
    "brier": 0.2794,
    "wilson_95": {
     "lo": 0.3648,
     "hi": 0.4368
    }
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "legacy": false,
    "is_field": false,
    "n": 727,
    "coverage": 0.5466,
    "hit_rate": 0.414,
    "mean_conf": 59.8,
    "gap_pp": 18.4,
    "brier": 0.2825,
    "wilson_95": {
     "lo": 0.3788,
     "hi": 0.4502
    }
   },
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "legacy": false,
    "is_field": false,
    "n": 518,
    "coverage": 0.3895,
    "hit_rate": 0.3958,
    "mean_conf": 60.0,
    "gap_pp": 20.5,
    "brier": 0.2847,
    "wilson_95": {
     "lo": 0.3546,
     "hi": 0.4385
    }
   },
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "legacy": false,
    "is_field": false,
    "n": 624,
    "coverage": 0.4692,
    "hit_rate": 0.3654,
    "mean_conf": 60.1,
    "gap_pp": 23.6,
    "brier": 0.2921,
    "wilson_95": {
     "lo": 0.3285,
     "hi": 0.4039
    }
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "legacy": false,
    "is_field": false,
    "n": 510,
    "coverage": 0.3835,
    "hit_rate": 0.3647,
    "mean_conf": 61.0,
    "gap_pp": 24.5,
    "brier": 0.2932,
    "wilson_95": {
     "lo": 0.3241,
     "hi": 0.4073
    }
   },
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "legacy": false,
    "is_field": false,
    "n": 673,
    "coverage": 0.506,
    "hit_rate": 0.4175,
    "mean_conf": 65.4,
    "gap_pp": 23.6,
    "brier": 0.3057,
    "wilson_95": {
     "lo": 0.3808,
     "hi": 0.4552
    }
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "legacy": false,
    "is_field": false,
    "n": 709,
    "coverage": 0.5343,
    "hit_rate": 0.4006,
    "mean_conf": 68.5,
    "gap_pp": 28.4,
    "brier": 0.323,
    "wilson_95": {
     "lo": 0.3651,
     "hi": 0.4371
    }
   },
   {
    "model": "Field (all models)",
    "legacy": false,
    "is_field": true,
    "n": 4468,
    "coverage": null,
    "hit_rate": 0.3957,
    "mean_conf": 62.1,
    "gap_pp": 22.5,
    "brier": 0.2948,
    "wilson_95": {
     "lo": null,
     "hi": null
    }
   }
  ],
  "buckets": [
   {
    "model": "qwen-3.8-max",
    "bucket": "50-60",
    "n": 386,
    "hit_rate": 0.4352,
    "mean_conf": 56.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "60-70",
    "n": 308,
    "hit_rate": 0.3571,
    "mean_conf": 62,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "70-80",
    "n": 1,
    "hit_rate": 0.0,
    "mean_conf": 70,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "50-60",
    "n": 308,
    "hit_rate": 0.4773,
    "mean_conf": 56.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "60-70",
    "n": 410,
    "hit_rate": 0.3707,
    "mean_conf": 62.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "70-80",
    "n": 8,
    "hit_rate": 0.125,
    "mean_conf": 70.2,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "grok-4.6",
    "bucket": "50-60",
    "n": 199,
    "hit_rate": 0.4322,
    "mean_conf": 57.3,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "60-70",
    "n": 317,
    "hit_rate": 0.3722,
    "mean_conf": 61.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "70-80",
    "n": 1,
    "hit_rate": 0.0,
    "mean_conf": 70,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-fable-5",
    "bucket": "50-60",
    "n": 234,
    "hit_rate": 0.4103,
    "mean_conf": 56.5,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "60-70",
    "n": 390,
    "hit_rate": 0.3385,
    "mean_conf": 62.3,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-fable-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "50-60",
    "n": 117,
    "hit_rate": 0.3675,
    "mean_conf": 57.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "60-70",
    "n": 393,
    "hit_rate": 0.3639,
    "mean_conf": 61.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "50-60",
    "n": 6,
    "hit_rate": 0.1667,
    "mean_conf": 55,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "60-70",
    "n": 473,
    "hit_rate": 0.4334,
    "mean_conf": 62.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "70-80",
    "n": 193,
    "hit_rate": 0.3834,
    "mean_conf": 72.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "50-60",
    "n": 11,
    "hit_rate": 0.3636,
    "mean_conf": 58.5,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "60-70",
    "n": 428,
    "hit_rate": 0.3902,
    "mean_conf": 65.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "70-80",
    "n": 269,
    "hit_rate": 0.4201,
    "mean_conf": 74.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "80-100",
    "n": 1,
    "hit_rate": 0.0,
    "mean_conf": 82,
    "insufficient": true,
    "no_data": false
   }
  ],
  "sub50": [
   {
    "model": "qwen-3.8-max",
    "n": 12,
    "hit_rate": 0.4167
   },
   {
    "model": "deepseek-v4-pro",
    "n": 1,
    "hit_rate": 1.0
   },
   {
    "model": "grok-4.6",
    "n": 1,
    "hit_rate": 1.0
   },
   {
    "model": "claude-fable-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "claude-opus-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gemini-3.1-pro",
    "n": 1,
    "hit_rate": 1.0
   },
   {
    "model": "gpt-5.6-sol",
    "n": 0,
    "hit_rate": null
   }
  ],
  "by_fh": [
   {
    "model": "qwen-3.8-max",
    "fh": "1h",
    "n": 439,
    "hit_rate": 0.4374,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "4h",
    "n": 231,
    "hit_rate": 0.368,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1d",
    "n": 37,
    "hit_rate": 0.1622,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1h",
    "n": 439,
    "hit_rate": 0.4533,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "4h",
    "n": 237,
    "hit_rate": 0.3966,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1d",
    "n": 51,
    "hit_rate": 0.1569,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "1h",
    "n": 327,
    "hit_rate": 0.4251,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "4h",
    "n": 169,
    "hit_rate": 0.3609,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "1d",
    "n": 22,
    "hit_rate": 0.2273,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "1h",
    "n": 387,
    "hit_rate": 0.4109,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "4h",
    "n": 203,
    "hit_rate": 0.3054,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "1d",
    "n": 34,
    "hit_rate": 0.2059,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1h",
    "n": 299,
    "hit_rate": 0.4214,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "4h",
    "n": 177,
    "hit_rate": 0.2994,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1d",
    "n": 34,
    "hit_rate": 0.2059,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1h",
    "n": 437,
    "hit_rate": 0.46,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "4h",
    "n": 201,
    "hit_rate": 0.3632,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1d",
    "n": 35,
    "hit_rate": 0.2,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1h",
    "n": 434,
    "hit_rate": 0.4447,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "4h",
    "n": 235,
    "hit_rate": 0.3489,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1d",
    "n": 40,
    "hit_rate": 0.225,
    "flag": null
   }
  ],
  "trading": [
   {
    "model": "grok-4.6",
    "n_trades": 518,
    "wr": 0.3243,
    "pnl_net_usd": -114.37,
    "pnl_gross_usd": -62.57,
    "max_dd_usd": -114.37
   },
   {
    "model": "gemini-3.1-pro",
    "n_trades": 673,
    "wr": 0.3373,
    "pnl_net_usd": -131.95,
    "pnl_gross_usd": -64.65,
    "max_dd_usd": -134.16
   },
   {
    "model": "claude-opus-5",
    "n_trades": 510,
    "wr": 0.2902,
    "pnl_net_usd": -137.47,
    "pnl_gross_usd": -86.47,
    "max_dd_usd": -137.47
   },
   {
    "model": "deepseek-v4-pro",
    "n_trades": 727,
    "wr": 0.3425,
    "pnl_net_usd": -152.22,
    "pnl_gross_usd": -79.52,
    "max_dd_usd": -152.76
   },
   {
    "model": "qwen-3.8-max",
    "n_trades": 707,
    "wr": 0.3267,
    "pnl_net_usd": -158.82,
    "pnl_gross_usd": -88.12,
    "max_dd_usd": -159.24
   },
   {
    "model": "claude-fable-5",
    "n_trades": 624,
    "wr": 0.2933,
    "pnl_net_usd": -160.85,
    "pnl_gross_usd": -98.45,
    "max_dd_usd": -160.85
   },
   {
    "model": "gpt-5.6-sol",
    "n_trades": 709,
    "wr": 0.323,
    "pnl_net_usd": -163.27,
    "pnl_gross_usd": -92.37,
    "max_dd_usd": -163.27
   }
  ],
  "gap_week_over_week": [
   {
    "model": "deepseek-v4-pro",
    "gap_pp_issue5": 16.5,
    "gap_pp_issue6": 18.4,
    "delta_pp": 1.9
   },
   {
    "model": "qwen-3.8-max",
    "gap_pp_issue5": 16.6,
    "gap_pp_issue6": 18.8,
    "delta_pp": 2.2
   },
   {
    "model": "grok-4.6",
    "gap_pp_issue5": 22.3,
    "gap_pp_issue6": 20.5,
    "delta_pp": -1.8
   },
   {
    "model": "claude-fable-5",
    "gap_pp_issue5": 17.9,
    "gap_pp_issue6": 23.6,
    "delta_pp": 5.7
   },
   {
    "model": "gemini-3.1-pro",
    "gap_pp_issue5": 19.7,
    "gap_pp_issue6": 23.6,
    "delta_pp": 3.9
   },
   {
    "model": "claude-opus-5",
    "gap_pp_issue5": 18.8,
    "gap_pp_issue6": 24.5,
    "delta_pp": 5.7
   },
   {
    "model": "gpt-5.6-sol",
    "gap_pp_issue5": 26.2,
    "gap_pp_issue6": 28.4,
    "delta_pp": 2.2
   },
   {
    "model": "Field",
    "gap_pp_issue5": 19.6,
    "gap_pp_issue6": 22.5,
    "delta_pp": 2.9
   }
  ],
  "ok_rate_by_model": {
   "qwen-3.8-max": 0.9959,
   "deepseek-v4-pro": 1.0,
   "grok-4.6": 1.0,
   "claude-fable-5": 1.0,
   "claude-opus-5": 1.0,
   "gemini-3.1-pro": 1.0,
   "gpt-5.6-sol": 0.998
  },
  "counters": {
   "forecasts_total": 10290,
   "ok_in_gate": 9301,
   "out_of_gate": 980,
   "out_of_gate_by_fh": {
    "1w": 490,
    "1m": 490
   },
   "invalid": 9,
   "mature": 9301,
   "mature_directional": 4468,
   "mature_sideways": 4833,
   "pending_next_issue": 0,
   "pending_by_fh": {},
   "late_closes": 0,
   "uptime": [
    {
     "fh": "1h",
     "tf": "1h",
     "slots_seen": 168,
     "slots_expected": 168
    },
    {
     "fh": "4h",
     "tf": "4h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "4h",
     "tf": "1h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "1d",
     "tf": "1d",
     "slots_seen": 7,
     "slots_expected": 7
    },
    {
     "fh": "1d",
     "tf": "4h",
     "slots_seen": 7,
     "slots_expected": 7
    }
   ],
   "regime_days": {
    "2026-09-07": {
     "move_pct": -1.52,
     "regime": "trend"
    },
    "2026-09-08": {
     "move_pct": -0.72,
     "regime": "flat"
    },
    "2026-09-09": {
     "move_pct": -0.25,
     "regime": "flat"
    },
    "2026-09-10": {
     "move_pct": -2.14,
     "regime": "trend"
    },
    "2026-09-11": {
     "move_pct": 0.85,
     "regime": "flat"
    },
    "2026-09-12": {
     "move_pct": 0.0,
     "regime": "flat"
    },
    "2026-09-13": {
     "move_pct": -0.57,
     "regime": "flat"
    }
   },
   "grok_era": {
    "ids_in_window": [
     "grok-4.6"
    ],
    "n_grok45": 0,
    "n_grok46": 1470,
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470,
    "first_grok46_slot": "2026-09-07T00:01:00+00:00",
    "last_grok45_slot": null,
    "first_grok45_slot": null,
    "flip_at": "2026-08-24T09:22:00+00:00",
    "flip_inside_window": false,
    "prev_window": {
     "n_grok45_rows": 0,
     "n_grok46_rows": 1470
    },
    "merged_model_id": "grok-4.6",
    "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "lineage_ids": [
     "grok-4.5",
     "grok-4.6"
    ],
    "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
    "lineage_warnings": [],
    "rows_remapped_a_plus_b": 0,
    "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- two windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #5) and week B (this window) are pure grok-4.6.",
    "cells_with_two_grok_rows": 0
   },
   "series_density": {
    "1d": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-07",
      "2026-09-08",
      "2026-09-09",
      "2026-09-10",
      "2026-09-11",
      "2026-09-12",
      "2026-09-13"
     ],
     "missing_days": []
    },
    "1h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 168,
     "n_rows": 5880,
     "days": [
      "2026-09-07",
      "2026-09-08",
      "2026-09-09",
      "2026-09-10",
      "2026-09-11",
      "2026-09-12",
      "2026-09-13"
     ],
     "missing_days": []
    },
    "1m": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-07",
      "2026-09-08",
      "2026-09-09",
      "2026-09-10",
      "2026-09-11",
      "2026-09-12",
      "2026-09-13"
     ],
     "missing_days": []
    },
    "1w": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-07",
      "2026-09-08",
      "2026-09-09",
      "2026-09-10",
      "2026-09-11",
      "2026-09-12",
      "2026-09-13"
     ],
     "missing_days": []
    },
    "4h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 42,
     "n_rows": 2940,
     "days": [
      "2026-09-07",
      "2026-09-08",
      "2026-09-09",
      "2026-09-10",
      "2026-09-11",
      "2026-09-12",
      "2026-09-13"
     ],
     "missing_days": []
    }
   },
   "model_roster": {
    "current": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "current_count": 7,
    "current_raw_ids_in_window": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "archived_legacy": [
     "grok-4.5",
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "archived_legacy_count": 3,
    "archived_legacy_detail": [
     {
      "model": "grok-4.5",
      "n_rows_since": 11128,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-24T09:01:00+00:00"
     },
     {
      "model": "qwen-3.7-max",
      "n_rows_since": 7479,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-05T09:01:00+00:00"
     },
     {
      "model": "claude-opus-4.8",
      "n_rows_since": 6315,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-07-30T08:01:00+00:00"
     }
    ],
    "archived_excluding_lineage_merged": [
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "tracked_total": 10,
    "canon_expected": {
     "current": 7,
     "archived_legacy": 4,
     "tracked_total": 11
    },
    "matches_canon_11_7_4": false,
    "roster_since": "2026-07-11T00:00:00+00:00",
    "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
   }
  }
 },
 "key_finding": {
  "label": "KEY FINDING · OBSERVATION (one weekly window)",
  "text": "The field's overconfidence gap moved +19.6pp -> +22.5pp week over week, 1 of 7 model lines narrowed, and field Brier went 0.2874 -> 0.2948.",
  "evidence_level": "observation"
 },
 "key_findings": [
  "**OBSERVATION -- The gap read +22.5pp this week.** Directional calls hit 39.6% against 62.1 stated mean confidence (issue #5: +19.6pp on 42.8% vs 62.4). Field Brier is 0.2948 (was 0.2874) and 0 of 7 model lines scored below 0.25, the score of an uninformative always-50% predictor. One window, descriptive.",
  "**1 of 7 models narrowed their gap.** Gap moves ran from grok-4.6's -1.8pp (+22.3pp -> +20.5pp) to claude-fable-5's +5.7pp (+17.9pp -> +23.6pp); 1 of 7 model lines narrowed week over week. qwen-3.8-max is the best-calibrated line (Brier 0.2794); gemini-3.1-pro leads on hit-rate (41.8%).",
  "**The high-confidence buckets, N>=10 only.** gemini-3.1-pro 70-80 38.3% (n=193), gpt-5.6-sol 70-80 42.0% (n=269). Cells below N=10 are marked insufficient and rank nothing.",
  "**Trading, kept separate from every calibration table.** 0 of 7 lines finished net-positive; net PnL ran -$114.37 (grok-4.6, best) to -$163.27 (gpt-5.6-sol, worst), field -$1018.93 against issue #5's -$669.03."
 ],
 "market_check": {
  "label": "MARKET CHECK · OBSERVATION (two adjacent calendar weeks)",
  "text": "OBSERVATION — mean |1d move| 2.10% -> 1.35% (-36% rel), BTC realized vol 35.2% -> 31.4% (ann., hourly); field directional accuracy 42.8% -> 39.6% (-3.3pp), 1 of 7 models improved, sim win-rate up for 1 of 7, field sim PnL -$669.03 -> -$1018.93 (adjacent calendar weeks Aug 31-Sep 6 vs Sep 7-13; descriptive, one pair of weeks, not a claim).",
  "robustness": "Robustness: raw price-sign accuracy 51.6% -> 38.7% — the same direction as the trade-based hit rule.",
  "weeks": {
   "A": {
    "days": "2026-08-31..2026-09-06",
    "slots": [
     "2026-08-31T00:00:00+00:00",
     "2026-09-07T00:00:00+00:00"
    ],
    "cutoff": "2026-09-07T16:00:00+00:00"
   },
   "B": {
    "days": "2026-09-07..2026-09-13",
    "slots": [
     "2026-09-07T00:00:00+00:00",
     "2026-09-14T00:00:00+00:00"
    ],
    "cutoff": "2026-09-14T16:00:00+00:00"
   }
  },
  "price_source": "crypto_spot/1h@:00",
  "accuracy_field": {
   "A": {
    "n": 4520,
    "hit_rate": 0.4283,
    "wilson_95": [
     0.414,
     0.4428
    ]
   },
   "B": {
    "n": 4468,
    "hit_rate": 0.3957,
    "wilson_95": [
     0.3815,
     0.4101
    ]
   },
   "delta_pp": -3.3
  },
  "accuracy_per_model": [
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "A": {
     "n": 476,
     "hit_rate": 0.3761,
     "wilson_95": [
      0.3337,
      0.4204
     ]
    },
    "B": {
     "n": 518,
     "hit_rate": 0.3958,
     "wilson_95": [
      0.3546,
      0.4385
     ]
    },
    "delta_pp": 2.0
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "A": {
     "n": 689,
     "hit_rate": 0.4165,
     "wilson_95": [
      0.3803,
      0.4537
     ]
    },
    "B": {
     "n": 709,
     "hit_rate": 0.4006,
     "wilson_95": [
      0.3651,
      0.4371
     ]
    },
    "delta_pp": -1.6
   },
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "A": {
     "n": 697,
     "hit_rate": 0.4247,
     "wilson_95": [
      0.3885,
      0.4617
     ]
    },
    "B": {
     "n": 707,
     "hit_rate": 0.4003,
     "wilson_95": [
      0.3648,
      0.4368
     ]
    },
    "delta_pp": -2.4
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "A": {
     "n": 724,
     "hit_rate": 0.442,
     "wilson_95": [
      0.4062,
      0.4784
     ]
    },
    "B": {
     "n": 727,
     "hit_rate": 0.414,
     "wilson_95": [
      0.3788,
      0.4502
     ]
    },
    "delta_pp": -2.8
   },
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "A": {
     "n": 761,
     "hit_rate": 0.4652,
     "wilson_95": [
      0.43,
      0.5007
     ]
    },
    "B": {
     "n": 673,
     "hit_rate": 0.4175,
     "wilson_95": [
      0.3808,
      0.4552
     ]
    },
    "delta_pp": -4.8
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "A": {
     "n": 545,
     "hit_rate": 0.4239,
     "wilson_95": [
      0.383,
      0.4657
     ]
    },
    "B": {
     "n": 510,
     "hit_rate": 0.3647,
     "wilson_95": [
      0.3241,
      0.4073
     ]
    },
    "delta_pp": -5.9
   },
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "A": {
     "n": 628,
     "hit_rate": 0.4283,
     "wilson_95": [
      0.3902,
      0.4674
     ]
    },
    "B": {
     "n": 624,
     "hit_rate": 0.3654,
     "wilson_95": [
      0.3285,
      0.4039
     ]
    },
    "delta_pp": -6.3
   }
  ],
  "raw_sign_field": {
   "A": {
    "n_scored": 4507,
    "hit_rate": 0.5156
   },
   "B": {
    "n_scored": 4466,
    "hit_rate": 0.3871
   }
  },
  "abs_move_1d": {
   "A": 2.102,
   "B": 1.347,
   "delta_pct_rel": -35.9
  },
  "btc_realized_vol_ann_pct_hourly": {
   "A": 35.22,
   "B": 31.4
  },
  "trading_field": {
   "A": {
    "n_trades": 4520,
    "wr": 0.3564,
    "pnl_net_usd": -669.03,
    "pnl_gross_usd": -217.03
   },
   "B": {
    "n_trades": 4468,
    "wr": 0.3212,
    "pnl_net_usd": -1018.93,
    "pnl_gross_usd": -572.13
   }
  },
  "verdict": {
   "models_with_hit_improved": "1 of 7",
   "models_with_rawsign_improved": "0 of 7",
   "models_with_wr_improved": "1 of 7",
   "field_hit_delta_pp": -3.3,
   "field_pnl_net": {
    "A": -669.03,
    "B": -1018.93
   },
   "abs_move_delta": {
    "1h": {
     "A": 0.331,
     "B": 0.333,
     "delta_pct_rel": 0.6
    },
    "4h": {
     "A": 0.664,
     "B": 0.605,
     "delta_pct_rel": -8.9
    },
    "1d": {
     "A": 2.102,
     "B": 1.347,
     "delta_pct_rel": -35.9
    }
   },
   "btc_vol_ann_pct": {
    "A": 35.22,
    "B": 31.4
   },
   "note": "descriptive, two adjacent calendar weeks; hit rule = methodology v1.1 trade-based; raw_sign = price-sign robustness check (closes at :00 vs slots at :01)"
  },
  "lineage_note": "lineage splice ACTIVE: grok-4.5 rows are aggregated into grok-4.6 (flip 2026-08-24T09:22:00+00:00, two windows back, inside issue #4's window). Week A (2026-08-31..2026-09-06) is pure grok-4.6 (it reproduces the published issue #5); week B is pure grok-4.6."
 },
 "why_it_matters": "Every MarketMania forecast carries a model-stated confidence from 0 to 100 alongside its direction call. This report checks whether that number tracks reality: on a well-calibrated forecaster, calls made at 70% confidence should hit their direction about 70% of the time. With issue #6 the series has six points on every model's gap -- enough to see movement, not enough to claim a trend: calibration and market regime move together, and durability claims belong in the Monthly series, not here.",
 "how_to_read_this": "Gap pp = mean stated confidence minus hit-rate, in percentage points; positive means overconfident. Brier is the mean squared error of the stated probability against the realised outcome, so lower is better and 0.25 is what an uninformative always-50% predictor scores. Coverage = calls scored divided by the mature calls available to that model. Confidence buckets are the models' own natural breakpoints, not an arbitrary binning; cells below N=10 are marked insufficient and never used to rank anything. Prediction and trading metrics live in separate sections and are never combined. The grok 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits two windows back (issue #4): both weeks of this issue are pure grok-4.6, with no 4.5-era rows in either window, so the grok week-over-week row is the first pure-4.6 against pure-4.6 comparison of the series.",
 "practical_implications": [
  "Stated confidence is still a ranking hint at best, not a probability: the field's gap is +22.5pp this week against +19.6pp in issue #5.",
  "Gap and regime move together. The field hit 39.6% against 42.8% in issue #5 and stated mean confidence read 62.1 against 62.4, so the gap held at +22.5pp: a field that hits 39.6% instead of 42.8% closes or opens a confidence gap without a single stated number changing.",
  "Trading direction this week (0 of 7 lines net-positive) is a regime read, not a strategy result; the same lines printed the opposite sign inside four weeks."
 ],
 "limitations": [
  "Prediction metrics (this report) and trading metrics are kept in separate sections per methodology -- they are never combined into a single score.",
  "95% Wilson CIs shown are descriptive, not inferential: observations inside one window are dependent, so read them as a range, not a formal coverage guarantee.",
  "All 4,468 scored calls sit inside one market regime. This window ran 2 trend days of 7 (issue #5: 3 of 7); every comparison with issue #5 is a comparison across regimes as well as across weeks. Six week-over-week points cannot separate drift from regime; no durability claim is made.",
  "Models report confidence at discrete levels, not a continuous scale; the buckets reflect those natural breakpoints. Cells below N=10 are marked insufficient.",
  "Underlying price series are reconstructed from trade entry prices (median per symbol-slot), not an independent tick feed. Series density and lineage notes (wave 6): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 7-13) daily coverage is FULL, as it was in issues #4 and #5: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 490, 1M 490). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, two windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#5 window), so every grok number in this issue and in issue #5 is a pure grok-4.6 line -- the grok week-over-week row is the first pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #5 after the audit that found the raw table mixes two exchanges; issue #5 values are as published.",
  "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 14 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4 and #5, which printed 42,905 at the Sep 7 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 47,373 is an additive step under one definition; counter deltas against issue #4 and earlier remain definitional."
 ],
 "notes": {
  "series_density_and_lineage_wave6": "Series density and lineage notes (wave 6): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 7-13) daily coverage is FULL, as it was in issues #4 and #5: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 490, 1M 490). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, two windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#5 window), so every grok number in this issue and in issue #5 is a pure grok-4.6 line -- the grok week-over-week row is the first pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "grok_era": {
   "ids_in_window": [
    "grok-4.6"
   ],
   "n_grok45": 0,
   "n_grok46": 1470,
   "n_grok45_rows": 0,
   "n_grok46_rows": 1470,
   "first_grok46_slot": "2026-09-07T00:01:00+00:00",
   "last_grok45_slot": null,
   "first_grok45_slot": null,
   "flip_at": "2026-08-24T09:22:00+00:00",
   "flip_inside_window": false,
   "prev_window": {
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470
   },
   "merged_model_id": "grok-4.6",
   "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
   "lineage_ids": [
    "grok-4.5",
    "grok-4.6"
   ],
   "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
   "lineage_warnings": [],
   "rows_remapped_a_plus_b": 0,
   "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- two windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #5) and week B (this window) are pure grok-4.6.",
   "cells_with_two_grok_rows": 0
  },
  "model_ids_line": [
   "claude-fable-5",
   "claude-opus-5",
   "deepseek-v4-pro",
   "gemini-3.1-pro",
   "gpt-5.6-sol",
   "grok-4.6",
   "qwen-3.8-max"
  ],
  "model_roster": {
   "current": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "current_count": 7,
   "current_raw_ids_in_window": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "archived_legacy": [
    "grok-4.5",
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "archived_legacy_count": 3,
   "archived_legacy_detail": [
    {
     "model": "grok-4.5",
     "n_rows_since": 11128,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-24T09:01:00+00:00"
    },
    {
     "model": "qwen-3.7-max",
     "n_rows_since": 7479,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-05T09:01:00+00:00"
    },
    {
     "model": "claude-opus-4.8",
     "n_rows_since": 6315,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-07-30T08:01:00+00:00"
    }
   ],
   "archived_excluding_lineage_merged": [
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "tracked_total": 10,
   "canon_expected": {
    "current": 7,
    "archived_legacy": 4,
    "tracked_total": 11
   },
   "matches_canon_11_7_4": false,
   "roster_since": "2026-07-11T00:00:00+00:00",
   "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
  },
  "series_density": {
   "1d": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-07",
     "2026-09-08",
     "2026-09-09",
     "2026-09-10",
     "2026-09-11",
     "2026-09-12",
     "2026-09-13"
    ],
    "missing_days": []
   },
   "1h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 168,
    "n_rows": 5880,
    "days": [
     "2026-09-07",
     "2026-09-08",
     "2026-09-09",
     "2026-09-10",
     "2026-09-11",
     "2026-09-12",
     "2026-09-13"
    ],
    "missing_days": []
   },
   "1m": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-07",
     "2026-09-08",
     "2026-09-09",
     "2026-09-10",
     "2026-09-11",
     "2026-09-12",
     "2026-09-13"
    ],
    "missing_days": []
   },
   "1w": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-07",
     "2026-09-08",
     "2026-09-09",
     "2026-09-10",
     "2026-09-11",
     "2026-09-12",
     "2026-09-13"
    ],
    "missing_days": []
   },
   "4h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 42,
    "n_rows": 2940,
    "days": [
     "2026-09-07",
     "2026-09-08",
     "2026-09-09",
     "2026-09-10",
     "2026-09-11",
     "2026-09-12",
     "2026-09-13"
    ],
    "missing_days": []
   }
  },
  "week_over_week_note": "Week-over-week columns compare back-to-back windows: Aug 31-Sep 6 (issue #5, and week A of this issue's alive slice) vs Sep 7-13 (this issue). \"Was\" values are the numbers published in issue #5; deltas are computed on unrounded rates.",
  "research_to_date_pin": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 14 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4 and #5, which printed 42,905 at the Sep 7 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 47,373 is an additive step under one definition; counter deltas against issue #4 and earlier remain definitional.",
  "market_state_audit": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #5 after the audit that found the raw table mixes two exchanges; issue #5 values are as published.",
  "engine": "Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
  "report_template": "PDF built with the canonical mm_report.py template module; preview card with report_preview.render_preview"
 },
 "platform_note_tpsl": "**Context, not a finding.** Since Aug 13 the public sandbox can trade with model-specific TP/SL multipliers learned from this same weekly history (v1); since Aug 19, v2 adds per-ticker and confidence-bucket (50-70 / 70-100) resolution. That feature consumes calibration history; it does not feed back into any table in this report, which measures stated confidence vs. direction-hit only. No effect size is claimed for v2. The toggle lives at marketmania.ai/indices.",
 "living_series_note": "**Issue #6.** Weekly Calibration is a living series. Issue #5's open questions -- does the field gap hold near +19.6pp for a third issue or track the base rate one-for-one, and does the best-Brier line change hands again? -- read this issue as: field gap +19.6pp -> +22.5pp (1 of 7 lines narrowed) on a field base that moved -3.3pp, and the best-Brier line did not change hands: qwen-3.8-max holds it two issues running (0.2794, issue #5: qwen-3.8-max 0.2755). Also on the record: gpt-5.6-sol's 70-80 bucket read 42.0% (n=269) against its own 60-70 at 39.0% (n=428). The grok row is pure grok-4.6 this issue and was pure grok-4.6 in issue #5; the 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits two windows back. Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
 "testing_next": [
  "Next issue: the gap moved +2.9pp while the field base moved -3.3pp -- does the gap keep tracking the base one-for-one, or does it hold a level of its own?",
  "Does qwen-3.8-max hold the best-Brier line for a third issue, or does it change hands?",
  "Monthly series: is the overconfidence gap stable across market regimes (trend vs flat) at monthly n?"
 ],
 "related_research": [
  {
   "title": "Consensus Watch #6",
   "url": "https://marketmania.ai/research/reports/consensus-watch-2026-09-07.pdf"
  },
  {
   "title": "Weekly Model Watch #6",
   "url": "https://marketmania.ai/research/reports/model-watch-2026-09-07.pdf"
  },
  {
   "title": "Weekly Calibration #5",
   "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-08-31.pdf"
  }
 ],
 "related_research_note": "The three weekly reports publish together as one issue each week; each links straight to the others' PDF and to its own previous issue. Direct links are the posting rule from wave 2 on.",
 "citation": {
  "bibtex_key": "mm_calibration_2026w37",
  "title": "Weekly Calibration #6: confidence vs. direction-hit, Sep 7-13 2026",
  "author": "MarketMania Research",
  "year": 2026,
  "month": "September",
  "day": 15,
  "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-07.pdf",
  "note": "Methodology v1.1; window Sep 7-13, 2026 UTC; source weekly_metrics_2026-09-07.json"
 },
 "research_to_date": {
  "this_report": {
   "scored_observations": 4468
  },
  "platform": {
   "as_of_cutoff": "2026-09-14",
   "resolved_forecasts": 47373,
   "resolved_forecasts_definition": "directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, at the issue cutoff",
   "resolved_forecasts_since": "2026-07-11",
   "models_tracked": 10,
   "models_tracked_detail": "7 models tracked (current line-up; earlier versions folded into their successors' lineage)",
   "assets": 5,
   "forecast_horizons": 5,
   "cadence": "hourly",
   "published_research_reports": 28
  },
  "source": "weekly_metrics_2026-09-07.json",
  "control_reproduces_wave3_pin": true,
  "pin_note": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 14 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4 and #5, which printed 42,905 at the Sep 7 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 47,373 is an additive step under one definition; counter deltas against issue #4 and earlier remain definitional."
 },
 "site_entry": {
  "slug": "weekly-calibration-2026-09-07",
  "series": "weekly",
  "seriesLabel": "Weekly · Calibration",
  "date": "September 15, 2026",
  "title": "Weekly Calibration #6",
  "summary": "4,468 directional calls asked whether stated confidence tracks reality. The field hit 39.6% against 62.1 stated mean confidence, an overconfidence gap of +22.5pp against +19.6pp in issue #5, and 1 of 7 model lines narrowed week over week. Field Brier moved 0.2874 -> 0.2948; qwen-3.8-max is the best-calibrated line at 0.2794 and gemini-3.1-pro leads on hit-rate at 41.8%. Trading, kept in its own section, finished with 0 of 7 lines net-positive as field accuracy moved 42.8% -> 39.6% (-3.3pp).",
  "stats": [
   "4,468 directional calls",
   "gap +22.5pp (was +19.6pp)",
   "0 of 7 net-positive"
  ],
  "manifest_key_finding": "The overconfidence gap read +22.5pp against +19.6pp in issue #5, with 1 of 7 model lines narrowing and field Brier moving 0.2874 -> 0.2948; trading, kept separate, finished 0 of 7 lines net-positive.",
  "files": {
   "pdf": "/research/reports/weekly-calibration-2026-09-07.pdf",
   "md": "/research/reports/weekly-calibration-2026-09-07.md",
   "json": "/research/reports/weekly-calibration-2026-09-07.json"
  }
 },
 "social": {
  "telegram": {
   "photo_caption_html": "🎯 <b>Weekly Calibration #6</b> — weekly series (window Sep 7–13)\n\nResearch question: when a model says 70, does it hit 70% of the time?\n\n4,468 directional calls across 7 model lines:\n• Field hit <b>39.6%</b> vs <b>62.1</b> stated — an overconfidence gap of <b>+22.5pp</b> (issue #5: +19.6pp)\n• Gap week over week: <b>1 of 7 lines narrowed</b>; field Brier 0.2874 -> 0.2948\n• Best-calibrated line: <b>qwen-3.8-max</b> (Brier 0.2794); hit-rate leader gemini-3.1-pro at 41.8%\n• Market check: field accuracy 42.8% -> 39.6% (-3.3pp); trading kept separate — <b>0 of 7 net-positive</b>\n\nNote: grok is a pure 4.6 line in both weeks of this issue; the 4.5 -> 4.6 flip (Aug 24 09:22 UTC) sits two windows back\n\n📄 <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-07.pdf\">Full report (PDF)</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "document_caption_html": "📄 Weekly Calibration #6 — full report (PDF).\nWeb copy: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-07.pdf\">weekly-calibration-2026-09-07.pdf</a> · machine-readable: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-07.json\">JSON</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "caption_len": 935,
   "document_caption_len": 416,
   "limit": 1024
  },
  "x": {
   "main": "Weekly Calibration #6 is out.\n\n4,468 LLM calls: 39.6% hit vs 62.1 stated confidence — an overconfidence gap of +22.5pp against +19.6pp a week earlier, with 1 of 7 lines narrowing.\n\nFull report: https://marketmania.ai/research/reports/weekly-calibration-2026-09-07.pdf\n\n#AI #Crypto #Research #Calibration",
   "main_tco_len": 253,
   "reply": "Overconfidence gap, w/w (pp):\ndeepseek: 16.5 -> 18.4\nqwen-3.8: 16.6 -> 18.8\ngrok-4.6: 22.3 -> 20.5\nfable-5: 17.9 -> 23.6\ngemini-3.1: 19.7 -> 23.6\nopus-5: 18.8 -> 24.5\ngpt-5.6: 26.2 -> 28.4\nField: +19.6pp -> +22.5pp. Methodology v1.1.\nData: https://marketmania.ai/research/reports/weekly-calibration-2026-09-07.json",
   "reply_tco_len": 263,
   "limit_tco": 280,
   "replies_max": 1
  }
 }
}
