{
 "slug": "weekly-calibration-2026-09-14",
 "series": "weekly",
 "series_label": "WEEKLY · CALIBRATION",
 "issue": 7,
 "title": "Weekly Calibration #7",
 "language": "en",
 "window_start": "2026-09-14T00:00:00+00:00",
 "window_end": "2026-09-21T00:00:00+00:00",
 "window_text": "Sep 14-20, 2026 UTC",
 "cutoff": "2026-09-21T16:00:00+00:00",
 "generated_at": "2026-09-22T07:27:22+00:00",
 "methodology": "v1.1 (2026-08-10)",
 "methodology_hash": "e66c7e8c864a2233",
 "source": "weekly_metrics_2026-09-14.json",
 "hit_rule": "direction: exit tp1/tp2 -> hit, sl -> miss, expiry -> sign of gross pnl",
 "bibtex_key": "mm_calibration_2026w38",
 "market_state": {
  "panel": "market_state_6p2",
  "week": {
   "start": "2026-09-14",
   "end": "2026-09-21"
  },
  "prev_week": {
   "start": "2026-09-07",
   "end": "2026-09-14"
  },
  "current": {
   "btc_net_pct": 5.64,
   "realized_vol_ann_pct": 51.01,
   "volume_top5_usd_bn": 18.7386,
   "volume_top5_wow_pct": 16.8,
   "avg_pairwise_corr": 0.93
  },
  "previous_as_published_issue6": {
   "btc_net_pct": -4.36,
   "realized_vol_ann_pct": 19.68,
   "volume_top5_usd_bn": 16.0492,
   "avg_pairwise_corr": 0.74
  },
  "snapshot_line": "BTC net +5.64% (prior -4.36%) - ann vol 51.0% (was 19.7%) - TOP5 volume $18.74B, +17% w/w - pairwise corr 0.93 (was 0.74)",
  "exchange": "binance",
  "audit_filter_applied": true,
  "source": "single-exchange (binance) daily candle recompute, market_state_2026-09-14.json; prior-week values as published in issue #6",
  "audit_note": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #6 after the audit that found the raw table mixes two exchanges; issue #6 values are as published.",
  "prev_week_control": {
   "published_issue3": {
    "btc_net_pct": -4.36,
    "realized_vol_ann_pct": 19.68,
    "volume_top5_usd_bn": 16.0492,
    "avg_pairwise_corr": 0.74
   },
   "published_issue6": {
    "btc_net_pct": -4.36,
    "realized_vol_ann_pct": 19.68,
    "volume_top5_usd_bn": 16.0492,
    "avg_pairwise_corr": 0.74
   },
   "got": {
    "btc_net_pct": -4.36,
    "realized_vol_ann_pct": 19.68,
    "volume_top5_usd_bn": 16.0492,
    "avg_pairwise_corr": 0.74
   },
   "match": {
    "btc_net_pct": true,
    "realized_vol_ann_pct": true,
    "volume_top5_usd_bn": true,
    "avg_pairwise_corr": true
   },
   "tolerance": {
    "btc_net_pct": 0.05,
    "realized_vol_ann_pct": 0.5,
    "volume_top5_usd_bn": 0.05,
    "avg_pairwise_corr": 0.02
   },
   "all_match": true,
   "note": "previous week of this run == published issue-#6 week; the four Snapshot pins must reproduce (key published_issue3 kept for the builder schema, published_issue6 is the add-only alias)"
  },
  "regime_days": {
   "2026-09-14": {
    "move_pct": 1.79,
    "regime": "trend"
   },
   "2026-09-15": {
    "move_pct": -3.35,
    "regime": "trend"
   },
   "2026-09-16": {
    "move_pct": 0.86,
    "regime": "flat"
   },
   "2026-09-17": {
    "move_pct": 0.21,
    "regime": "flat"
   },
   "2026-09-18": {
    "move_pct": 5.82,
    "regime": "trend"
   },
   "2026-09-19": {
    "move_pct": 0.46,
    "regime": "flat"
   },
   "2026-09-20": {
    "move_pct": 0.13,
    "regime": "flat"
   }
  },
  "alive_slice_context": {
   "abs_move_1d_pct": {
    "A": 1.347,
    "B": 2.693
   },
   "btc_realized_vol_ann_pct_hourly": {
    "A": 31.4,
    "B": 35.07
   }
  }
 },
 "tables": {
  "calibration_main": [
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "legacy": false,
    "is_field": false,
    "n": 799,
    "coverage": 0.6008,
    "hit_rate": 0.5232,
    "mean_conf": 61.4,
    "gap_pp": 9.1,
    "brier": 0.2593,
    "wilson_95": {
     "lo": 0.4885,
     "hi": 0.5576
    }
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "legacy": false,
    "is_field": false,
    "n": 711,
    "coverage": 0.5346,
    "hit_rate": 0.5162,
    "mean_conf": 61.5,
    "gap_pp": 9.9,
    "brier": 0.2603,
    "wilson_95": {
     "lo": 0.4795,
     "hi": 0.5527
    }
   },
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "legacy": false,
    "is_field": false,
    "n": 819,
    "coverage": 0.6163,
    "hit_rate": 0.5079,
    "mean_conf": 60.0,
    "gap_pp": 9.2,
    "brier": 0.2604,
    "wilson_95": {
     "lo": 0.4737,
     "hi": 0.5421
    }
   },
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "legacy": false,
    "is_field": false,
    "n": 629,
    "coverage": 0.4733,
    "hit_rate": 0.496,
    "mean_conf": 60.3,
    "gap_pp": 10.7,
    "brier": 0.262,
    "wilson_95": {
     "lo": 0.4571,
     "hi": 0.535
    }
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "legacy": false,
    "is_field": false,
    "n": 936,
    "coverage": 0.7038,
    "hit_rate": 0.484,
    "mean_conf": 60.8,
    "gap_pp": 12.4,
    "brier": 0.266,
    "wilson_95": {
     "lo": 0.4521,
     "hi": 0.516
    }
   },
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "legacy": false,
    "is_field": false,
    "n": 883,
    "coverage": 0.6639,
    "hit_rate": 0.5289,
    "mean_conf": 66.8,
    "gap_pp": 14.0,
    "brier": 0.2701,
    "wilson_95": {
     "lo": 0.4959,
     "hi": 0.5616
    }
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "legacy": false,
    "is_field": false,
    "n": 821,
    "coverage": 0.6173,
    "hit_rate": 0.497,
    "mean_conf": 68.1,
    "gap_pp": 18.4,
    "brier": 0.2876,
    "wilson_95": {
     "lo": 0.4628,
     "hi": 0.5311
    }
   },
   {
    "model": "Field (all models)",
    "legacy": false,
    "is_field": true,
    "n": 5598,
    "coverage": null,
    "hit_rate": 0.5075,
    "mean_conf": 62.8,
    "gap_pp": 12.1,
    "brier": 0.2669,
    "wilson_95": {
     "lo": null,
     "hi": null
    }
   }
  ],
  "buckets": [
   {
    "model": "claude-fable-5",
    "bucket": "50-60",
    "n": 186,
    "hit_rate": 0.543,
    "mean_conf": 56.6,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "60-70",
    "n": 613,
    "hit_rate": 0.5171,
    "mean_conf": 62.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-fable-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "50-60",
    "n": 117,
    "hit_rate": 0.5385,
    "mean_conf": 57.6,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "60-70",
    "n": 594,
    "hit_rate": 0.5118,
    "mean_conf": 62.3,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "50-60",
    "n": 349,
    "hit_rate": 0.5129,
    "mean_conf": 56.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "60-70",
    "n": 461,
    "hit_rate": 0.5054,
    "mean_conf": 62.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "70-80",
    "n": 8,
    "hit_rate": 0.375,
    "mean_conf": 70.8,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "grok-4.6",
    "bucket": "50-60",
    "n": 216,
    "hit_rate": 0.4907,
    "mean_conf": 57.4,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "60-70",
    "n": 412,
    "hit_rate": 0.5,
    "mean_conf": 61.8,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "70-80",
    "n": 1,
    "hit_rate": 0.0,
    "mean_conf": 70,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "grok-4.6",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "50-60",
    "n": 288,
    "hit_rate": 0.4722,
    "mean_conf": 57.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "60-70",
    "n": 622,
    "hit_rate": 0.492,
    "mean_conf": 62.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "70-80",
    "n": 26,
    "hit_rate": 0.4231,
    "mean_conf": 71.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "50-60",
    "n": 6,
    "hit_rate": 0.3333,
    "mean_conf": 55,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "60-70",
    "n": 530,
    "hit_rate": 0.5321,
    "mean_conf": 63.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "70-80",
    "n": 342,
    "hit_rate": 0.5292,
    "mean_conf": 72.5,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "80-100",
    "n": 5,
    "hit_rate": 0.4,
    "mean_conf": 80,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "50-60",
    "n": 4,
    "hit_rate": 0.75,
    "mean_conf": 58.5,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "60-70",
    "n": 547,
    "hit_rate": 0.5192,
    "mean_conf": 65.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "70-80",
    "n": 267,
    "hit_rate": 0.4494,
    "mean_conf": 73.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "80-100",
    "n": 3,
    "hit_rate": 0.3333,
    "mean_conf": 82,
    "insufficient": true,
    "no_data": false
   }
  ],
  "sub50": [
   {
    "model": "claude-fable-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "claude-opus-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "qwen-3.8-max",
    "n": 1,
    "hit_rate": 1.0
   },
   {
    "model": "grok-4.6",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "deepseek-v4-pro",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gemini-3.1-pro",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gpt-5.6-sol",
    "n": 0,
    "hit_rate": null
   }
  ],
  "by_fh": [
   {
    "model": "claude-fable-5",
    "fh": "1h",
    "n": 496,
    "hit_rate": 0.5323,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "4h",
    "n": 253,
    "hit_rate": 0.5455,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "1d",
    "n": 50,
    "hit_rate": 0.32,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1h",
    "n": 422,
    "hit_rate": 0.5118,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "4h",
    "n": 244,
    "hit_rate": 0.5615,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1d",
    "n": 45,
    "hit_rate": 0.3111,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1h",
    "n": 505,
    "hit_rate": 0.5188,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "4h",
    "n": 269,
    "hit_rate": 0.5167,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1d",
    "n": 45,
    "hit_rate": 0.3333,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "1h",
    "n": 385,
    "hit_rate": 0.4909,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "4h",
    "n": 205,
    "hit_rate": 0.5317,
    "flag": null
   },
   {
    "model": "grok-4.6",
    "fh": "1d",
    "n": 39,
    "hit_rate": 0.359,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1h",
    "n": 586,
    "hit_rate": 0.4881,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "4h",
    "n": 297,
    "hit_rate": 0.5051,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1d",
    "n": 53,
    "hit_rate": 0.3208,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1h",
    "n": 561,
    "hit_rate": 0.5383,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "4h",
    "n": 272,
    "hit_rate": 0.5294,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1d",
    "n": 50,
    "hit_rate": 0.42,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1h",
    "n": 505,
    "hit_rate": 0.497,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "4h",
    "n": 266,
    "hit_rate": 0.5263,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1d",
    "n": 50,
    "hit_rate": 0.34,
    "flag": null
   }
  ],
  "trading": [
   {
    "model": "claude-opus-5",
    "n_trades": 711,
    "wr": 0.4346,
    "pnl_net_usd": -30.12,
    "pnl_gross_usd": 40.98,
    "max_dd_usd": -60.18
   },
   {
    "model": "claude-fable-5",
    "n_trades": 799,
    "wr": 0.4518,
    "pnl_net_usd": -31.4,
    "pnl_gross_usd": 48.5,
    "max_dd_usd": -63.4
   },
   {
    "model": "gemini-3.1-pro",
    "n_trades": 883,
    "wr": 0.4439,
    "pnl_net_usd": -34.19,
    "pnl_gross_usd": 54.11,
    "max_dd_usd": -77.86
   },
   {
    "model": "grok-4.6",
    "n_trades": 629,
    "wr": 0.4213,
    "pnl_net_usd": -44.11,
    "pnl_gross_usd": 18.79,
    "max_dd_usd": -54.55
   },
   {
    "model": "qwen-3.8-max",
    "n_trades": 819,
    "wr": 0.4335,
    "pnl_net_usd": -56.3,
    "pnl_gross_usd": 25.6,
    "max_dd_usd": -63.65
   },
   {
    "model": "gpt-5.6-sol",
    "n_trades": 821,
    "wr": 0.4251,
    "pnl_net_usd": -62.22,
    "pnl_gross_usd": 19.88,
    "max_dd_usd": -92.02
   },
   {
    "model": "deepseek-v4-pro",
    "n_trades": 936,
    "wr": 0.4241,
    "pnl_net_usd": -79.39,
    "pnl_gross_usd": 14.21,
    "max_dd_usd": -80.25
   }
  ],
  "gap_week_over_week": [
   {
    "model": "claude-fable-5",
    "gap_pp_issue6": 23.6,
    "gap_pp_issue7": 9.1,
    "delta_pp": -14.5
   },
   {
    "model": "qwen-3.8-max",
    "gap_pp_issue6": 18.8,
    "gap_pp_issue7": 9.2,
    "delta_pp": -9.6
   },
   {
    "model": "claude-opus-5",
    "gap_pp_issue6": 24.5,
    "gap_pp_issue7": 9.9,
    "delta_pp": -14.6
   },
   {
    "model": "grok-4.6",
    "gap_pp_issue6": 20.5,
    "gap_pp_issue7": 10.7,
    "delta_pp": -9.8
   },
   {
    "model": "deepseek-v4-pro",
    "gap_pp_issue6": 18.4,
    "gap_pp_issue7": 12.4,
    "delta_pp": -6.0
   },
   {
    "model": "gemini-3.1-pro",
    "gap_pp_issue6": 23.6,
    "gap_pp_issue7": 14.0,
    "delta_pp": -9.6
   },
   {
    "model": "gpt-5.6-sol",
    "gap_pp_issue6": 28.4,
    "gap_pp_issue7": 18.4,
    "delta_pp": -10.0
   },
   {
    "model": "Field",
    "gap_pp_issue6": 22.5,
    "gap_pp_issue7": 12.1,
    "delta_pp": -10.4
   }
  ],
  "ok_rate_by_model": {
   "claude-fable-5": 1.0,
   "claude-opus-5": 1.0,
   "qwen-3.8-max": 0.9946,
   "grok-4.6": 0.9993,
   "deepseek-v4-pro": 1.0,
   "gemini-3.1-pro": 1.0,
   "gpt-5.6-sol": 1.0
  },
  "counters": {
   "forecasts_total": 10290,
   "ok_in_gate": 9308,
   "out_of_gate": 973,
   "out_of_gate_by_fh": {
    "1w": 488,
    "1m": 485
   },
   "invalid": 9,
   "mature": 9308,
   "mature_directional": 5598,
   "mature_sideways": 3710,
   "pending_next_issue": 0,
   "pending_by_fh": {},
   "late_closes": 0,
   "uptime": [
    {
     "fh": "1h",
     "tf": "1h",
     "slots_seen": 168,
     "slots_expected": 168
    },
    {
     "fh": "4h",
     "tf": "4h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "4h",
     "tf": "1h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "1d",
     "tf": "1d",
     "slots_seen": 7,
     "slots_expected": 7
    },
    {
     "fh": "1d",
     "tf": "4h",
     "slots_seen": 7,
     "slots_expected": 7
    }
   ],
   "regime_days": {
    "2026-09-14": {
     "move_pct": 1.79,
     "regime": "trend"
    },
    "2026-09-15": {
     "move_pct": -3.35,
     "regime": "trend"
    },
    "2026-09-16": {
     "move_pct": 0.86,
     "regime": "flat"
    },
    "2026-09-17": {
     "move_pct": 0.21,
     "regime": "flat"
    },
    "2026-09-18": {
     "move_pct": 5.82,
     "regime": "trend"
    },
    "2026-09-19": {
     "move_pct": 0.46,
     "regime": "flat"
    },
    "2026-09-20": {
     "move_pct": 0.13,
     "regime": "flat"
    }
   },
   "grok_era": {
    "ids_in_window": [
     "grok-4.6"
    ],
    "n_grok45": 0,
    "n_grok46": 1470,
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470,
    "first_grok46_slot": "2026-09-14T00:01:00+00:00",
    "last_grok45_slot": null,
    "first_grok45_slot": null,
    "flip_at": "2026-08-24T09:22:00+00:00",
    "flip_inside_window": false,
    "prev_window": {
     "n_grok45_rows": 0,
     "n_grok46_rows": 1470
    },
    "merged_model_id": "grok-4.6",
    "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "lineage_ids": [
     "grok-4.5",
     "grok-4.6"
    ],
    "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
    "lineage_warnings": [],
    "rows_remapped_a_plus_b": 0,
    "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- three windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #6) and week B (this window) are pure grok-4.6.",
    "cells_with_two_grok_rows": 0
   },
   "series_density": {
    "1d": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-14",
      "2026-09-15",
      "2026-09-16",
      "2026-09-17",
      "2026-09-18",
      "2026-09-19",
      "2026-09-20"
     ],
     "missing_days": []
    },
    "1h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 168,
     "n_rows": 5880,
     "days": [
      "2026-09-14",
      "2026-09-15",
      "2026-09-16",
      "2026-09-17",
      "2026-09-18",
      "2026-09-19",
      "2026-09-20"
     ],
     "missing_days": []
    },
    "1m": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-14",
      "2026-09-15",
      "2026-09-16",
      "2026-09-17",
      "2026-09-18",
      "2026-09-19",
      "2026-09-20"
     ],
     "missing_days": []
    },
    "1w": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-14",
      "2026-09-15",
      "2026-09-16",
      "2026-09-17",
      "2026-09-18",
      "2026-09-19",
      "2026-09-20"
     ],
     "missing_days": []
    },
    "4h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 42,
     "n_rows": 2940,
     "days": [
      "2026-09-14",
      "2026-09-15",
      "2026-09-16",
      "2026-09-17",
      "2026-09-18",
      "2026-09-19",
      "2026-09-20"
     ],
     "missing_days": []
    }
   },
   "model_roster": {
    "current": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "current_count": 7,
    "current_raw_ids_in_window": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "archived_legacy": [
     "grok-4.5",
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "archived_legacy_count": 3,
    "archived_legacy_detail": [
     {
      "model": "grok-4.5",
      "n_rows_since": 11128,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-24T09:01:00+00:00"
     },
     {
      "model": "qwen-3.7-max",
      "n_rows_since": 7479,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-05T09:01:00+00:00"
     },
     {
      "model": "claude-opus-4.8",
      "n_rows_since": 6315,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-07-30T08:01:00+00:00"
     }
    ],
    "archived_excluding_lineage_merged": [
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "tracked_total": 10,
    "canon_expected": {
     "current": 7,
     "archived_legacy": 4,
     "tracked_total": 11
    },
    "matches_canon_11_7_4": false,
    "roster_since": "2026-07-11T00:00:00+00:00",
    "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
   }
  }
 },
 "key_finding": {
  "label": "KEY FINDING · OBSERVATION (one weekly window)",
  "text": "The field's overconfidence gap moved +22.5pp -> +12.1pp week over week, 7 of 7 model lines narrowed, and field Brier went 0.2948 -> 0.2669.",
  "evidence_level": "observation"
 },
 "key_findings": [
  "**OBSERVATION -- The gap read +12.1pp this week.** Directional calls hit 50.7% against 62.8 stated mean confidence (issue #6: +22.5pp on 39.6% vs 62.1). Field Brier is 0.2669 (was 0.2948) and 0 of 7 model lines scored below 0.25, the score of an uninformative always-50% predictor. One window, descriptive.",
  "**7 of 7 models narrowed their gap.** Gap moves ran from claude-opus-5's -14.6pp (+24.5pp -> +9.9pp) to deepseek-v4-pro's -6.0pp (+18.4pp -> +12.4pp); 7 of 7 model lines narrowed week over week. claude-fable-5 is the best-calibrated line (Brier 0.2593); gemini-3.1-pro leads on hit-rate (52.9%).",
  "**The high-confidence buckets, N>=10 only.** deepseek-v4-pro 70-80 42.3% (n=26), gemini-3.1-pro 70-80 52.9% (n=342), gpt-5.6-sol 70-80 44.9% (n=267). Cells below N=10 are marked insufficient and rank nothing.",
  "**Trading, kept separate from every calibration table.** 0 of 7 lines finished net-positive; net PnL ran -$30.12 (claude-opus-5, best) to -$79.39 (deepseek-v4-pro, worst), field -$337.72 against issue #6's -$1018.93."
 ],
 "market_check": {
  "label": "MARKET CHECK · OBSERVATION (two adjacent calendar weeks)",
  "text": "OBSERVATION — mean |1d move| 1.35% -> 2.69% (+100% rel), BTC realized vol 31.4% -> 35.1% (ann., hourly); field directional accuracy 39.6% -> 50.7% (+11.2pp), 7 of 7 models improved, sim win-rate up for 7 of 7, field sim PnL -$1018.93 -> -$337.72 (adjacent calendar weeks Sep 7-13 vs Sep 14-20; descriptive, one pair of weeks, not a claim).",
  "robustness": "Robustness: raw price-sign accuracy 38.7% -> 54.2% — the same direction as the trade-based hit rule.",
  "weeks": {
   "A": {
    "days": "2026-09-07..2026-09-13",
    "slots": [
     "2026-09-07T00:00:00+00:00",
     "2026-09-14T00:00:00+00:00"
    ],
    "cutoff": "2026-09-14T16:00:00+00:00"
   },
   "B": {
    "days": "2026-09-14..2026-09-20",
    "slots": [
     "2026-09-14T00:00:00+00:00",
     "2026-09-21T00:00:00+00:00"
    ],
    "cutoff": "2026-09-21T16:00:00+00:00"
   }
  },
  "price_source": "crypto_spot/1h@:00",
  "accuracy_field": {
   "A": {
    "n": 4468,
    "hit_rate": 0.3957,
    "wilson_95": [
     0.3815,
     0.4101
    ]
   },
   "B": {
    "n": 5598,
    "hit_rate": 0.5075,
    "wilson_95": [
     0.4944,
     0.5206
    ]
   },
   "delta_pp": 11.2
  },
  "accuracy_per_model": [
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "A": {
     "n": 624,
     "hit_rate": 0.3654,
     "wilson_95": [
      0.3285,
      0.4039
     ]
    },
    "B": {
     "n": 799,
     "hit_rate": 0.5232,
     "wilson_95": [
      0.4885,
      0.5576
     ]
    },
    "delta_pp": 15.8
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "A": {
     "n": 510,
     "hit_rate": 0.3647,
     "wilson_95": [
      0.3241,
      0.4073
     ]
    },
    "B": {
     "n": 711,
     "hit_rate": 0.5162,
     "wilson_95": [
      0.4795,
      0.5527
     ]
    },
    "delta_pp": 15.1
   },
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "A": {
     "n": 673,
     "hit_rate": 0.4175,
     "wilson_95": [
      0.3808,
      0.4552
     ]
    },
    "B": {
     "n": 883,
     "hit_rate": 0.5289,
     "wilson_95": [
      0.4959,
      0.5616
     ]
    },
    "delta_pp": 11.1
   },
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "A": {
     "n": 707,
     "hit_rate": 0.4003,
     "wilson_95": [
      0.3648,
      0.4368
     ]
    },
    "B": {
     "n": 819,
     "hit_rate": 0.5079,
     "wilson_95": [
      0.4737,
      0.5421
     ]
    },
    "delta_pp": 10.8
   },
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "A": {
     "n": 518,
     "hit_rate": 0.3958,
     "wilson_95": [
      0.3546,
      0.4385
     ]
    },
    "B": {
     "n": 629,
     "hit_rate": 0.496,
     "wilson_95": [
      0.4571,
      0.535
     ]
    },
    "delta_pp": 10.0
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "A": {
     "n": 709,
     "hit_rate": 0.4006,
     "wilson_95": [
      0.3651,
      0.4371
     ]
    },
    "B": {
     "n": 821,
     "hit_rate": 0.497,
     "wilson_95": [
      0.4628,
      0.5311
     ]
    },
    "delta_pp": 9.6
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "A": {
     "n": 727,
     "hit_rate": 0.414,
     "wilson_95": [
      0.3788,
      0.4502
     ]
    },
    "B": {
     "n": 936,
     "hit_rate": 0.484,
     "wilson_95": [
      0.4521,
      0.516
     ]
    },
    "delta_pp": 7.0
   }
  ],
  "raw_sign_field": {
   "A": {
    "n_scored": 4466,
    "hit_rate": 0.3871
   },
   "B": {
    "n_scored": 5583,
    "hit_rate": 0.5416
   }
  },
  "abs_move_1d": {
   "A": 1.347,
   "B": 2.693,
   "delta_pct_rel": 99.9
  },
  "btc_realized_vol_ann_pct_hourly": {
   "A": 31.4,
   "B": 35.07
  },
  "trading_field": {
   "A": {
    "n_trades": 4468,
    "wr": 0.3212,
    "pnl_net_usd": -1018.93,
    "pnl_gross_usd": -572.13
   },
   "B": {
    "n_trades": 5598,
    "wr": 0.4337,
    "pnl_net_usd": -337.72,
    "pnl_gross_usd": 222.08
   }
  },
  "verdict": {
   "models_with_hit_improved": "7 of 7",
   "models_with_rawsign_improved": "7 of 7",
   "models_with_wr_improved": "7 of 7",
   "field_hit_delta_pp": 11.2,
   "field_pnl_net": {
    "A": -1018.93,
    "B": -337.72
   },
   "abs_move_delta": {
    "1h": {
     "A": 0.333,
     "B": 0.369,
     "delta_pct_rel": 10.8
    },
    "4h": {
     "A": 0.605,
     "B": 0.835,
     "delta_pct_rel": 38.0
    },
    "1d": {
     "A": 1.347,
     "B": 2.693,
     "delta_pct_rel": 99.9
    }
   },
   "btc_vol_ann_pct": {
    "A": 31.4,
    "B": 35.07
   },
   "note": "descriptive, two adjacent calendar weeks; hit rule = methodology v1.1 trade-based; raw_sign = price-sign robustness check (closes at :00 vs slots at :01)"
  },
  "lineage_note": "lineage splice ACTIVE: grok-4.5 rows are aggregated into grok-4.6 (flip 2026-08-24T09:22:00+00:00, three windows back, inside issue #4's window). Week A (2026-09-07..2026-09-13) is pure grok-4.6 (it reproduces the published issue #6); week B is pure grok-4.6."
 },
 "why_it_matters": "Every MarketMania forecast carries a model-stated confidence from 0 to 100 alongside its direction call. This report checks whether that number tracks reality: on a well-calibrated forecaster, calls made at 70% confidence should hit their direction about 70% of the time. With issue #7 the series has seven points on every model's gap -- enough to see movement, not enough to claim a trend: calibration and market regime move together, and durability claims belong in the Monthly series, not here.",
 "how_to_read_this": "Gap pp = mean stated confidence minus hit-rate, in percentage points; positive means overconfident. Brier is the mean squared error of the stated probability against the realised outcome, so lower is better and 0.25 is what an uninformative always-50% predictor scores. Coverage = calls scored divided by the mature calls available to that model. Confidence buckets are the models' own natural breakpoints, not an arbitrary binning; cells below N=10 are marked insufficient and never used to rank anything. Prediction and trading metrics live in separate sections and are never combined. The grok 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits three windows back (issue #4): both weeks of this issue are pure grok-4.6, with no 4.5-era rows in either window, so the grok week-over-week row is the second pure-4.6 against pure-4.6 comparison of the series.",
 "practical_implications": [
  "Stated confidence is still a ranking hint at best, not a probability: the field's gap is +12.1pp this week against +22.5pp in issue #6.",
  "Gap and regime move together. The field hit 50.7% against 39.6% in issue #6 and stated mean confidence read 62.8 against 62.1, so the gap came out at +12.1pp: a field that hits 50.7% instead of 39.6% closes or opens a confidence gap without a single stated number changing.",
  "Trading direction this week (0 of 7 lines net-positive) is a regime read, not a strategy result; the same lines printed the opposite sign inside five weeks."
 ],
 "limitations": [
  "Prediction metrics (this report) and trading metrics are kept in separate sections per methodology -- they are never combined into a single score.",
  "95% Wilson CIs shown are descriptive, not inferential: observations inside one window are dependent, so read them as a range, not a formal coverage guarantee.",
  "All 5,598 scored calls sit inside one market regime. This window ran 3 trend days of 7 (issue #6: 2 of 7); every comparison with issue #6 is a comparison across regimes as well as across weeks. Seven week-over-week points cannot separate drift from regime; no durability claim is made.",
  "Models report confidence at discrete levels, not a continuous scale; the buckets reflect those natural breakpoints. Cells below N=10 are marked insufficient.",
  "Underlying price series are reconstructed from trade entry prices (median per symbol-slot), not an independent tick feed. Series density and lineage notes (wave 7): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 14-20) daily coverage is FULL, as it was in issues #4, #5 and #6: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 488, 1M 485). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, three windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#6 window), so every grok number in this issue and in issue #6 is a pure grok-4.6 line -- the grok week-over-week row is the second pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #6 after the audit that found the raw table mixes two exchanges; issue #6 values are as published.",
  "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 21 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5 and #6, which printed 47,373 at the Sep 14 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 52,971 is an additive step under one definition; counter deltas against issue #5 and earlier remain definitional."
 ],
 "notes": {
  "series_density_and_lineage_wave7": "Series density and lineage notes (wave 7): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 14-20) daily coverage is FULL, as it was in issues #4, #5 and #6: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 488, 1M 485). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, three windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#6 window), so every grok number in this issue and in issue #6 is a pure grok-4.6 line -- the grok week-over-week row is the second pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "grok_era": {
   "ids_in_window": [
    "grok-4.6"
   ],
   "n_grok45": 0,
   "n_grok46": 1470,
   "n_grok45_rows": 0,
   "n_grok46_rows": 1470,
   "first_grok46_slot": "2026-09-14T00:01:00+00:00",
   "last_grok45_slot": null,
   "first_grok45_slot": null,
   "flip_at": "2026-08-24T09:22:00+00:00",
   "flip_inside_window": false,
   "prev_window": {
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470
   },
   "merged_model_id": "grok-4.6",
   "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
   "lineage_ids": [
    "grok-4.5",
    "grok-4.6"
   ],
   "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
   "lineage_warnings": [],
   "rows_remapped_a_plus_b": 0,
   "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- three windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #6) and week B (this window) are pure grok-4.6.",
   "cells_with_two_grok_rows": 0
  },
  "model_ids_line": [
   "claude-fable-5",
   "claude-opus-5",
   "deepseek-v4-pro",
   "gemini-3.1-pro",
   "gpt-5.6-sol",
   "grok-4.6",
   "qwen-3.8-max"
  ],
  "model_roster": {
   "current": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "current_count": 7,
   "current_raw_ids_in_window": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "archived_legacy": [
    "grok-4.5",
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "archived_legacy_count": 3,
   "archived_legacy_detail": [
    {
     "model": "grok-4.5",
     "n_rows_since": 11128,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-24T09:01:00+00:00"
    },
    {
     "model": "qwen-3.7-max",
     "n_rows_since": 7479,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-05T09:01:00+00:00"
    },
    {
     "model": "claude-opus-4.8",
     "n_rows_since": 6315,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-07-30T08:01:00+00:00"
    }
   ],
   "archived_excluding_lineage_merged": [
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "tracked_total": 10,
   "canon_expected": {
    "current": 7,
    "archived_legacy": 4,
    "tracked_total": 11
   },
   "matches_canon_11_7_4": false,
   "roster_since": "2026-07-11T00:00:00+00:00",
   "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
  },
  "series_density": {
   "1d": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-14",
     "2026-09-15",
     "2026-09-16",
     "2026-09-17",
     "2026-09-18",
     "2026-09-19",
     "2026-09-20"
    ],
    "missing_days": []
   },
   "1h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 168,
    "n_rows": 5880,
    "days": [
     "2026-09-14",
     "2026-09-15",
     "2026-09-16",
     "2026-09-17",
     "2026-09-18",
     "2026-09-19",
     "2026-09-20"
    ],
    "missing_days": []
   },
   "1m": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-14",
     "2026-09-15",
     "2026-09-16",
     "2026-09-17",
     "2026-09-18",
     "2026-09-19",
     "2026-09-20"
    ],
    "missing_days": []
   },
   "1w": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-14",
     "2026-09-15",
     "2026-09-16",
     "2026-09-17",
     "2026-09-18",
     "2026-09-19",
     "2026-09-20"
    ],
    "missing_days": []
   },
   "4h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 42,
    "n_rows": 2940,
    "days": [
     "2026-09-14",
     "2026-09-15",
     "2026-09-16",
     "2026-09-17",
     "2026-09-18",
     "2026-09-19",
     "2026-09-20"
    ],
    "missing_days": []
   }
  },
  "week_over_week_note": "Week-over-week columns compare back-to-back windows: Sep 7-13 (issue #6, and week A of this issue's alive slice) vs Sep 14-20 (this issue). \"Was\" values are the numbers published in issue #6; deltas are computed on unrounded rates.",
  "research_to_date_pin": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 21 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5 and #6, which printed 47,373 at the Sep 14 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 52,971 is an additive step under one definition; counter deltas against issue #5 and earlier remain definitional.",
  "market_state_audit": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #6 after the audit that found the raw table mixes two exchanges; issue #6 values are as published.",
  "engine": "Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
  "report_template": "PDF built with the canonical mm_report.py template module; preview card with report_preview.render_preview"
 },
 "platform_note_tpsl": "**Context, not a finding.** Since Aug 13 the public sandbox can trade with model-specific TP/SL multipliers learned from this same weekly history (v1); since Aug 19, v2 adds per-ticker and confidence-bucket (50-70 / 70-100) resolution. That feature consumes calibration history; it does not feed back into any table in this report, which measures stated confidence vs. direction-hit only. No effect size is claimed for v2. The toggle lives at marketmania.ai/indices.",
 "living_series_note": "**Issue #7.** Weekly Calibration is a living series. Issue #6's open questions -- does the gap keep tracking the field base one-for-one or hold a level of its own, and does qwen-3.8-max hold the best-Brier line for a third issue? -- read this issue as: field gap +22.5pp -> +12.1pp (7 of 7 lines narrowed) on a field base that moved +11.2pp, and the best-Brier line changed hands: claude-fable-5 at 0.2593 (issue #6: qwen-3.8-max at 0.2794). Also on the record: gpt-5.6-sol's 70-80 bucket read 44.9% (n=267) against its own 60-70 at 51.9% (n=547). The grok row is pure grok-4.6 this issue and was pure grok-4.6 in issue #6; the 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits three windows back. Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
 "testing_next": [
  "Next issue: the gap moved -10.4pp while the field base moved +11.2pp -- the second issue running in which the two move nearly one-for-one in opposite directions; does that hold for a third?",
  "The best-Brier line changed hands to claude-fable-5 -- does it hold for a second issue, or change hands again?",
  "Monthly series: is the overconfidence gap stable across market regimes (trend vs flat) at monthly n?"
 ],
 "related_research": [
  {
   "title": "Consensus Watch #7",
   "url": "https://marketmania.ai/research/reports/consensus-watch-2026-09-14.pdf"
  },
  {
   "title": "Weekly Model Watch #7",
   "url": "https://marketmania.ai/research/reports/model-watch-2026-09-14.pdf"
  },
  {
   "title": "Weekly Calibration #6",
   "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-07.pdf"
  }
 ],
 "related_research_note": "The three weekly reports publish together as one issue each week; each links straight to the others' PDF and to its own previous issue. Direct links are the posting rule from wave 2 on.",
 "citation": {
  "bibtex_key": "mm_calibration_2026w38",
  "title": "Weekly Calibration #7: confidence vs. direction-hit, Sep 14-20 2026",
  "author": "MarketMania Research",
  "year": 2026,
  "month": "September",
  "day": 22,
  "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-14.pdf",
  "note": "Methodology v1.1; window Sep 14-20, 2026 UTC; source weekly_metrics_2026-09-14.json"
 },
 "research_to_date": {
  "this_report": {
   "scored_observations": 5598
  },
  "platform": {
   "as_of_cutoff": "2026-09-21",
   "resolved_forecasts": 52971,
   "resolved_forecasts_definition": "directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, at the issue cutoff",
   "resolved_forecasts_since": "2026-07-11",
   "models_tracked": 10,
   "models_tracked_detail": "7 models tracked (current line-up; earlier versions folded into their successors' lineage)",
   "assets": 5,
   "forecast_horizons": 5,
   "cadence": "hourly",
   "published_research_reports": 32
  },
  "source": "weekly_metrics_2026-09-14.json",
  "control_reproduces_wave3_pin": true,
  "pin_note": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 21 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5 and #6, which printed 47,373 at the Sep 14 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 52,971 is an additive step under one definition; counter deltas against issue #5 and earlier remain definitional."
 },
 "site_entry": {
  "slug": "weekly-calibration-2026-09-14",
  "series": "weekly",
  "seriesLabel": "Weekly · Calibration",
  "date": "September 22, 2026",
  "title": "Weekly Calibration #7",
  "summary": "5,598 directional calls asked whether stated confidence tracks reality. The field hit 50.7% against 62.8 stated mean confidence, an overconfidence gap of +12.1pp against +22.5pp in issue #6, and 7 of 7 model lines narrowed week over week. Field Brier moved 0.2948 -> 0.2669; claude-fable-5 is the best-calibrated line at 0.2593 and gemini-3.1-pro leads on hit-rate at 52.9%. Trading, kept in its own section, finished with 0 of 7 lines net-positive as field accuracy moved 39.6% -> 50.7% (+11.2pp).",
  "stats": [
   "5,598 directional calls",
   "gap +12.1pp (was +22.5pp)",
   "0 of 7 net-positive"
  ],
  "manifest_key_finding": "The overconfidence gap read +12.1pp against +22.5pp in issue #6, with 7 of 7 model lines narrowing and field Brier moving 0.2948 -> 0.2669; trading, kept separate, finished 0 of 7 lines net-positive.",
  "files": {
   "pdf": "/research/reports/weekly-calibration-2026-09-14.pdf",
   "md": "/research/reports/weekly-calibration-2026-09-14.md",
   "json": "/research/reports/weekly-calibration-2026-09-14.json"
  }
 },
 "social": {
  "telegram": {
   "photo_caption_html": "🎯 <b>Weekly Calibration #7</b> — weekly series (window Sep 14–20)\n\nResearch question: when a model says 70, does it hit 70% of the time?\n\n5,598 directional calls across 7 model lines:\n• Field hit <b>50.7%</b> vs <b>62.8</b> stated — an overconfidence gap of <b>+12.1pp</b> (issue #6: +22.5pp)\n• Gap week over week: <b>7 of 7 lines narrowed</b>; field Brier 0.2948 -> 0.2669\n• Best-calibrated line: <b>claude-fable-5</b> (Brier 0.2593); hit-rate leader gemini-3.1-pro at 52.9%\n• Market check: field accuracy 39.6% -> 50.7% (+11.2pp); trading kept separate — <b>0 of 7 net-positive</b>\n\nNote: grok is a pure 4.6 line in both weeks of this issue; the 4.5 -> 4.6 flip (Aug 24 09:22 UTC) sits three windows back\n\n📄 <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-14.pdf\">Full report (PDF)</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "document_caption_html": "📄 Weekly Calibration #7 — full report (PDF).\nWeb copy: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-14.pdf\">weekly-calibration-2026-09-14.pdf</a> · machine-readable: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-09-14.json\">JSON</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "caption_len": 941,
   "document_caption_len": 416,
   "limit": 1024
  },
  "x": {
   "main": "Weekly Calibration #7 is out.\n\n5,598 LLM calls: 50.7% hit vs 62.8 stated confidence — an overconfidence gap of +12.1pp against +22.5pp a week earlier, with 7 of 7 lines narrowing.\n\nFull report: https://marketmania.ai/research/reports/weekly-calibration-2026-09-14.pdf\n\n#AI #Crypto #Research #Calibration",
   "main_tco_len": 253,
   "reply": "Overconfidence gap, w/w (pp):\nfable-5: 23.6 -> 9.1\nqwen-3.8: 18.8 -> 9.2\nopus-5: 24.5 -> 9.9\ngrok-4.6: 20.5 -> 10.7\ndeepseek: 18.4 -> 12.4\ngemini-3.1: 23.6 -> 14.0\ngpt-5.6: 28.4 -> 18.4\nField: +22.5pp -> +12.1pp. Methodology v1.1.\nData: https://marketmania.ai/research/reports/weekly-calibration-2026-09-14.json",
   "reply_tco_len": 260,
   "limit_tco": 280,
   "replies_max": 1
  }
 }
}
