{
 "slug": "weekly-calibration-2026-08-17",
 "series": "weekly",
 "series_label": "WEEKLY · CALIBRATION",
 "issue": 3,
 "title": "Weekly Calibration #3",
 "language": "en",
 "window_start": "2026-08-17T00:00:00+00:00",
 "window_end": "2026-08-24T00:00:00+00:00",
 "window_text": "Aug 17-23, 2026 UTC",
 "cutoff": "2026-08-24T16:00:00+00:00",
 "generated_at": "2026-08-26T07:00:00+00:00",
 "methodology": "v1.1 (2026-08-10)",
 "methodology_hash": "e66c7e8c864a2233",
 "source": "weekly_metrics_2026-08-17.json",
 "hit_rule": "direction: exit tp1/tp2 -> hit, sl -> miss, expiry -> sign of gross pnl",
 "bibtex_key": "mm_calibration_2026w34",
 "market_state": {
  "panel": "market_state_6p2",
  "week": {
   "start": "2026-08-17",
   "end": "2026-08-24"
  },
  "prev_week": {
   "start": "2026-08-10",
   "end": "2026-08-17"
  },
  "current": {
   "btc_net_pct": 23.58,
   "realized_vol_ann_pct": 71.1,
   "volume_top5_usd_bn": 25.97,
   "volume_top5_wow_pct": 235.0,
   "avg_pairwise_corr": 0.81
  },
  "previous_as_published_issue2": {
   "btc_net_pct": -3.08,
   "realized_vol_ann_pct": 10.0,
   "volume_top5_usd_bn": 7.7521,
   "avg_pairwise_corr": 0.52
  },
  "snapshot_line": "BTC net +23.58% (prior -3.08%) - ann vol 71.1% (was 10.0%) - TOP5 volume $25.97B, +235% w/w - pairwise corr 0.81 (was 0.52)",
  "exchange": "binance",
  "source": "owner-verified single-exchange (binance) daily candle recompute, 2026-08-26; prior-week values as published in issue #2",
  "audit_note": "Market-state row is computed on a single-exchange (binance) daily candle series this issue, after an audit found the raw table mixes two exchanges; issue #2 row is as published.",
  "regime_days": {
   "2026-08-17": {
    "move_pct": 2.55,
    "regime": "trend"
   },
   "2026-08-18": {
    "move_pct": 0.21,
    "regime": "flat"
   },
   "2026-08-19": {
    "move_pct": 7.82,
    "regime": "trend"
   },
   "2026-08-20": {
    "move_pct": 5.07,
    "regime": "trend"
   },
   "2026-08-21": {
    "move_pct": 7.26,
    "regime": "trend"
   },
   "2026-08-22": {
    "move_pct": -1.67,
    "regime": "trend"
   },
   "2026-08-23": {
    "move_pct": 0.69,
    "regime": "flat"
   }
  },
  "alive_slice_context": {
   "abs_move_1d_pct": {
    "A": 0.734,
    "B": 4.283
   },
   "btc_realized_vol_ann_pct_hourly": {
    "A": 19.26,
    "B": 58.72
   }
  }
 },
 "tables": {
  "calibration_main": [
   {
    "model": "claude-opus-5",
    "legacy": false,
    "is_field": false,
    "n": 672,
    "coverage": 0.5053,
    "hit_rate": 0.5908,
    "mean_conf": 61.2,
    "gap_pp": 2.2,
    "brier": 0.2419,
    "wilson_95": {
     "lo": 0.5532,
     "hi": 0.6273
    }
   },
   {
    "model": "claude-fable-5",
    "legacy": false,
    "is_field": false,
    "n": 779,
    "coverage": 0.5857,
    "hit_rate": 0.5648,
    "mean_conf": 61.2,
    "gap_pp": 4.7,
    "brier": 0.2464,
    "wilson_95": {
     "lo": 0.5298,
     "hi": 0.5992
    }
   },
   {
    "model": "qwen-3.8-max",
    "legacy": false,
    "is_field": false,
    "n": 706,
    "coverage": 0.5511,
    "hit_rate": 0.5227,
    "mean_conf": 59.7,
    "gap_pp": 7.4,
    "brier": 0.254,
    "wilson_95": {
     "lo": 0.4858,
     "hi": 0.5593
    }
   },
   {
    "model": "grok-4.5",
    "legacy": false,
    "is_field": false,
    "n": 710,
    "coverage": 0.5338,
    "hit_rate": 0.5141,
    "mean_conf": 60.8,
    "gap_pp": 9.4,
    "brier": 0.2557,
    "wilson_95": {
     "lo": 0.4773,
     "hi": 0.5507
    }
   },
   {
    "model": "deepseek-v4-pro",
    "legacy": false,
    "is_field": false,
    "n": 777,
    "coverage": 0.5842,
    "hit_rate": 0.5161,
    "mean_conf": 61.0,
    "gap_pp": 9.4,
    "brier": 0.2574,
    "wilson_95": {
     "lo": 0.481,
     "hi": 0.5511
    }
   },
   {
    "model": "gemini-3.1-pro",
    "legacy": false,
    "is_field": false,
    "n": 798,
    "coverage": 0.6005,
    "hit_rate": 0.5414,
    "mean_conf": 67.2,
    "gap_pp": 13.1,
    "brier": 0.2646,
    "wilson_95": {
     "lo": 0.5067,
     "hi": 0.5756
    }
   },
   {
    "model": "gpt-5.6-sol",
    "legacy": false,
    "is_field": false,
    "n": 688,
    "coverage": 0.5173,
    "hit_rate": 0.5363,
    "mean_conf": 67.5,
    "gap_pp": 13.9,
    "brier": 0.271,
    "wilson_95": {
     "lo": 0.499,
     "hi": 0.5733
    }
   },
   {
    "model": "Field (all models)",
    "legacy": false,
    "is_field": true,
    "n": 5130,
    "coverage": null,
    "hit_rate": 0.5405,
    "mean_conf": 62.7,
    "gap_pp": 8.6,
    "brier": 0.2559,
    "wilson_95": null
   }
  ],
  "buckets": [
   {
    "model": "claude-fable-5",
    "bucket": "50-60",
    "n": 183,
    "hit_rate": 0.541,
    "mean_conf": 56.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "60-70",
    "n": 596,
    "hit_rate": 0.5721,
    "mean_conf": 62.6,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-fable-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-fable-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "50-60",
    "n": 126,
    "hit_rate": 0.6349,
    "mean_conf": 57.6,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "60-70",
    "n": 546,
    "hit_rate": 0.5806,
    "mean_conf": 62.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "claude-opus-5",
    "bucket": "70-80",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "claude-opus-5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "50-60",
    "n": 236,
    "hit_rate": 0.4576,
    "mean_conf": 57.0,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "60-70",
    "n": 524,
    "hit_rate": 0.5382,
    "mean_conf": 62.5,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "70-80",
    "n": 17,
    "hit_rate": 0.6471,
    "mean_conf": 70.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "deepseek-v4-pro",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "50-60",
    "n": 2,
    "hit_rate": 0.5,
    "mean_conf": 55,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "60-70",
    "n": 476,
    "hit_rate": 0.5273,
    "mean_conf": 63.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "70-80",
    "n": 307,
    "hit_rate": 0.5505,
    "mean_conf": 72.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gemini-3.1-pro",
    "bucket": "80-100",
    "n": 13,
    "hit_rate": 0.8462,
    "mean_conf": 81.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "50-60",
    "n": 2,
    "hit_rate": 0.5,
    "mean_conf": 58,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "60-70",
    "n": 501,
    "hit_rate": 0.5569,
    "mean_conf": 65.6,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "70-80",
    "n": 185,
    "hit_rate": 0.4811,
    "mean_conf": 72.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "gpt-5.6-sol",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "grok-4.5",
    "bucket": "50-60",
    "n": 315,
    "hit_rate": 0.454,
    "mean_conf": 57.7,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.5",
    "bucket": "60-70",
    "n": 389,
    "hit_rate": 0.5578,
    "mean_conf": 63.2,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "grok-4.5",
    "bucket": "70-80",
    "n": 6,
    "hit_rate": 0.8333,
    "mean_conf": 71,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "grok-4.5",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "50-60",
    "n": 318,
    "hit_rate": 0.5094,
    "mean_conf": 56.9,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "60-70",
    "n": 379,
    "hit_rate": 0.5435,
    "mean_conf": 62.1,
    "insufficient": false,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "70-80",
    "n": 3,
    "hit_rate": 0.3333,
    "mean_conf": 70,
    "insufficient": true,
    "no_data": false
   },
   {
    "model": "qwen-3.8-max",
    "bucket": "80-100",
    "n": 0,
    "hit_rate": null,
    "mean_conf": null,
    "insufficient": false,
    "no_data": true
   }
  ],
  "sub50": [
   {
    "model": "claude-fable-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "claude-opus-5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "deepseek-v4-pro",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gemini-3.1-pro",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "gpt-5.6-sol",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "grok-4.5",
    "n": 0,
    "hit_rate": null
   },
   {
    "model": "qwen-3.8-max",
    "n": 6,
    "hit_rate": 0.0
   }
  ],
  "by_fh": [
   {
    "model": "claude-fable-5",
    "fh": "1h",
    "n": 472,
    "hit_rate": 0.5191,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "4h",
    "n": 254,
    "hit_rate": 0.6102,
    "flag": null
   },
   {
    "model": "claude-fable-5",
    "fh": "1d",
    "n": 53,
    "hit_rate": 0.7547,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1h",
    "n": 385,
    "hit_rate": 0.5299,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "4h",
    "n": 236,
    "hit_rate": 0.6525,
    "flag": null
   },
   {
    "model": "claude-opus-5",
    "fh": "1d",
    "n": 51,
    "hit_rate": 0.7647,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1h",
    "n": 463,
    "hit_rate": 0.4795,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "4h",
    "n": 264,
    "hit_rate": 0.5606,
    "flag": null
   },
   {
    "model": "deepseek-v4-pro",
    "fh": "1d",
    "n": 50,
    "hit_rate": 0.62,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1h",
    "n": 492,
    "hit_rate": 0.4919,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "4h",
    "n": 257,
    "hit_rate": 0.5953,
    "flag": null
   },
   {
    "model": "gemini-3.1-pro",
    "fh": "1d",
    "n": 49,
    "hit_rate": 0.7551,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1h",
    "n": 420,
    "hit_rate": 0.4881,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "4h",
    "n": 221,
    "hit_rate": 0.6199,
    "flag": null
   },
   {
    "model": "gpt-5.6-sol",
    "fh": "1d",
    "n": 47,
    "hit_rate": 0.5745,
    "flag": null
   },
   {
    "model": "grok-4.5",
    "fh": "1h",
    "n": 425,
    "hit_rate": 0.4753,
    "flag": null
   },
   {
    "model": "grok-4.5",
    "fh": "4h",
    "n": 242,
    "hit_rate": 0.562,
    "flag": null
   },
   {
    "model": "grok-4.5",
    "fh": "1d",
    "n": 43,
    "hit_rate": 0.6279,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1h",
    "n": 432,
    "hit_rate": 0.4745,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "4h",
    "n": 219,
    "hit_rate": 0.5845,
    "flag": null
   },
   {
    "model": "qwen-3.8-max",
    "fh": "1d",
    "n": 55,
    "hit_rate": 0.6545,
    "flag": null
   }
  ],
  "trading": [
   {
    "model": "claude-opus-5",
    "n_trades": 672,
    "wr": 0.5074,
    "pnl_net_usd": 197.44,
    "pnl_gross_usd": 264.64,
    "max_dd_usd": -55.65
   },
   {
    "model": "claude-fable-5",
    "n_trades": 779,
    "wr": 0.484,
    "pnl_net_usd": 180.33,
    "pnl_gross_usd": 258.23,
    "max_dd_usd": -57.23
   },
   {
    "model": "gemini-3.1-pro",
    "n_trades": 798,
    "wr": 0.4687,
    "pnl_net_usd": 168.37,
    "pnl_gross_usd": 248.17,
    "max_dd_usd": -84.8
   },
   {
    "model": "qwen-3.8-max",
    "n_trades": 706,
    "wr": 0.4448,
    "pnl_net_usd": 111.62,
    "pnl_gross_usd": 182.22,
    "max_dd_usd": -39.89
   },
   {
    "model": "gpt-5.6-sol",
    "n_trades": 688,
    "wr": 0.4477,
    "pnl_net_usd": 111.61,
    "pnl_gross_usd": 180.41,
    "max_dd_usd": -75.04
   },
   {
    "model": "deepseek-v4-pro",
    "n_trades": 777,
    "wr": 0.462,
    "pnl_net_usd": 65.84,
    "pnl_gross_usd": 143.54,
    "max_dd_usd": -45.0
   },
   {
    "model": "grok-4.5",
    "n_trades": 710,
    "wr": 0.4225,
    "pnl_net_usd": 54.16,
    "pnl_gross_usd": 125.16,
    "max_dd_usd": -57.04
   }
  ],
  "gap_week_over_week": [
   {
    "model": "claude-opus-5",
    "gap_pp_issue2": 19.2,
    "gap_pp_issue3": 2.2,
    "delta_pp": -17.0
   },
   {
    "model": "claude-fable-5",
    "gap_pp_issue2": 18.1,
    "gap_pp_issue3": 4.7,
    "delta_pp": -13.4
   },
   {
    "model": "qwen-3.8-max",
    "gap_pp_issue2": 14.1,
    "gap_pp_issue3": 7.4,
    "delta_pp": -6.7
   },
   {
    "model": "deepseek-v4-pro",
    "gap_pp_issue2": 18.2,
    "gap_pp_issue3": 9.4,
    "delta_pp": -8.8
   },
   {
    "model": "grok-4.5",
    "gap_pp_issue2": 15.8,
    "gap_pp_issue3": 9.4,
    "delta_pp": -6.4
   },
   {
    "model": "gemini-3.1-pro",
    "gap_pp_issue2": 18.4,
    "gap_pp_issue3": 13.1,
    "delta_pp": -5.3
   },
   {
    "model": "gpt-5.6-sol",
    "gap_pp_issue2": 24.0,
    "gap_pp_issue3": 13.9,
    "delta_pp": -10.1
   },
   {
    "model": "Field (all models)",
    "gap_pp_issue2": 18.3,
    "gap_pp_issue3": 8.6,
    "delta_pp": -9.7
   }
  ],
  "ok_rate_by_model": {
   "claude-opus-5": 1.0,
   "claude-fable-5": 1.0,
   "qwen-3.8-max": 0.962,
   "grok-4.5": 1.0,
   "deepseek-v4-pro": 1.0,
   "gemini-3.1-pro": 0.9993,
   "gpt-5.6-sol": 1.0
  },
  "counters": {
   "forecasts_total": 9590,
   "ok_in_gate": 9260,
   "invalid": 53,
   "out_of_gate": 277,
   "out_of_gate_by_fh": {
    "1w": 209,
    "1m": 68
   },
   "mature": 9260,
   "mature_directional": 5130,
   "mature_sideways": 4130,
   "pending_next_issue": 0,
   "late_closes": 0,
   "uptime": [
    {
     "fh": "1h",
     "tf": "1h",
     "slots_seen": 168,
     "slots_expected": 168
    },
    {
     "fh": "4h",
     "tf": "4h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "4h",
     "tf": "1h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "1d",
     "tf": "1d",
     "slots_seen": 7,
     "slots_expected": 7
    },
    {
     "fh": "1d",
     "tf": "4h",
     "slots_seen": 7,
     "slots_expected": 7
    }
   ]
  }
 },
 "key_finding": {
  "label": "KEY FINDING · OBSERVATION (one weekly window)",
  "text": "The overconfidence gap halved -- field +18.3pp -> +8.6pp, with all 7 models narrowing for a second straight week -- and high confidence finally ranked something: gemini-3.1-pro's 70-80 bucket held at 55.1% on doubled volume while its 80-100 bucket opened at 84.6%.",
  "evidence_level": "observation"
 },
 "key_findings": [
  "OBSERVATION -- The gap halved in a live tape. Directional calls hit 54.1% against 62.7 stated mean confidence -- a +8.6pp overconfidence gap (issue #2: +18.3pp). Field Brier improved to 0.2559 (was 0.2836) and 2 of 7 models scored below 0.25 for the first time.",
  "Every model narrowed its gap again -- second straight field-wide narrowing: from claude-opus-5's -17.0pp move (19.2 -> 2.2) to gemini-3.1-pro's -5.3pp (18.4 -> 13.1).",
  "gemini-3.1-pro's 70-80 bucket held at 55.1% on n=307 (issue #2: 50.7% on n=152); its 80-100 bucket opened at 84.6% (n=13, insufficient); deepseek-v4-pro un-inverted 26.3% -> 64.7% (n=17, thin); gpt-5.6-sol entered 70-80 heavy at 48.1% (n=185).",
  "Trading flipped with the tape: all 7 models net-positive AND gross-positive (issue #2: all 7 negative on both); net PnL +$197.44 to +$54.16, field +$889.37 vs -$527.07."
 ],
 "market_check": {
  "label": "MARKET CHECK · OBSERVATION (two adjacent calendar weeks)",
  "text": "OBSERVATION — the tape sped up and accuracy followed: mean |1d move| 0.73% -> 4.28% (483% rel), BTC realized vol 19.3% -> 58.7% (ann., hourly); field directional accuracy 43.8% -> 54.1% (+10.3pp), 7 of 7 models improved, sim win-rate up for 7 of 7, field sim PnL -$527 -> +$889 (adjacent calendar weeks Aug 10-16 vs Aug 17-23; descriptive, one pair of weeks, not a claim).",
  "robustness": "Robustness: raw price-sign accuracy 40.1% -> 53.5% — consistent with the trade-based hit rule.",
  "weeks": {
   "A": {
    "days": "2026-08-10..2026-08-16",
    "slots": [
     "2026-08-10T00:00:00+00:00",
     "2026-08-17T00:00:00+00:00"
    ],
    "cutoff": "2026-08-17T16:00:00+00:00"
   },
   "B": {
    "days": "2026-08-17..2026-08-23",
    "slots": [
     "2026-08-17T00:00:00+00:00",
     "2026-08-24T00:00:00+00:00"
    ],
    "cutoff": "2026-08-24T16:00:00+00:00"
   }
  },
  "price_source": "crypto_spot/1h@:00",
  "accuracy_field": {
   "A": {
    "n": 3670,
    "hit_rate": 0.4379,
    "wilson_95": [
     0.4219,
     0.454
    ]
   },
   "B": {
    "n": 5130,
    "hit_rate": 0.5405,
    "wilson_95": [
     0.5269,
     0.5541
    ]
   },
   "delta_pp": 10.3
  },
  "accuracy_per_model": [
   {
    "model": "claude-opus-5",
    "A": {
     "n": 361,
     "hit_rate": 0.4155,
     "wilson_95": [
      0.3658,
      0.467
     ]
    },
    "B": {
     "n": 672,
     "hit_rate": 0.5908,
     "wilson_95": [
      0.5532,
      0.6273
     ]
    },
    "delta_pp": 17.5
   },
   {
    "model": "claude-fable-5",
    "A": {
     "n": 472,
     "hit_rate": 0.4174,
     "wilson_95": [
      0.3737,
      0.4624
     ]
    },
    "B": {
     "n": 779,
     "hit_rate": 0.5648,
     "wilson_95": [
      0.5298,
      0.5992
     ]
    },
    "delta_pp": 14.7
   },
   {
    "model": "gpt-5.6-sol",
    "A": {
     "n": 636,
     "hit_rate": 0.4418,
     "wilson_95": [
      0.4037,
      0.4807
     ]
    },
    "B": {
     "n": 688,
     "hit_rate": 0.5363,
     "wilson_95": [
      0.499,
      0.5733
     ]
    },
    "delta_pp": 9.4
   },
   {
    "model": "deepseek-v4-pro",
    "A": {
     "n": 454,
     "hit_rate": 0.4251,
     "wilson_95": [
      0.3805,
      0.471
     ]
    },
    "B": {
     "n": 777,
     "hit_rate": 0.5161,
     "wilson_95": [
      0.481,
      0.5511
     ]
    },
    "delta_pp": 9.1
   },
   {
    "model": "qwen-3.8-max",
    "A": {
     "n": 572,
     "hit_rate": 0.4406,
     "wilson_95": [
      0.4004,
      0.4815
     ]
    },
    "B": {
     "n": 706,
     "hit_rate": 0.5227,
     "wilson_95": [
      0.4858,
      0.5593
     ]
    },
    "delta_pp": 8.2
   },
   {
    "model": "grok-4.5",
    "A": {
     "n": 620,
     "hit_rate": 0.4403,
     "wilson_95": [
      0.4017,
      0.4796
     ]
    },
    "B": {
     "n": 710,
     "hit_rate": 0.5141,
     "wilson_95": [
      0.4773,
      0.5507
     ]
    },
    "delta_pp": 7.4
   },
   {
    "model": "gemini-3.1-pro",
    "A": {
     "n": 555,
     "hit_rate": 0.4703,
     "wilson_95": [
      0.4291,
      0.5119
     ]
    },
    "B": {
     "n": 798,
     "hit_rate": 0.5414,
     "wilson_95": [
      0.5067,
      0.5756
     ]
    },
    "delta_pp": 7.1
   }
  ],
  "raw_sign_field": {
   "A": {
    "n_scored": 3632,
    "hit_rate": 0.4009
   },
   "B": {
    "n_scored": 5118,
    "hit_rate": 0.5348
   }
  },
  "abs_move_1d": {
   "A": 0.734,
   "B": 4.283,
   "delta_pct_rel": 483.5
  },
  "btc_realized_vol_ann_pct_hourly": {
   "A": 19.26,
   "B": 58.72
  },
  "trading_field": {
   "A": {
    "n_trades": 3670,
    "wr": 0.2965,
    "pnl_net_usd": -527.07,
    "pnl_gross_usd": -160.07
   },
   "B": {
    "n_trades": 5130,
    "wr": 0.4626,
    "pnl_net_usd": 889.37,
    "pnl_gross_usd": 1402.37
   }
  },
  "verdict": {
   "models_with_hit_improved": "7 of 7",
   "models_with_rawsign_improved": "7 of 7",
   "models_with_wr_improved": "7 of 7",
   "field_hit_delta_pp": 10.3,
   "field_pnl_net": {
    "A": -527.07,
    "B": 889.37
   },
   "abs_move_delta": {
    "1h": {
     "A": 0.181,
     "B": 0.541,
     "delta_pct_rel": 198.9
    },
    "4h": {
     "A": 0.367,
     "B": 1.314,
     "delta_pct_rel": 258.0
    },
    "1d": {
     "A": 0.734,
     "B": 4.283,
     "delta_pct_rel": 483.5
    }
   },
   "btc_vol_ann_pct": {
    "A": 19.26,
    "B": 58.72
   },
   "note": "descriptive, two adjacent calendar weeks; hit rule = methodology v1.1 trade-based; raw_sign = price-sign robustness check (closes at :00 vs slots at :01)"
  },
  "lineage_note": "no lineage stitching needed: the grok 4.5->4.6 flip (2026-08-24 09:22 UTC) is after this window -- weeks A and B are pure grok-4.5 (verified: 0 grok-4.6 rows in the pull)"
 },
 "why_it_matters": "Every MarketMania forecast carries a model-stated confidence from 0 to 100 alongside its direction call. This report checks whether that number tracks reality: on a well-calibrated forecaster, calls made at 70% confidence should hit their direction about 70% of the time. With issue #3 the series has three points on every model's gap -- enough to see a direction, not enough to claim one: this window was also the first live-tape week of the series, and calibration and market regime move together. Durability claims belong in the Monthly series, not here.",
 "how_to_read_this": "Gap pp = mean stated confidence minus hit-rate, in percentage points; positive means overconfident. Brier is the mean squared error of the stated probability against the realised outcome, so lower is better and 0.25 is what an uninformative always-50% predictor scores. Coverage = calls scored divided by the mature calls available to that model. Confidence buckets are the models' own natural breakpoints, not an arbitrary binning; cells below N=10 are marked insufficient and never used to rank anything. Prediction and trading metrics live in separate sections and are never combined.",
 "practical_implications": [
  "Stated confidence moved from useless to weakly informative: 4 of 5 sufficient-N models out-ranked 50-60 with 60-70 this week (issue #2: 1 of 5), and the field's two heaviest 70-80 buckets split. It is still not a probability.",
  "The gap halved in the series' first live-tape week. The two facts are not separable on one week of data: a field that hits 54.1% instead of 43.8% closes a confidence gap without changing a single stated number.",
  "All 7 net-positive is a regime read, not a strategy result. The same seven were all net-negative one week earlier at the same stated confidences."
 ],
 "limitations": [
  "Prediction metrics (this report) and trading metrics are kept in separate sections per methodology -- they are never combined into a single score.",
  "95% Wilson CIs shown are descriptive, not inferential: observations inside one window are dependent, so read them as a range, not a formal coverage guarantee.",
  "All 5,130 scored calls sit inside one market regime, and this window's regime is the opposite of issue #2's: 5 of 7 days were trend days against 1 of 7 last week. Three week-over-week points cannot separate drift from regime; no durability claim is made.",
  "Models report confidence at discrete levels, not a continuous scale; the buckets reflect those natural breakpoints. Cells below N=10 are marked insufficient.",
  "Underlying price series are reconstructed from trade entry prices (median per symbol-slot), not an independent tick feed. Series density and lineage notes (wave 3): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Aug 17-23) daily coverage is PARTIAL by design: the 1w series has the Mon Aug 17 anchor plus daily slots on Aug 22-23 only, and the 1M series has a daily slot on Aug 23 only; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 209, 1M 68). (2) After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "Market-state row is computed on a single-exchange (binance) daily candle series this issue, after an audit found the raw table mixes two exchanges; issue #2 row is as published.",
  "Research-to-date counter pinned from this issue: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Aug 24 16:00 UTC). Issue #2 printed 23,458 under an earlier, unpinned definition; treat cross-issue counter deltas across the pin as definitional, not additive."
 ],
 "notes": {
  "series_density_and_lineage_wave3": "Series density and lineage notes (wave 3): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Aug 17-23) daily coverage is PARTIAL by design: the 1w series has the Mon Aug 17 anchor plus daily slots on Aug 22-23 only, and the 1M series has a daily slot on Aug 23 only; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 209, 1M 68). (2) After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "grok_era": {
   "ids_in_window": [
    "grok-4.5"
   ],
   "n_grok45": 1370,
   "n_grok46": 0,
   "note": "After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5"
  },
  "model_ids_line": [
   "claude-fable-5",
   "claude-opus-5",
   "deepseek-v4-pro",
   "gemini-3.1-pro",
   "gpt-5.6-sol",
   "grok-4.5",
   "qwen-3.8-max"
  ],
  "week_over_week_note": "Week-over-week columns compare back-to-back windows: Aug 10-16 (issue #2, and week A of this issue's alive slice) vs Aug 17-23 (this issue). \"Was\" values are the numbers published in issue #2; deltas are computed on unrounded rates.",
  "research_to_date_pin": "Research-to-date counter pinned from this issue: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Aug 24 16:00 UTC). Issue #2 printed 23,458 under an earlier, unpinned definition; treat cross-issue counter deltas across the pin as definitional, not additive.",
  "market_state_audit": "Market-state row is computed on a single-exchange (binance) daily candle series this issue, after an audit found the raw table mixes two exchanges; issue #2 row is as published.",
  "engine": "Engine 1.1 has powered the sandbox since Aug 18, i.e. from day 2 of this window; no cross-engine PnL comparisons are claimed.",
  "report_template": "PDF built with the canonical mm_report.py template module; preview card with report_preview.render_preview"
 },
 "platform_note_tpsl": "Since Aug 13 the public sandbox can trade with model-specific TP/SL multipliers learned from this same weekly history (v1); since Aug 19, v2 adds per-ticker and confidence-bucket (50-70 / 70-100) resolution. That feature consumes calibration history; it does not feed back into any table in this report. No effect size is claimed for v2 -- it went live inside this window. Context, not a finding.",
 "living_series_note": "Issue #3. Weekly Calibration is a living series. Issue #2's open questions -- does gemini-3.1-pro's 70-80 bucket keep outperforming at n>150, does deepseek-v4-pro's inversion survive a third week, and does the field-wide gap keep narrowing? -- closed 'yes' (55.1% on n=307), 'no' (26.3% -> 64.7% on a thin n=17) and 'yes' (all 7 narrowed again, field 18.3 -> 8.6). After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5. Engine 1.1 has powered the sandbox since Aug 18, i.e. from day 2 of this window; no cross-engine PnL comparisons are claimed.",
 "testing_next": [
  "Next issue: does the halved gap survive a flat week, or was +8.6pp the tape rather than the models?",
  "Does gpt-5.6-sol's 70-80 bucket (48.1% on n=185) stay below its own 60-70 -- the field's new inversion candidate?",
  "Monthly test (Sep 2): is the overconfidence gap stable across market regimes (trend vs flat) at monthly n?"
 ],
 "related_research": [
  {
   "title": "Consensus Watch #3",
   "url": "https://marketmania.ai/research/reports/consensus-watch-2026-08-17.pdf"
  },
  {
   "title": "Weekly Model Watch #3",
   "url": "https://marketmania.ai/research/reports/model-watch-2026-08-17.pdf"
  },
  {
   "title": "Weekly Calibration #2",
   "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-08-10.pdf"
  }
 ],
 "related_research_note": "The three weekly reports publish together as one issue each week; each links straight to the others' PDF and to its own previous issue. Direct links are the posting rule from wave 2 on.",
 "citation": {
  "bibtex_key": "mm_calibration_2026w34",
  "title": "Weekly Calibration #3: confidence vs. direction-hit, Aug 17-23 2026",
  "author": "MarketMania Research",
  "year": 2026,
  "month": "August",
  "day": 26,
  "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-08-17.pdf",
  "note": "Methodology v1.1, hash e66c7e8c864a2233; source weekly_metrics_2026-08-17.json"
 },
 "research_to_date": {
  "this_report": {
   "scored_observations": 5130
  },
  "platform": {
   "as_of_cutoff": "2026-08-24",
   "resolved_forecasts": 33809,
   "resolved_forecasts_definition": "directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, at the issue cutoff",
   "resolved_forecasts_since": "2026-07-11",
   "models_tracked": 11,
   "models_tracked_detail": "7 current + 4 archived",
   "assets": 5,
   "forecast_horizons": 5,
   "cadence": "hourly",
   "published_research_reports": 12
  },
  "source": "weekly_metrics_2026-08-17.json",
  "pin_note": "Research-to-date counter pinned from this issue: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Aug 24 16:00 UTC). Issue #2 printed 23,458 under an earlier, unpinned definition; treat cross-issue counter deltas across the pin as definitional, not additive."
 },
 "site_entry": {
  "slug": "weekly-calibration-2026-08-17",
  "series": "weekly",
  "seriesLabel": "Weekly · Calibration",
  "date": "August 26, 2026",
  "title": "Weekly Calibration #3",
  "summary": "5,130 directional calls asked whether stated confidence tracks reality. The overconfidence gap halved: field +8.6pp (was +18.3), with all 7 models narrowing for a second straight week — claude-opus-5 down to +2.2pp. The gemini-3.1-pro 70-80 bucket held at 55.1% on doubled volume (n=307, was 50.7%), and trading flipped with the tape: all 7 models net-positive (was all 7 negative). Field accuracy rose 43.8% -> 54.1% (+10.3pp) as the market came alive.",
  "stats": [
   "5,130 directional calls",
   "gap +8.6pp (was +18.3)",
   "all 7 net-positive"
  ],
  "files": {
   "pdf": "/research/reports/weekly-calibration-2026-08-17.pdf",
   "md": "/research/reports/weekly-calibration-2026-08-17.md",
   "json": "/research/reports/weekly-calibration-2026-08-17.json"
  }
 },
 "social": {
  "telegram": {
   "photo_caption_html": "🎯 <b>Weekly Calibration #3</b> — weekly series (window Aug 17–23)\n\nResearch question: when a model says 70, does it hit 70% of the time?\n\n5,130 directional calls across 7 models:\n• Field hit <b>54.1%</b> vs <b>62.7</b> stated — a <b>+8.6pp</b> overconfidence gap (was +18.3): the gap halved\n• Gap week-over-week: <b>all 7 narrowed again</b> (second straight week; claude-opus-5 down to +2.2pp)\n• 70-80 bucket follow-up: gemini-3.1-pro <b>held at 55.1%</b> (n=307, was 50.7); gpt-5.6-sol entered heavy at 48.1% (n=185)\n• Market check: field accuracy 43.8% -> 54.1% (+10.3pp); trading flipped — <b>all 7 net-positive</b> (was all 7 negative)\n\n📄 <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-08-17.pdf\">Full report (PDF)</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "document_caption_html": "📄 Weekly Calibration #3 — full report (PDF).\nWeb copy: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-08-17.pdf\">weekly-calibration-2026-08-17.pdf</a> · machine-readable: <a href=\"https://marketmania.ai/research/reports/weekly-calibration-2026-08-17.json\">JSON</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #calibration #research",
   "caption_len": 874,
   "document_caption_len": 416,
   "limit": 1024
  },
  "x": {
   "main": "Weekly Calibration #3 is out.\n\n5,130 LLM calls: 54.1% hit vs 62.7 stated confidence. The gap halved to +8.6pp — all 7 models narrowed again, and all 7 turned net-positive on trading.\n\nFull report: https://marketmania.ai/research/reports/weekly-calibration-2026-08-17.pdf\n\n#AI #Crypto #Research #Calibration",
   "main_tco_len": 256,
   "reply": "Overconfidence gap, w/w (pp):\nqwen-3.8: 14.1 -> 7.4\ngrok-4.5: 15.8 -> 9.4\nfable-5: 18.1 -> 4.7\ndeepseek: 18.2 -> 9.4\ngemini-3.1: 18.4 -> 13.1\nopus-5: 19.2 -> 2.2\ngpt-5.6: 24.0 -> 13.9\nField: 18.3 -> 8.6 — all seven narrowed again. Methodology v1.1.\nData: https://marketmania.ai/research/reports/weekly-calibration-2026-08-17.json",
   "reply_tco_len": 278,
   "limit_tco": 280,
   "replies_max": 1
  }
 }
}
