{
  "slug": "weekly-calibration-2026-08-03",
  "series": "weekly",
  "series_label": "WEEKLY · CALIBRATION",
  "issue": 1,
  "title": "Weekly Calibration #1",
  "window_start": "2026-08-03T00:00:00+00:00",
  "window_end": "2026-08-10T00:00:00+00:00",
  "cutoff": "2026-08-10T16:00:00+00:00",
  "generated_at": "2026-08-11T14:24:23.984020+00:00",
  "methodology": "v1.1 (2026-08-10)",
  "methodology_hash": "e66c7e8c864a2233",
  "source": "weekly_metrics_2026-08-03.json",
  "hit_rule": "direction: exit tp1/tp2 -> hit, sl -> miss, expiry -> sign of gross pnl",
  "bibtex_key": "mm_weekly_calibration_2026w32",
  "tables": {
    "calibration_main": [
      {
        "model": "qwen-3.8-max",
        "legacy": false,
        "is_field": false,
        "n": 447,
        "coverage": 0.5392,
        "hit_rate": 0.4407,
        "mean_conf": 59.2,
        "gap_pp": 15.1,
        "brier": 0.2713,
        "wilson_95": {
          "lo": 0.3954,
          "hi": 0.4871
        }
      },
      {
        "model": "grok-4.5",
        "legacy": false,
        "is_field": false,
        "n": 607,
        "coverage": 0.4669,
        "hit_rate": 0.4217,
        "mean_conf": 60.5,
        "gap_pp": 18.3,
        "brier": 0.2811,
        "wilson_95": {
          "lo": 0.3831,
          "hi": 0.4614
        }
      },
      {
        "model": "claude-fable-5",
        "legacy": false,
        "is_field": false,
        "n": 642,
        "coverage": 0.492,
        "hit_rate": 0.4143,
        "mean_conf": 60.7,
        "gap_pp": 19.2,
        "brier": 0.2811,
        "wilson_95": {
          "lo": 0.3768,
          "hi": 0.4528
        }
      },
      {
        "model": "claude-opus-5",
        "legacy": false,
        "is_field": false,
        "n": 465,
        "coverage": 0.3563,
        "hit_rate": 0.4086,
        "mean_conf": 60.8,
        "gap_pp": 20.0,
        "brier": 0.2825,
        "wilson_95": {
          "lo": 0.3648,
          "hi": 0.4539
        }
      },
      {
        "model": "deepseek-v4-pro",
        "legacy": false,
        "is_field": false,
        "n": 377,
        "coverage": 0.2889,
        "hit_rate": 0.4164,
        "mean_conf": 62.0,
        "gap_pp": 20.3,
        "brier": 0.2915,
        "wilson_95": {
          "lo": 0.3678,
          "hi": 0.4668
        }
      },
      {
        "model": "gemini-3.1-pro",
        "legacy": false,
        "is_field": false,
        "n": 629,
        "coverage": 0.4831,
        "hit_rate": 0.4372,
        "mean_conf": 65.9,
        "gap_pp": 22.2,
        "brier": 0.3038,
        "wilson_95": {
          "lo": 0.3989,
          "hi": 0.4762
        }
      },
      {
        "model": "gpt-5.6-sol",
        "legacy": false,
        "is_field": false,
        "n": 612,
        "coverage": 0.469,
        "hit_rate": 0.4297,
        "mean_conf": 67.4,
        "gap_pp": 24.4,
        "brier": 0.3076,
        "wilson_95": {
          "lo": 0.3911,
          "hi": 0.4693
        }
      },
      {
        "model": "qwen-3.7-max",
        "legacy": true,
        "is_field": false,
        "n": 263,
        "coverage": 0.5596,
        "hit_rate": 0.4221,
        "mean_conf": 66.0,
        "gap_pp": 23.8,
        "brier": 0.3129,
        "wilson_95": {
          "lo": 0.3639,
          "hi": 0.4824
        }
      },
      {
        "model": "Field (all models)",
        "legacy": false,
        "is_field": true,
        "n": 4042,
        "coverage": null,
        "hit_rate": 0.4243,
        "mean_conf": 62.8,
        "gap_pp": 20.4,
        "brier": 0.2908,
        "wilson_95": null
      }
    ],
    "buckets": [
      {
        "model": "qwen-3.8-max",
        "bucket": "50-60",
        "n": 223,
        "hit_rate": 0.4843,
        "mean_conf": 56.6,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "qwen-3.8-max",
        "bucket": "60-70",
        "n": 221,
        "hit_rate": 0.4027,
        "mean_conf": 61.9,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "qwen-3.8-max",
        "bucket": "70-80",
        "n": 0,
        "hit_rate": null,
        "mean_conf": null,
        "insufficient": true,
        "no_data": true
      },
      {
        "model": "qwen-3.8-max",
        "bucket": "80-100",
        "n": 0,
        "hit_rate": null,
        "mean_conf": null,
        "insufficient": true,
        "no_data": true
      },
      {
        "model": "grok-4.5",
        "bucket": "50-60",
        "n": 323,
        "hit_rate": 0.4396,
        "mean_conf": 57.6,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "grok-4.5",
        "bucket": "60-70",
        "n": 280,
        "hit_rate": 0.4,
        "mean_conf": 63.7,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "grok-4.5",
        "bucket": "70-80",
        "n": 4,
        "hit_rate": 0.5,
        "mean_conf": 71.5,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "grok-4.5",
        "bucket": "80-100",
        "n": 0,
        "hit_rate": null,
        "mean_conf": null,
        "insufficient": true,
        "no_data": true
      },
      {
        "model": "claude-fable-5",
        "bucket": "50-60",
        "n": 208,
        "hit_rate": 0.4279,
        "mean_conf": 56.8,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "claude-fable-5",
        "bucket": "60-70",
        "n": 434,
        "hit_rate": 0.4078,
        "mean_conf": 62.5,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "claude-fable-5",
        "bucket": "70-80",
        "n": 0,
        "hit_rate": null,
        "mean_conf": null,
        "insufficient": true,
        "no_data": true
      },
      {
        "model": "claude-fable-5",
        "bucket": "80-100",
        "n": 0,
        "hit_rate": null,
        "mean_conf": null,
        "insufficient": true,
        "no_data": true
      },
      {
        "model": "claude-opus-5",
        "bucket": "50-60",
        "n": 114,
        "hit_rate": 0.3947,
        "mean_conf": 58.0,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "claude-opus-5",
        "bucket": "60-70",
        "n": 351,
        "hit_rate": 0.4131,
        "mean_conf": 61.8,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "claude-opus-5",
        "bucket": "70-80",
        "n": 0,
        "hit_rate": null,
        "mean_conf": null,
        "insufficient": true,
        "no_data": true
      },
      {
        "model": "claude-opus-5",
        "bucket": "80-100",
        "n": 0,
        "hit_rate": null,
        "mean_conf": null,
        "insufficient": true,
        "no_data": true
      },
      {
        "model": "deepseek-v4-pro",
        "bucket": "50-60",
        "n": 93,
        "hit_rate": 0.3656,
        "mean_conf": 55.5,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "deepseek-v4-pro",
        "bucket": "60-70",
        "n": 218,
        "hit_rate": 0.4495,
        "mean_conf": 63.0,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "deepseek-v4-pro",
        "bucket": "70-80",
        "n": 53,
        "hit_rate": 0.3585,
        "mean_conf": 72.0,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "deepseek-v4-pro",
        "bucket": "80-100",
        "n": 3,
        "hit_rate": 0.3333,
        "mean_conf": 80,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "gemini-3.1-pro",
        "bucket": "50-60",
        "n": 6,
        "hit_rate": 0.5,
        "mean_conf": 55,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "gemini-3.1-pro",
        "bucket": "60-70",
        "n": 421,
        "hit_rate": 0.4727,
        "mean_conf": 63.0,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "gemini-3.1-pro",
        "bucket": "70-80",
        "n": 201,
        "hit_rate": 0.3632,
        "mean_conf": 72.3,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "gemini-3.1-pro",
        "bucket": "80-100",
        "n": 1,
        "hit_rate": 0.0,
        "mean_conf": 80,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "gpt-5.6-sol",
        "bucket": "50-60",
        "n": 6,
        "hit_rate": 0.3333,
        "mean_conf": 58.5,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "gpt-5.6-sol",
        "bucket": "60-70",
        "n": 444,
        "hit_rate": 0.4324,
        "mean_conf": 65.4,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "gpt-5.6-sol",
        "bucket": "70-80",
        "n": 161,
        "hit_rate": 0.4224,
        "mean_conf": 73.1,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "gpt-5.6-sol",
        "bucket": "80-100",
        "n": 1,
        "hit_rate": 1.0,
        "mean_conf": 82,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "qwen-3.7-max",
        "bucket": "50-60",
        "n": 12,
        "hit_rate": 0.6667,
        "mean_conf": 54.8,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "qwen-3.7-max",
        "bucket": "60-70",
        "n": 159,
        "hit_rate": 0.4528,
        "mean_conf": 63.5,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "qwen-3.7-max",
        "bucket": "70-80",
        "n": 79,
        "hit_rate": 0.3418,
        "mean_conf": 74.2,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "qwen-3.7-max",
        "bucket": "80-100",
        "n": 4,
        "hit_rate": 0.25,
        "mean_conf": 82.2,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "qwen-3.8-max",
        "bucket": "sub50",
        "n": 3,
        "hit_rate": 0.0,
        "mean_conf": null,
        "insufficient": true,
        "no_data": false
      },
      {
        "model": "deepseek-v4-pro",
        "bucket": "sub50",
        "n": 10,
        "hit_rate": 0.5,
        "mean_conf": null,
        "insufficient": false,
        "no_data": false
      },
      {
        "model": "qwen-3.7-max",
        "bucket": "sub50",
        "n": 9,
        "hit_rate": 0.3333,
        "mean_conf": null,
        "insufficient": true,
        "no_data": false
      }
    ],
    "by_fh": [
      {
        "model": "qwen-3.8-max",
        "fh": "1h",
        "n": 279,
        "hit_rate": 0.4337,
        "flag": null
      },
      {
        "model": "qwen-3.8-max",
        "fh": "4h",
        "n": 149,
        "hit_rate": 0.4698,
        "flag": null
      },
      {
        "model": "qwen-3.8-max",
        "fh": "1d",
        "n": 19,
        "hit_rate": 0.3158,
        "flag": "low"
      },
      {
        "model": "grok-4.5",
        "fh": "1h",
        "n": 404,
        "hit_rate": 0.4208,
        "flag": null
      },
      {
        "model": "grok-4.5",
        "fh": "4h",
        "n": 180,
        "hit_rate": 0.4278,
        "flag": null
      },
      {
        "model": "grok-4.5",
        "fh": "1d",
        "n": 23,
        "hit_rate": 0.3913,
        "flag": null
      },
      {
        "model": "claude-fable-5",
        "fh": "1h",
        "n": 397,
        "hit_rate": 0.4232,
        "flag": null
      },
      {
        "model": "claude-fable-5",
        "fh": "4h",
        "n": 216,
        "hit_rate": 0.3935,
        "flag": null
      },
      {
        "model": "claude-fable-5",
        "fh": "1d",
        "n": 29,
        "hit_rate": 0.4483,
        "flag": null
      },
      {
        "model": "claude-opus-5",
        "fh": "1h",
        "n": 275,
        "hit_rate": 0.4109,
        "flag": null
      },
      {
        "model": "claude-opus-5",
        "fh": "4h",
        "n": 165,
        "hit_rate": 0.3939,
        "flag": null
      },
      {
        "model": "claude-opus-5",
        "fh": "1d",
        "n": 25,
        "hit_rate": 0.48,
        "flag": null
      },
      {
        "model": "deepseek-v4-pro",
        "fh": "1h",
        "n": 250,
        "hit_rate": 0.432,
        "flag": null
      },
      {
        "model": "deepseek-v4-pro",
        "fh": "4h",
        "n": 118,
        "hit_rate": 0.4153,
        "flag": null
      },
      {
        "model": "deepseek-v4-pro",
        "fh": "1d",
        "n": 9,
        "hit_rate": 0.0,
        "flag": "insufficient"
      },
      {
        "model": "gemini-3.1-pro",
        "fh": "1h",
        "n": 405,
        "hit_rate": 0.4593,
        "flag": null
      },
      {
        "model": "gemini-3.1-pro",
        "fh": "4h",
        "n": 198,
        "hit_rate": 0.404,
        "flag": null
      },
      {
        "model": "gemini-3.1-pro",
        "fh": "1d",
        "n": 26,
        "hit_rate": 0.3462,
        "flag": null
      },
      {
        "model": "gpt-5.6-sol",
        "fh": "1h",
        "n": 386,
        "hit_rate": 0.4456,
        "flag": null
      },
      {
        "model": "gpt-5.6-sol",
        "fh": "4h",
        "n": 201,
        "hit_rate": 0.403,
        "flag": null
      },
      {
        "model": "gpt-5.6-sol",
        "fh": "1d",
        "n": 25,
        "hit_rate": 0.4,
        "flag": null
      },
      {
        "model": "qwen-3.7-max",
        "fh": "1h",
        "n": 165,
        "hit_rate": 0.4485,
        "flag": null
      },
      {
        "model": "qwen-3.7-max",
        "fh": "4h",
        "n": 85,
        "hit_rate": 0.4118,
        "flag": null
      },
      {
        "model": "qwen-3.7-max",
        "fh": "1d",
        "n": 13,
        "hit_rate": 0.1538,
        "flag": "low"
      }
    ],
    "trading": [
      {
        "model": "qwen-3.8-max",
        "n_trades": 447,
        "wr": 0.2975,
        "pnl_net_usd": -43.32,
        "pnl_gross_usd": 1.38,
        "max_dd_usd": -43.63
      },
      {
        "model": "qwen-3.7-max",
        "n_trades": 263,
        "wr": 0.2738,
        "pnl_net_usd": -50.32,
        "pnl_gross_usd": -24.02,
        "max_dd_usd": -56.4
      },
      {
        "model": "deepseek-v4-pro",
        "n_trades": 377,
        "wr": 0.2626,
        "pnl_net_usd": -53.93,
        "pnl_gross_usd": -16.23,
        "max_dd_usd": -57.41
      },
      {
        "model": "claude-opus-5",
        "n_trades": 465,
        "wr": 0.2796,
        "pnl_net_usd": -61.79,
        "pnl_gross_usd": -15.29,
        "max_dd_usd": -61.79
      },
      {
        "model": "gpt-5.6-sol",
        "n_trades": 612,
        "wr": 0.2876,
        "pnl_net_usd": -78.59,
        "pnl_gross_usd": -17.39,
        "max_dd_usd": -80.26
      },
      {
        "model": "gemini-3.1-pro",
        "n_trades": 629,
        "wr": 0.2734,
        "pnl_net_usd": -78.71,
        "pnl_gross_usd": -15.81,
        "max_dd_usd": -79.73
      },
      {
        "model": "claude-fable-5",
        "n_trades": 642,
        "wr": 0.2664,
        "pnl_net_usd": -87.15,
        "pnl_gross_usd": -22.95,
        "max_dd_usd": -87.15
      },
      {
        "model": "grok-4.5",
        "n_trades": 607,
        "wr": 0.2718,
        "pnl_net_usd": -89.08,
        "pnl_gross_usd": -28.38,
        "max_dd_usd": -91.01
      }
    ],
    "counters": {
      "forecasts_total": 9194,
      "ok_in_gate": 9121,
      "invalid": 3,
      "out_of_gate_1w": 70,
      "mature": 9121,
      "mature_directional": 4042,
      "mature_sideways": 5079,
      "pending_next_issue": 0,
      "uptime": [
        {
          "fh": "1h",
          "tf": "1h",
          "slots_seen": 165,
          "slots_expected": 168
        },
        {
          "fh": "4h",
          "tf": "4h",
          "slots_seen": 41,
          "slots_expected": 42
        },
        {
          "fh": "4h",
          "tf": "1h",
          "slots_seen": 41,
          "slots_expected": 42
        },
        {
          "fh": "1d",
          "tf": "1d",
          "slots_seen": 7,
          "slots_expected": 7
        },
        {
          "fh": "1d",
          "tf": "4h",
          "slots_seen": 7,
          "slots_expected": 7
        }
      ]
    }
  },
  "key_findings": [
    "OBSERVATION -- A flat, hard week for the whole field. Directional calls hit 42.4% of the time against 62.8 stated mean confidence -- a +20.4pp overconfidence gap. Every model's Brier score topped 0.25 this week, worse than an uninformative always-50% predictor.",
    "qwen-3.8-max is this week's best-calibrated model: lowest Brier (0.2713), smallest gap (+15.1pp) and highest hit-rate (44.1%) of the 8-model field -- in line with the caution it showed in Stability Index Run 2 (a different metric, different sample; read as context, not confirmation).",
    "Confidence mostly failed to rank outcomes this week: of the 6 models with sufficient N (>=10) in both the 50-60 and 60-70 buckets, 4 did not out-hit 50-60 from 60-70. Higher up, deepseek-v4-pro, gemini-3.1-pro, qwen-3.7-max all hit meaningfully less in 70-80 than 60-70 (inverted); gpt-5.6-sol alone stayed roughly flat.",
    "Brier and hit-rate rankings reward different things: gemini-3.1-pro is 2nd of 8 by hit-rate (43.7%) but only 6th by Brier (0.3038), with the field's third-largest gap (22.2pp). gpt-5.6-sol posts the single largest gap this week, at 24.4pp."
  ],
  "limitations": [
    "Prediction metrics (this report) and trading metrics are kept in separate sections per methodology -- they are never combined into a single score.",
    "95% Wilson CIs shown are descriptive, not inferential: observations inside one window are dependent (a single market wave can move many forecasts together), so read them as a range, not a formal coverage guarantee.",
    "All 4,042 scored calls this week sit inside one market regime (Aug 3-9). A single week cannot separate a persistent calibration pattern from this week's specific conditions.",
    "This is issue #1 of Weekly Calibration -- there is no prior issue to compare against, so no week-over-week deltas are reported. They begin with issue #2.",
    "Models report confidence at discrete levels, not a continuous scale; the 50-60 / 60-70 / 70-80 / 80-100 buckets reflect those natural breakpoints, not an arbitrary binning choice.",
    "Underlying price series are reconstructed from trade entry prices (median per symbol-slot), not an independent tick feed."
  ],
  "key_finding": {
    "label": "KEY FINDING · OBSERVATION (one weekly window)",
    "text": "Frontier models were systematically overconfident this week: 62.8% stated confidence vs 42.4% realized directional accuracy.",
    "evidence_level": "observation"
  },
  "testing_next": [
    "Next issue: does the confidence-bucket inversion persist? Does the 60-70 vs 50-60 non-ranking repeat?",
    "Monthly test: is the overconfidence gap stable across market regimes (trend vs flat)?"
  ],
  "related_research": [
    {
      "title": "Stability Index Run 2",
      "url": "marketmania.ai/research/reports/si-run-2.pdf"
    },
    {
      "title": "Consensus Watch #1",
      "url": "marketmania.ai/research/reports/consensus-watch-2026-08-03.pdf"
    },
    {
      "title": "Weekly Model Watch #1",
      "url": "marketmania.ai/research/reports/model-watch-2026-08-03.pdf"
    }
  ],
  "related_research_note": "These publish together as one wave; links are stable.",
  "research_to_date": {
    "this_report": {
      "scored_observations": 4042
    },
    "platform": {
      "as_of_cutoff": "2026-08-10",
      "resolved_forecasts": 14151,
      "since": "2026-07-11",
      "models_tracked": 9,
      "models_current_frontier": 7,
      "models_archived_legacy": 2,
      "assets": 5,
      "forecast_horizons": 5,
      "cadence": "hourly",
      "published_reports": 4
    },
    "source": "platform_counters_2026-08-10"
  }
}
