{
  "slug": "consensus-watch-2026-08-03",
  "series": "weekly",
  "series_label": "WEEKLY · CONSENSUS",
  "issue": 1,
  "title": "Consensus Watch #1",
  "language": "en",
  "window_start": "2026-08-03T00:00:00+00:00",
  "window_end": "2026-08-10T00:00:00+00:00",
  "window_text": "Aug 3-9, 2026 UTC",
  "cutoff": "2026-08-10T16:00:00+00:00",
  "generated_at": "2026-08-11T14:24:23.984020+00:00",
  "methodology": "v1.1 (2026-08-10)",
  "methodology_hash": "e66c7e8c864a2233",
  "source": "weekly_metrics_2026-08-03.json",
  "research_question": "if a model's forecast matches the consensus of the other models, is the forecast more reliable?",
  "method": "leave-one-out strict majority, >=3 directional peers",
  "base_field_hit": {
    "n": 4042,
    "hit_rate": 0.4243
  },
  "tables": {
    "pooled": {
      "agree": {
        "n": 3308,
        "hit_rate": 0.4039,
        "wilson_95": [
          0.3873,
          0.4207
        ]
      },
      "disagree": {
        "n": 19,
        "hit_rate": 0.6316,
        "wilson_95": [
          0.4104,
          0.8085
        ]
      },
      "lift_pp": -22.8,
      "thin_or_tied": 715,
      "total_scoreable": 3327,
      "caveat": "lift_pp rests on disagree n=19; Wilson 95% CI [41.0%, 80.9%] overlaps the agree bucket's CI [38.7%, 42.1%]; descriptive, one week only, not a durability claim"
    },
    "by_cell": [
      {
        "cell": "fh_1h_tf_1h",
        "fh": "1h",
        "tf": "1h",
        "n_subjects": 2561,
        "agree": {
          "n": 2070,
          "hit_rate": 0.4174
        },
        "disagree": {
          "n": 19,
          "hit_rate": 0.6316
        },
        "lift_pp": -21.4,
        "thin_or_tied": 472
      },
      {
        "cell": "fh_4h_tf_4h",
        "fh": "4h",
        "tf": "4h",
        "n_subjects": 665,
        "agree": {
          "n": 562,
          "hit_rate": 0.3915
        },
        "disagree": {
          "n": 0,
          "hit_rate": null
        },
        "lift_pp": null,
        "thin_or_tied": 103
      },
      {
        "cell": "fh_4h_tf_1h",
        "fh": "4h",
        "tf": "1h",
        "n_subjects": 647,
        "agree": {
          "n": 544,
          "hit_rate": 0.3768
        },
        "disagree": {
          "n": 0,
          "hit_rate": null
        },
        "lift_pp": null,
        "thin_or_tied": 103
      },
      {
        "cell": "fh_1d_tf_4h",
        "fh": "1d",
        "tf": "4h",
        "n_subjects": 91,
        "agree": {
          "n": 77,
          "hit_rate": 0.3377
        },
        "disagree": {
          "n": 0,
          "hit_rate": null
        },
        "lift_pp": null,
        "thin_or_tied": 14
      },
      {
        "cell": "fh_1d_tf_1d",
        "fh": "1d",
        "tf": "1d",
        "n_subjects": 78,
        "agree": {
          "n": 55,
          "hit_rate": 0.3818
        },
        "disagree": {
          "n": 0,
          "hit_rate": null
        },
        "lift_pp": null,
        "thin_or_tied": 23
      }
    ],
    "by_margin": [
      {
        "margin": 2,
        "n": 3,
        "hit_rate": 1.0,
        "note": "insufficient n — footnote only, excluded from chart"
      },
      {
        "margin": 3,
        "n": 404,
        "hit_rate": 0.3045,
        "note": null
      },
      {
        "margin": 4,
        "n": 515,
        "hit_rate": 0.433,
        "note": null
      },
      {
        "margin": 5,
        "n": 1140,
        "hit_rate": 0.4991,
        "note": null
      },
      {
        "margin": 6,
        "n": 1246,
        "hit_rate": 0.3355,
        "note": "full unanimity of the LOO peer set"
      }
    ],
    "per_model": [
      {
        "model": "claude-fable-5",
        "agree": {
          "n": 526,
          "hit_rate": 0.4049,
          "wilson_95": [
            0.3638,
            0.4474
          ]
        },
        "disagree": {
          "n": 2,
          "hit_rate": 0.0,
          "wilson_95": [
            0.0,
            0.6576
          ]
        },
        "thin_or_tied": 114,
        "note": "lift not rankable at these n"
      },
      {
        "model": "claude-opus-5",
        "agree": {
          "n": 438,
          "hit_rate": 0.4018,
          "wilson_95": [
            0.357,
            0.4484
          ]
        },
        "disagree": {
          "n": 0,
          "hit_rate": null,
          "wilson_95": [
            null,
            null
          ]
        },
        "thin_or_tied": 27,
        "note": "lift not rankable at these n"
      },
      {
        "model": "deepseek-v4-pro",
        "agree": {
          "n": 311,
          "hit_rate": 0.4019,
          "wilson_95": [
            0.349,
            0.4573
          ]
        },
        "disagree": {
          "n": 3,
          "hit_rate": 0.6667,
          "wilson_95": [
            0.2077,
            0.9385
          ]
        },
        "thin_or_tied": 63,
        "note": "lift not rankable at these n"
      },
      {
        "model": "gemini-3.1-pro",
        "agree": {
          "n": 494,
          "hit_rate": 0.3927,
          "wilson_95": [
            0.3506,
            0.4364
          ]
        },
        "disagree": {
          "n": 6,
          "hit_rate": 0.6667,
          "wilson_95": [
            0.3,
            0.9032
          ]
        },
        "thin_or_tied": 129,
        "note": "lift not rankable at these n"
      },
      {
        "model": "gpt-5.6-sol",
        "agree": {
          "n": 527,
          "hit_rate": 0.4042,
          "wilson_95": [
            0.3631,
            0.4466
          ]
        },
        "disagree": {
          "n": 0,
          "hit_rate": null,
          "wilson_95": [
            null,
            null
          ]
        },
        "thin_or_tied": 85,
        "note": "lift not rankable at these n"
      },
      {
        "model": "grok-4.5",
        "agree": {
          "n": 489,
          "hit_rate": 0.4131,
          "wilson_95": [
            0.3703,
            0.4572
          ]
        },
        "disagree": {
          "n": 4,
          "hit_rate": 1.0,
          "wilson_95": [
            0.5101,
            1.0
          ]
        },
        "thin_or_tied": 114,
        "note": "lift not rankable at these n"
      },
      {
        "model": "qwen-3.7-max",
        "agree": {
          "n": 168,
          "hit_rate": 0.3988,
          "wilson_95": [
            0.3278,
            0.4743
          ]
        },
        "disagree": {
          "n": 1,
          "hit_rate": 1.0,
          "wilson_95": [
            0.2065,
            1.0
          ]
        },
        "thin_or_tied": 94,
        "note": "lift not rankable at these n"
      },
      {
        "model": "qwen-3.8-max",
        "agree": {
          "n": 355,
          "hit_rate": 0.4113,
          "wilson_95": [
            0.3613,
            0.4631
          ]
        },
        "disagree": {
          "n": 3,
          "hit_rate": 0.3333,
          "wilson_95": [
            0.0615,
            0.7923
          ]
        },
        "thin_or_tied": 89,
        "note": "lift not rankable at these n"
      }
    ],
    "herd_cards": [
      {
        "slot": "2026-08-04T19:01:00+00:00",
        "symbol": "BTC",
        "fh": "1h",
        "tf": "1h",
        "side": "long",
        "n_models": 7,
        "mean_conf": 71.7,
        "hit_rate": 0.0
      },
      {
        "slot": "2026-08-03T16:01:00+00:00",
        "symbol": "XRP",
        "fh": "4h",
        "tf": "1h",
        "side": "long",
        "n_models": 7,
        "mean_conf": 70,
        "hit_rate": 0.0
      },
      {
        "slot": "2026-08-05T20:01:00+00:00",
        "symbol": "BTC",
        "fh": "4h",
        "tf": "1h",
        "side": "long",
        "n_models": 7,
        "mean_conf": 69.9,
        "hit_rate": 0.0
      },
      {
        "slot": "2026-08-05T06:01:00+00:00",
        "symbol": "BNB",
        "fh": "1h",
        "tf": "1h",
        "side": "long",
        "n_models": 7,
        "mean_conf": 69.7,
        "hit_rate": 0.0
      },
      {
        "slot": "2026-08-09T16:01:00+00:00",
        "symbol": "BNB",
        "fh": "4h",
        "tf": "4h",
        "side": "long",
        "n_models": 7,
        "mean_conf": 69.7,
        "hit_rate": 0.0
      }
    ],
    "counters": {
      "forecasts_total": 9194,
      "ok_in_gate": 9121,
      "out_of_gate_1w": 70,
      "invalid": 3,
      "mature": 9121,
      "mature_directional": 4042,
      "mature_sideways": 5079,
      "pending_next_issue": 0,
      "late_closes": 0,
      "unanimous_cells_ge6": 367,
      "loo_thin_or_tied": 715
    }
  },
  "herd_tie_in": {
    "note": "independently cross-checked against model_watch.fail_of_week (not otherwise used in this report)",
    "slot": "2026-08-04T19:01:00+00:00",
    "model": "qwen-3.7-max",
    "symbol": "BTC",
    "side": "long",
    "confidence": 85
  },
  "smart_consensus_note": "Smart-vs-Outlook deferred: as-of weights pipeline not yet recording; will appear once weights are logged per-slot (methodology 3.2.3 fallback)",
  "key_finding": {
    "label": "KEY FINDING · OBSERVATION (one weekly window)",
    "text": "More agreement did not mean more reliability: full peer unanimity was the weakest high-consensus tier this week.",
    "evidence_level": "observation"
  },
  "key_findings": [
    "OBSERVATION -- Herding was near-total, and it did not pay. Of 3,327 LOO-scoreable directional calls, 3,308 (99.4%) sided with peers' majority (the other 19 were all forecast-horizon 1h calls); agree-hit was 40.4% vs. 63.2% for those 19 disagreers, a pooled lift of -22.8pp on n=19 with a Wilson 95% CI of [41.0%, 80.9%] that overlaps the agree bucket's own CI [38.7%, 42.1%]. Descriptive, one week, not a claim of a durable edge.",
    "The margin curve is not monotonic. Hit rate ran 30.5% at 3 agreeing peers (n=404), up to 43.3% at 4 (n=515) and a peak of 49.9% at 5 (n=1,140), then back down to 33.6% once all peers agreed (6 of 6, n=1,246) — the largest bucket in the data. More agreement did not mean a safer call.",
    "Herd failures were real. 367 cells had 6 or more directional models on the same side; the week's 5 worst were total wipeouts — all 7 active models long, 0% hit — and all 5 were long calls in a week where 'sideways' was the field's single most common call (55.7% of mature forecasts).",
    "Nobody escaped the herd. Every model's LOO-disagree count this week is 0-6, far too small to rank models by contrarian lift; grok-4.5's 4 contrarian calls all hit and claude-fable-5's 2 both missed, but both are anecdotes, not findings."
  ],
  "practical_implications": [
    "This week, peer-unanimity behaved like a crowding signal, not a safety signal — the largest 'everyone agrees' bucket (6 peers, n=1,246) hit worse (33.6%) than the 4- and 5-peer tiers below it. Treat a fully unanimous herd as a prompt to check position sizing, not as confirmation.",
    "Treat any consensus filter as regime-dependent, not as a fixed rule. This week's -22.8pp 'agree vs. disagree' lift (n=19 disagree; Wilson CI [41.0%, 80.9%], overlapping the agree bucket's CI) should not be read as 'contrarian calls are better' — it is one thin week, reported descriptively, not a strategy.",
    "Consensus Watch is a living series: each issue adds one more week to the LOO comparison and one more point on the margin curve. Only after several issues will it be possible to ask whether this week's non-monotonic curve and negative pooled lift are a pattern or noise."
  ],
  "limitations": [
    "Single week (Aug 3-9, 2026 UTC). Every finding in this issue is descriptive for this window only; no claim is made about next week or about any model's underlying skill.",
    "Observations are not independent: the same models watch overlapping symbol / FH / TF cells hour after hour, so LOO agree/disagree calls made close in time or on the same slot are correlated, not i.i.d. draws. Wilson confidence intervals in this issue are descriptive, not inferential.",
    "The design is correlational, not causal. All models see the same market data and broadly similar priors, so 'agreement' and 'hit' can rise together simply because a slot was easy to read that week — shared inputs mean shared blind spots, and a readable market can lift agreement and hit rate together without either causing the other. LOO removes self-match bias, not this confound.",
    "The headline contrarian result rests on 19 disagreeing calls pooled across all models, all of them in FH 1h; per-model disagree counts run 0-6, too small to rank models by contrarian lift.",
    "715 calls (17.7% of the 4,042-call mature directional universe) were thin (fewer than 3 directional peers in the cell) or tied (peers split evenly) and are excluded from every consensus table in this issue — they are not folded into agree or disagree either way."
  ],
  "citation": {
    "bibtex_key": "mm_consensus_watch_2026w32",
    "title": "Consensus Watch #1: does agreeing with the model consensus make a forecast more reliable?",
    "author": "MarketMania Research",
    "year": 2026,
    "month": "August",
    "day": 10,
    "url": "https://marketmania.ai/research"
  },
  "living_series_note": "NEW series — Issue #1. Consensus Watch is a living weekly comparison; each issue appends one more week of leave-one-out agreement-vs-hit data, and coverage of the slower forecast horizons (especially 1d, 78-91 scoreable subjects this week vs. 2,561 for 1h/1h) widens as more weeks mature. Read any single issue descriptively, and watch the margin curve across issues before drawing conclusions.",
  "testing_next": [
    "Next issue: does full unanimity continue to underperform?",
    "Monthly test: does the agreement curve survive across multiple weeks and regimes?"
  ],
  "related_research": [
    {
      "title": "Stability Index Run 2",
      "url": "marketmania.ai/research/reports/si-run-2.pdf"
    },
    {
      "title": "Weekly Calibration #1",
      "url": "marketmania.ai/research/reports/weekly-calibration-2026-08-03.pdf"
    },
    {
      "title": "Weekly Model Watch #1",
      "url": "marketmania.ai/research/reports/model-watch-2026-08-03.pdf"
    }
  ],
  "related_research_note": "These three weekly reports — Consensus Watch, Weekly Calibration and Weekly Model Watch — publish together as one wave each week.",
  "research_to_date": {
    "this_report": {
      "loo_scored_observations": 3327
    },
    "platform": {
      "as_of_cutoff": "2026-08-10",
      "resolved_forecasts": 14151,
      "resolved_forecasts_since": "2026-07-11",
      "models_tracked": 9,
      "models_tracked_detail": "7 current frontier + 2 archived legacy generations",
      "assets": 5,
      "forecast_horizons": 5,
      "cadence": "hourly",
      "published_research_reports": 4
    },
    "source": "platform_counters_2026-08-10"
  }
}
