{
 "slug": "consensus-watch-2026-08-17",
 "series": "weekly",
 "series_label": "WEEKLY · CONSENSUS",
 "issue": 3,
 "title": "Consensus Watch #3",
 "language": "en",
 "window_start": "2026-08-17T00:00:00+00:00",
 "window_end": "2026-08-24T00:00:00+00:00",
 "window_text": "Aug 17-23, 2026 UTC",
 "cutoff": "2026-08-24T16:00:00+00:00",
 "generated_at": "2026-08-26T07:00:00+00:00",
 "methodology": "v1.1 (2026-08-10)",
 "methodology_hash": "e66c7e8c864a2233",
 "source": "weekly_metrics_2026-08-17.json",
 "research_question": "if a model's forecast matches the consensus of the other models, is the forecast more reliable?",
 "method": "leave-one-out strict majority, >=3 directional peers",
 "base_field_hit": {
  "n": 5130,
  "hit_rate": 0.5405
 },
 "market_state": {
  "panel": "market_state_6p2",
  "week": {
   "start": "2026-08-17",
   "end": "2026-08-24"
  },
  "prev_week": {
   "start": "2026-08-10",
   "end": "2026-08-17"
  },
  "current": {
   "btc_net_pct": 23.58,
   "realized_vol_ann_pct": 71.1,
   "volume_top5_usd_bn": 25.97,
   "volume_top5_wow_pct": 235.0,
   "avg_pairwise_corr": 0.81
  },
  "previous_as_published_issue2": {
   "btc_net_pct": -3.08,
   "realized_vol_ann_pct": 10.0,
   "volume_top5_usd_bn": 7.7521,
   "avg_pairwise_corr": 0.52
  },
  "snapshot_line": "BTC net +23.58% (prior -3.08%) - ann vol 71.1% (was 10.0%) - TOP5 volume $25.97B, +235% w/w - pairwise corr 0.81 (was 0.52)",
  "exchange": "binance",
  "source": "owner-verified single-exchange (binance) daily candle recompute, 2026-08-26; prior-week values as published in issue #2",
  "audit_note": "Market-state row is computed on a single-exchange (binance) daily candle series this issue, after an audit found the raw table mixes two exchanges; issue #2 row is as published.",
  "regime_days": {
   "2026-08-17": {
    "move_pct": 2.55,
    "regime": "trend"
   },
   "2026-08-18": {
    "move_pct": 0.21,
    "regime": "flat"
   },
   "2026-08-19": {
    "move_pct": 7.82,
    "regime": "trend"
   },
   "2026-08-20": {
    "move_pct": 5.07,
    "regime": "trend"
   },
   "2026-08-21": {
    "move_pct": 7.26,
    "regime": "trend"
   },
   "2026-08-22": {
    "move_pct": -1.67,
    "regime": "trend"
   },
   "2026-08-23": {
    "move_pct": 0.69,
    "regime": "flat"
   }
  },
  "alive_slice_context": {
   "abs_move_1d_pct": {
    "A": 0.734,
    "B": 4.283
   },
   "btc_realized_vol_ann_pct_hourly": {
    "A": 19.26,
    "B": 58.72
   }
  }
 },
 "tables": {
  "pooled": {
   "agree": {
    "n": 4232,
    "hit_rate": 0.5475,
    "wilson_95": [
     0.5325,
     0.5624
    ]
   },
   "disagree": {
    "n": 125,
    "hit_rate": 0.376,
    "wilson_95": [
     0.296,
     0.4634
    ]
   },
   "lift_pp": 17.1,
   "thin_or_tied": 773,
   "total_scoreable": 4357,
   "caveat": "first issue with a readable disagree bucket (n=125); the two Wilson intervals do not overlap, but observations inside one window are dependent -- descriptive, one week only, not a durability claim."
  },
  "by_cell": [
   {
    "cell": "fh_1h_tf_1h",
    "fh": "1h",
    "tf": "1h",
    "n_subjects": 3089,
    "agree": {
     "n": 2554,
     "hit_rate": 0.493,
     "wilson_95": [
      0.4736,
      0.5123
     ]
    },
    "disagree": {
     "n": 51,
     "hit_rate": 0.451,
     "wilson_95": [
      0.3227,
      0.5862
     ]
    },
    "lift_pp": 4.2,
    "thin_or_tied": 484,
    "by_margin": {
     "2": {
      "n": 45,
      "hit_rate": 0.5778
     },
     "3": {
      "n": 292,
      "hit_rate": 0.4966
     },
     "4": {
      "n": 470,
      "hit_rate": 0.3745
     },
     "5": {
      "n": 648,
      "hit_rate": 0.5108
     },
     "6": {
      "n": 1099,
      "hit_rate": 0.5287
     }
    }
   },
   {
    "cell": "fh_4h_tf_4h",
    "fh": "4h",
    "tf": "4h",
    "n_subjects": 840,
    "agree": {
     "n": 657,
     "hit_rate": 0.6317,
     "wilson_95": [
      0.5941,
      0.6677
     ]
    },
    "disagree": {
     "n": 50,
     "hit_rate": 0.36,
     "wilson_95": [
      0.2414,
      0.4986
     ]
    },
    "lift_pp": 27.2,
    "thin_or_tied": 133,
    "by_margin": {
     "2": {
      "n": 24,
      "hit_rate": 0.5417
     },
     "3": {
      "n": 88,
      "hit_rate": 0.6477
     },
     "4": {
      "n": 100,
      "hit_rate": 0.8
     },
     "5": {
      "n": 186,
      "hit_rate": 0.6398
     },
     "6": {
      "n": 259,
      "hit_rate": 0.5637
     }
    }
   },
   {
    "cell": "fh_4h_tf_1h",
    "fh": "4h",
    "tf": "1h",
    "n_subjects": 853,
    "agree": {
     "n": 703,
     "hit_rate": 0.5875,
     "wilson_95": [
      0.5507,
      0.6233
     ]
    },
    "disagree": {
     "n": 18,
     "hit_rate": 0.3333,
     "wilson_95": [
      0.1628,
      0.5625
     ]
    },
    "lift_pp": 25.4,
    "thin_or_tied": 132,
    "by_margin": {
     "2": {
      "n": 6,
      "hit_rate": 0.5
     },
     "3": {
      "n": 64,
      "hit_rate": 0.6094
     },
     "4": {
      "n": 100,
      "hit_rate": 0.63
     },
     "5": {
      "n": 204,
      "hit_rate": 0.5196
     },
     "6": {
      "n": 329,
      "hit_rate": 0.614
     }
    }
   },
   {
    "cell": "fh_1d_tf_1d",
    "fh": "1d",
    "tf": "1d",
    "n_subjects": 171,
    "agree": {
     "n": 159,
     "hit_rate": 0.7547,
     "wilson_95": [
      0.6824,
      0.8151
     ]
    },
    "disagree": {
     "n": 4,
     "hit_rate": 0.0,
     "wilson_95": [
      0.0,
      0.4899
     ]
    },
    "lift_pp": 75.5,
    "thin_or_tied": 8,
    "by_margin": {
     "2": {
      "n": 6,
      "hit_rate": 1.0
     },
     "3": {
      "n": 12,
      "hit_rate": 0.9167
     },
     "4": {
      "n": 15,
      "hit_rate": 0.8667
     },
     "5": {
      "n": 42,
      "hit_rate": 0.5238
     },
     "6": {
      "n": 84,
      "hit_rate": 0.8095
     }
    }
   },
   {
    "cell": "fh_1d_tf_4h",
    "fh": "1d",
    "tf": "4h",
    "n_subjects": 177,
    "agree": {
     "n": 159,
     "hit_rate": 0.6918,
     "wilson_95": [
      0.6162,
      0.7584
     ]
    },
    "disagree": {
     "n": 2,
     "hit_rate": 0.0,
     "wilson_95": [
      0.0,
      0.6576
     ]
    },
    "lift_pp": 69.2,
    "thin_or_tied": 16,
    "by_margin": {
     "3": {
      "n": 12,
      "hit_rate": 1.0
     },
     "4": {
      "n": 15,
      "hit_rate": 0.6667
     },
     "5": {
      "n": 48,
      "hit_rate": 0.75
     },
     "6": {
      "n": 84,
      "hit_rate": 0.619
     }
    }
   }
  ],
  "by_margin": [
   {
    "margin": 2,
    "n": 81,
    "hit_rate": 0.5926
   },
   {
    "margin": 3,
    "n": 468,
    "hit_rate": 0.5641
   },
   {
    "margin": 4,
    "n": 700,
    "hit_rate": 0.4886
   },
   {
    "margin": 5,
    "n": 1128,
    "hit_rate": 0.5443
   },
   {
    "margin": 6,
    "n": 1855,
    "hit_rate": 0.5655
   }
  ],
  "by_margin_week_over_week": [
   {
    "margin": 3,
    "issue2_hit_rate_pct": 50.2,
    "issue3_hit_rate_pct": 56.4,
    "issue3_n": 468
   },
   {
    "margin": 4,
    "issue2_hit_rate_pct": 39.4,
    "issue3_hit_rate_pct": 48.9,
    "issue3_n": 700
   },
   {
    "margin": 5,
    "issue2_hit_rate_pct": 41.0,
    "issue3_hit_rate_pct": 54.4,
    "issue3_n": 1128
   },
   {
    "margin": 6,
    "issue2_hit_rate_pct": 43.3,
    "issue3_hit_rate_pct": 56.6,
    "issue3_n": 1855
   }
  ],
  "per_model": [
   {
    "model": "claude-fable-5",
    "agree": {
     "n": 666,
     "hit_rate": 0.5556,
     "wilson_95": [
      0.5176,
      0.5929
     ]
    },
    "disagree": {
     "n": 5,
     "hit_rate": 0.4,
     "wilson_95": [
      0.1176,
      0.7693
     ]
    },
    "thin_or_tied": 108,
    "by_margin": {
     "2": {
      "n": 17,
      "hit_rate": 0.7059
     },
     "3": {
      "n": 81,
      "hit_rate": 0.5679
     },
     "4": {
      "n": 117,
      "hit_rate": 0.4701
     },
     "5": {
      "n": 186,
      "hit_rate": 0.5591
     },
     "6": {
      "n": 265,
      "hit_rate": 0.5774
     }
    }
   },
   {
    "model": "claude-opus-5",
    "agree": {
     "n": 604,
     "hit_rate": 0.5877,
     "wilson_95": [
      0.5481,
      0.6263
     ]
    },
    "disagree": {
     "n": 3,
     "hit_rate": 0.3333,
     "wilson_95": [
      0.0615,
      0.7923
     ]
    },
    "thin_or_tied": 65,
    "by_margin": {
     "2": {
      "n": 16,
      "hit_rate": 0.5
     },
     "3": {
      "n": 60,
      "hit_rate": 0.6333
     },
     "4": {
      "n": 101,
      "hit_rate": 0.5842
     },
     "5": {
      "n": 162,
      "hit_rate": 0.5864
     },
     "6": {
      "n": 265,
      "hit_rate": 0.5849
     }
    }
   },
   {
    "model": "deepseek-v4-pro",
    "agree": {
     "n": 621,
     "hit_rate": 0.5233,
     "wilson_95": [
      0.484,
      0.5624
     ]
    },
    "disagree": {
     "n": 22,
     "hit_rate": 0.4545,
     "wilson_95": [
      0.2692,
      0.6534
     ]
    },
    "thin_or_tied": 134,
    "by_margin": {
     "2": {
      "n": 13,
      "hit_rate": 0.5385
     },
     "3": {
      "n": 71,
      "hit_rate": 0.5352
     },
     "4": {
      "n": 101,
      "hit_rate": 0.4257
     },
     "5": {
      "n": 171,
      "hit_rate": 0.5439
     },
     "6": {
      "n": 265,
      "hit_rate": 0.5434
     }
    }
   },
   {
    "model": "gemini-3.1-pro",
    "agree": {
     "n": 578,
     "hit_rate": 0.5502,
     "wilson_95": [
      0.5094,
      0.5903
     ]
    },
    "disagree": {
     "n": 53,
     "hit_rate": 0.3774,
     "wilson_95": [
      0.2594,
      0.5119
     ]
    },
    "thin_or_tied": 167,
    "by_margin": {
     "2": {
      "n": 10,
      "hit_rate": 0.7
     },
     "3": {
      "n": 72,
      "hit_rate": 0.5833
     },
     "4": {
      "n": 97,
      "hit_rate": 0.4845
     },
     "5": {
      "n": 134,
      "hit_rate": 0.5299
     },
     "6": {
      "n": 265,
      "hit_rate": 0.5698
     }
    }
   },
   {
    "model": "gpt-5.6-sol",
    "agree": {
     "n": 600,
     "hit_rate": 0.545,
     "wilson_95": [
      0.505,
      0.5844
     ]
    },
    "disagree": {
     "n": 7,
     "hit_rate": 0.2857,
     "wilson_95": [
      0.0822,
      0.6411
     ]
    },
    "thin_or_tied": 81,
    "by_margin": {
     "2": {
      "n": 2,
      "hit_rate": 0.0
     },
     "3": {
      "n": 59,
      "hit_rate": 0.5593
     },
     "4": {
      "n": 104,
      "hit_rate": 0.4808
     },
     "5": {
      "n": 170,
      "hit_rate": 0.5294
     },
     "6": {
      "n": 265,
      "hit_rate": 0.5811
     }
    }
   },
   {
    "model": "grok-4.5",
    "agree": {
     "n": 559,
     "hit_rate": 0.5456,
     "wilson_95": [
      0.5042,
      0.5864
     ]
    },
    "disagree": {
     "n": 33,
     "hit_rate": 0.3636,
     "wilson_95": [
      0.2219,
      0.5338
     ]
    },
    "thin_or_tied": 118,
    "by_margin": {
     "2": {
      "n": 9,
      "hit_rate": 0.6667
     },
     "3": {
      "n": 49,
      "hit_rate": 0.5102
     },
     "4": {
      "n": 83,
      "hit_rate": 0.4819
     },
     "5": {
      "n": 153,
      "hit_rate": 0.5556
     },
     "6": {
      "n": 265,
      "hit_rate": 0.5623
     }
    }
   },
   {
    "model": "qwen-3.8-max",
    "agree": {
     "n": 604,
     "hit_rate": 0.5248,
     "wilson_95": [
      0.485,
      0.5644
     ]
    },
    "disagree": {
     "n": 2,
     "hit_rate": 0.0,
     "wilson_95": [
      0.0,
      0.6576
     ]
    },
    "thin_or_tied": 100,
    "by_margin": {
     "2": {
      "n": 14,
      "hit_rate": 0.5714
     },
     "3": {
      "n": 76,
      "hit_rate": 0.5526
     },
     "4": {
      "n": 97,
      "hit_rate": 0.4948
     },
     "5": {
      "n": 152,
      "hit_rate": 0.5
     },
     "6": {
      "n": 265,
      "hit_rate": 0.5396
     }
    }
   }
  ],
  "herd_worst_cards": [
   {
    "slot": "2026-08-18T16:01:00+00:00",
    "symbol": "BTC",
    "fh": "4h",
    "tf": "1h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 69.3,
    "hit_rate": 0.0
   },
   {
    "slot": "2026-08-23T22:01:00+00:00",
    "symbol": "BTC",
    "fh": "1h",
    "tf": "1h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 68.4,
    "hit_rate": 0.0
   },
   {
    "slot": "2026-08-19T09:01:00+00:00",
    "symbol": "SOL",
    "fh": "1h",
    "tf": "1h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 68.3,
    "hit_rate": 0.0
   },
   {
    "slot": "2026-08-18T19:01:00+00:00",
    "symbol": "SOL",
    "fh": "1h",
    "tf": "1h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 68,
    "hit_rate": 0.0
   },
   {
    "slot": "2026-08-18T16:01:00+00:00",
    "symbol": "SOL",
    "fh": "1h",
    "tf": "1h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 68,
    "hit_rate": 0.1429
   }
  ],
  "counters": {
   "forecasts_total": 9590,
   "ok_in_gate": 9260,
   "invalid": 53,
   "out_of_gate": 277,
   "out_of_gate_by_fh": {
    "1w": 209,
    "1m": 68
   },
   "mature": 9260,
   "mature_directional": 5130,
   "mature_sideways": 4130,
   "pending_next_issue": 0,
   "late_closes": 0,
   "uptime": [
    {
     "fh": "1h",
     "tf": "1h",
     "slots_seen": 168,
     "slots_expected": 168
    },
    {
     "fh": "4h",
     "tf": "4h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "4h",
     "tf": "1h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "1d",
     "tf": "1d",
     "slots_seen": 7,
     "slots_expected": 7
    },
    {
     "fh": "1d",
     "tf": "4h",
     "slots_seen": 7,
     "slots_expected": 7
    }
   ]
  }
 },
 "herd_note": "437 cells had 6+ directional models on one side; unanimous_cells_ge6 per weekly metrics",
 "smart_note": "Smart-vs-Outlook deferred: as-of weights pipeline not yet recording; will appear once weights are logged per-slot (methodology 3.2.3 fallback)",
 "key_finding": {
  "label": "KEY FINDING · OBSERVATION (one weekly window)",
  "text": "Herding eased to 97.1% and, for the first time in the series, leaving the crowd was the mistake: agree-hit 54.8% against 37.6% on disagreement, with the margin curve taking its third shape in three weeks.",
  "evidence_level": "observation"
 },
 "key_findings": [
  "OBSERVATION -- Herding eased, and this week it paid. Of 4,357 LOO-scoreable directional calls, 4,232 (97.1%) sided with peers' majority (issue #2: 99.8%); agree-hit was 54.8% [53.3%, 56.2%] vs. 37.6% [29.6%, 46.3%] for the 125 disagreers, a pooled lift of +17.1pp. Descriptive, one week, not a claim of a durable edge.",
  "The margin curve took a third shape: 3 peers 56.4% (was 50.2%), 4 peers 48.9% (was 39.4%), 5 peers 54.4% (was 41.0%), full unanimity 56.6% (was 43.3%). Only the 4-peer tier sits below the 54.1% base.",
  "Daily-horizon unanimity flipped: 1d/1d margin-6 hit 81.0% (n=84) and 1d/4h margin-6 hit 61.9% (n=84) -- issue #2 had the 1d/4h near-unanimous bucket at 0 for 28.",
  "Herd failures did not disappear: 437 cells with 6+ models on one side; the 5 worst -- all 7 models long at ~68-69 mean confidence -- resolved 0.0-14.3%, four of them on Aug 18-19."
 ],
 "market_check": {
  "label": "MARKET CHECK · OBSERVATION (two adjacent calendar weeks)",
  "text": "OBSERVATION — the tape sped up and accuracy followed: mean |1d move| 0.73% -> 4.28% (483% rel), BTC realized vol 19.3% -> 58.7% (ann., hourly); field directional accuracy 43.8% -> 54.1% (+10.3pp), 7 of 7 models improved, sim win-rate up for 7 of 7, field sim PnL -$527 -> +$889 (adjacent calendar weeks Aug 10-16 vs Aug 17-23; descriptive, one pair of weeks, not a claim).",
  "robustness": "Robustness: raw price-sign accuracy 40.1% -> 53.5% — consistent with the trade-based hit rule.",
  "weeks": {
   "A": {
    "days": "2026-08-10..2026-08-16",
    "slots": [
     "2026-08-10T00:00:00+00:00",
     "2026-08-17T00:00:00+00:00"
    ],
    "cutoff": "2026-08-17T16:00:00+00:00"
   },
   "B": {
    "days": "2026-08-17..2026-08-23",
    "slots": [
     "2026-08-17T00:00:00+00:00",
     "2026-08-24T00:00:00+00:00"
    ],
    "cutoff": "2026-08-24T16:00:00+00:00"
   }
  },
  "price_source": "crypto_spot/1h@:00",
  "accuracy_field": {
   "A": {
    "n": 3670,
    "hit_rate": 0.4379,
    "wilson_95": [
     0.4219,
     0.454
    ]
   },
   "B": {
    "n": 5130,
    "hit_rate": 0.5405,
    "wilson_95": [
     0.5269,
     0.5541
    ]
   },
   "delta_pp": 10.3
  },
  "accuracy_per_model": [
   {
    "model": "claude-opus-5",
    "A": {
     "n": 361,
     "hit_rate": 0.4155,
     "wilson_95": [
      0.3658,
      0.467
     ]
    },
    "B": {
     "n": 672,
     "hit_rate": 0.5908,
     "wilson_95": [
      0.5532,
      0.6273
     ]
    },
    "delta_pp": 17.5
   },
   {
    "model": "claude-fable-5",
    "A": {
     "n": 472,
     "hit_rate": 0.4174,
     "wilson_95": [
      0.3737,
      0.4624
     ]
    },
    "B": {
     "n": 779,
     "hit_rate": 0.5648,
     "wilson_95": [
      0.5298,
      0.5992
     ]
    },
    "delta_pp": 14.7
   },
   {
    "model": "gpt-5.6-sol",
    "A": {
     "n": 636,
     "hit_rate": 0.4418,
     "wilson_95": [
      0.4037,
      0.4807
     ]
    },
    "B": {
     "n": 688,
     "hit_rate": 0.5363,
     "wilson_95": [
      0.499,
      0.5733
     ]
    },
    "delta_pp": 9.4
   },
   {
    "model": "deepseek-v4-pro",
    "A": {
     "n": 454,
     "hit_rate": 0.4251,
     "wilson_95": [
      0.3805,
      0.471
     ]
    },
    "B": {
     "n": 777,
     "hit_rate": 0.5161,
     "wilson_95": [
      0.481,
      0.5511
     ]
    },
    "delta_pp": 9.1
   },
   {
    "model": "qwen-3.8-max",
    "A": {
     "n": 572,
     "hit_rate": 0.4406,
     "wilson_95": [
      0.4004,
      0.4815
     ]
    },
    "B": {
     "n": 706,
     "hit_rate": 0.5227,
     "wilson_95": [
      0.4858,
      0.5593
     ]
    },
    "delta_pp": 8.2
   },
   {
    "model": "grok-4.5",
    "A": {
     "n": 620,
     "hit_rate": 0.4403,
     "wilson_95": [
      0.4017,
      0.4796
     ]
    },
    "B": {
     "n": 710,
     "hit_rate": 0.5141,
     "wilson_95": [
      0.4773,
      0.5507
     ]
    },
    "delta_pp": 7.4
   },
   {
    "model": "gemini-3.1-pro",
    "A": {
     "n": 555,
     "hit_rate": 0.4703,
     "wilson_95": [
      0.4291,
      0.5119
     ]
    },
    "B": {
     "n": 798,
     "hit_rate": 0.5414,
     "wilson_95": [
      0.5067,
      0.5756
     ]
    },
    "delta_pp": 7.1
   }
  ],
  "raw_sign_field": {
   "A": {
    "n_scored": 3632,
    "hit_rate": 0.4009
   },
   "B": {
    "n_scored": 5118,
    "hit_rate": 0.5348
   }
  },
  "abs_move_1d": {
   "A": 0.734,
   "B": 4.283,
   "delta_pct_rel": 483.5
  },
  "btc_realized_vol_ann_pct_hourly": {
   "A": 19.26,
   "B": 58.72
  },
  "trading_field": {
   "A": {
    "n_trades": 3670,
    "wr": 0.2965,
    "pnl_net_usd": -527.07,
    "pnl_gross_usd": -160.07
   },
   "B": {
    "n_trades": 5130,
    "wr": 0.4626,
    "pnl_net_usd": 889.37,
    "pnl_gross_usd": 1402.37
   }
  },
  "verdict": {
   "models_with_hit_improved": "7 of 7",
   "models_with_rawsign_improved": "7 of 7",
   "models_with_wr_improved": "7 of 7",
   "field_hit_delta_pp": 10.3,
   "field_pnl_net": {
    "A": -527.07,
    "B": 889.37
   },
   "abs_move_delta": {
    "1h": {
     "A": 0.181,
     "B": 0.541,
     "delta_pct_rel": 198.9
    },
    "4h": {
     "A": 0.367,
     "B": 1.314,
     "delta_pct_rel": 258.0
    },
    "1d": {
     "A": 0.734,
     "B": 4.283,
     "delta_pct_rel": 483.5
    }
   },
   "btc_vol_ann_pct": {
    "A": 19.26,
    "B": 58.72
   },
   "note": "descriptive, two adjacent calendar weeks; hit rule = methodology v1.1 trade-based; raw_sign = price-sign robustness check (closes at :00 vs slots at :01)"
  },
  "lineage_note": "no lineage stitching needed: the grok 4.5->4.6 flip (2026-08-24 09:22 UTC) is after this window -- weeks A and B are pure grok-4.5 (verified: 0 grok-4.6 rows in the pull)"
 },
 "why_it_matters": "MarketMania publishes forecasts from 7 frontier models into the same slots every week; a natural question for anyone reading the feed is whether a model's call is worth more when it agrees with what everyone else is saying, or whether that agreement is just everyone making the same mistake together. Consensus Watch answers that question empirically, one week at a time, using a leave-one-out design built specifically to avoid the obvious trap of comparing a model to an index that includes its own vote.",
 "how_to_read_this": "For every model M and every forecast it made, we rebuild that slot's consensus using only the OTHER active models in the same symbol / forecast-horizon (FH) / timeframe (TF) cell -- M's own call never counts toward its own consensus. M is scored agree if its side (long or short) matches the strict majority of those peers, and disagree if it sits alone against that majority; a cell needs at least 3 directional peers to count at all, and tied peer splits are excluded rather than forced either way. LOO peers are genuinely external to the model being scored, so the comparison below is not comparing a model to itself.",
 "practical_implications": [
  "Three weeks, three margin-curve shapes. No consensus tier has shown a stable meaning, so a consensus filter still should not carry weight in a config.",
  "The agree-vs-disagree lift finally rests on a readable disagree bucket (n=125) and it points the opposite way from issue #2. Two opposite signs in two weeks is what a regime-dependent description looks like; treat +17.1pp as this week's weather.",
  "The daily-horizon flip (0 for 28 -> 81.0% on 1d/1d unanimity) tracks the market waking up, not a change in how the herd forms. Read the two together, never the consensus number alone."
 ],
 "limitations": [
  "Single week (Aug 17-23, 2026 UTC). Every finding is descriptive for this window only.",
  "Observations are not independent: the same models watch overlapping symbol / FH / TF cells hour after hour. Wilson intervals here are descriptive, not inferential.",
  "Correlational, not causal: all models see the same market data, so 'agreement' and 'hit' can rise together simply because a slot was easy to read. LOO removes self-match bias, not this confound.",
  "The pooled lift rests on 125 disagreeing calls; per-model counts run 2-53 and two cell-level disagree buckets are N<10. 773 calls (15.1% of the 5,130-call mature directional universe) were thin or tied and are excluded from every consensus table.",
  "This window ran 5 trend days of 7 (issue #2: 1 of 7); every comparison with issue #2 crosses two different regimes. Series density and lineage notes (wave 3): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Aug 17-23) daily coverage is PARTIAL by design: the 1w series has the Mon Aug 17 anchor plus daily slots on Aug 22-23 only, and the 1M series has a daily slot on Aug 23 only; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 209, 1M 68). (2) After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "Market-state row is computed on a single-exchange (binance) daily candle series this issue, after an audit found the raw table mixes two exchanges; issue #2 row is as published.",
  "Research-to-date counter pinned from this issue: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Aug 24 16:00 UTC). Issue #2 printed 23,458 under an earlier, unpinned definition; treat cross-issue counter deltas across the pin as definitional, not additive."
 ],
 "notes": {
  "series_density_and_lineage_wave3": "Series density and lineage notes (wave 3): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Aug 17-23) daily coverage is PARTIAL by design: the 1w series has the Mon Aug 17 anchor plus daily slots on Aug 22-23 only, and the 1M series has a daily slot on Aug 23 only; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 209, 1M 68). (2) After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "grok_era": {
   "ids_in_window": [
    "grok-4.5"
   ],
   "n_grok45": 1370,
   "n_grok46": 0,
   "note": "After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5"
  },
  "model_ids_line": [
   "claude-fable-5",
   "claude-opus-5",
   "deepseek-v4-pro",
   "gemini-3.1-pro",
   "gpt-5.6-sol",
   "grok-4.5",
   "qwen-3.8-max"
  ],
  "week_over_week_note": "Week-over-week columns compare back-to-back windows: Aug 10-16 (issue #2, and week A of this issue's alive slice) vs Aug 17-23 (this issue). \"Was\" values are the numbers published in issue #2; deltas are computed on unrounded rates.",
  "research_to_date_pin": "Research-to-date counter pinned from this issue: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Aug 24 16:00 UTC). Issue #2 printed 23,458 under an earlier, unpinned definition; treat cross-issue counter deltas across the pin as definitional, not additive.",
  "market_state_audit": "Market-state row is computed on a single-exchange (binance) daily candle series this issue, after an audit found the raw table mixes two exchanges; issue #2 row is as published.",
  "engine": "Engine 1.1 has powered the sandbox since Aug 18, i.e. from day 2 of this window; no cross-engine PnL comparisons are claimed.",
  "report_template": "PDF built with the canonical mm_report.py template module; preview card with report_preview.render_preview"
 },
 "bibtex_key": "mm_consensus_2026w34",
 "citation": {
  "bibtex_key": "mm_consensus_2026w34",
  "title": "Consensus Watch #3: does agreeing with the crowd make an LLM's market call safer?",
  "author": "MarketMania Research",
  "year": 2026,
  "month": "August",
  "day": 26,
  "url": "https://marketmania.ai/research/reports/consensus-watch-2026-08-17.pdf",
  "note": "Methodology v1.1; window Aug 17-23, 2026 UTC; source weekly_metrics_2026-08-17.json"
 },
 "living_series_note": "Issue #3. Consensus Watch is a living weekly comparison; each issue appends one more week of leave-one-out agreement-vs-hit data. Issue #2's open question -- does the margin-curve inversion persist, or is tier ordering week-to-week noise? -- closed as 'noise so far': the curve took a third, different shape. The new result is the sign flip on the pooled lift, on the first readable disagree bucket of the series. After the report window — from Aug 24, 2026 — the grok line runs Grok 4.6; every grok forecast in this window and in the week-2 comparison is grok-4.5. Engine 1.1 has powered the sandbox since Aug 18, i.e. from day 2 of this window; no cross-engine PnL comparisons are claimed.",
 "testing_next": [
  "Next issue: does the agree-side lift survive a fourth week -- and does it survive a flat week, or is it a trend-regime artifact?",
  "Does the 4-peer dip repeat, or is it this week's noise in the only below-base tier?",
  "Monthly test (Sep 2): the agreement curve across regimes at monthly n."
 ],
 "related_research": [
  {
   "title": "Weekly Calibration #3",
   "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-08-17.pdf"
  },
  {
   "title": "Weekly Model Watch #3",
   "url": "https://marketmania.ai/research/reports/model-watch-2026-08-17.pdf"
  },
  {
   "title": "Consensus Watch #2",
   "url": "https://marketmania.ai/research/reports/consensus-watch-2026-08-10.pdf"
  }
 ],
 "related_research_note": "The three weekly reports publish together as one issue each week; each links straight to the others' PDF and to its own previous issue. Direct links are the posting rule from wave 2 on.",
 "research_to_date": {
  "this_report": {
   "loo_scored_observations": 4357
  },
  "platform": {
   "as_of_cutoff": "2026-08-24",
   "resolved_forecasts": 33809,
   "resolved_forecasts_definition": "directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, at the issue cutoff",
   "resolved_forecasts_since": "2026-07-11",
   "models_tracked": 11,
   "models_tracked_detail": "7 current + 4 archived",
   "assets": 5,
   "forecast_horizons": 5,
   "cadence": "hourly",
   "published_research_reports": 11
  },
  "source": "weekly_metrics_2026-08-17.json",
  "pin_note": "Research-to-date counter pinned from this issue: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Aug 24 16:00 UTC). Issue #2 printed 23,458 under an earlier, unpinned definition; treat cross-issue counter deltas across the pin as definitional, not additive."
 },
 "site_entry": {
  "slug": "consensus-watch-2026-08-17",
  "series": "weekly",
  "seriesLabel": "Weekly · Consensus",
  "date": "August 26, 2026",
  "title": "Consensus Watch #3",
  "summary": "9,260 mature forecasts asked whether agreeing with the crowd makes an LLM market call safer. Herding eased to 97.1% (was 99.8%) and, for the first time, leaving the crowd was the mistake: agree-hit 54.8% vs 37.6% on disagreement, against a 54.1% field base. The margin curve took its third shape in three weeks — the 4-peer tier sat below base while unanimity hit 56.6% — and daily-horizon unanimity flipped from 0-for-28 to 81.0% on 1d/1d. The tape woke up and accuracy followed: field 43.8% -> 54.1% (+10.3pp), 7 of 7 models improved.",
  "stats": [
   "4,357 peer-judged calls",
   "97.1% ran with the herd",
   "margin curve reshuffled"
  ],
  "files": {
   "pdf": "/research/reports/consensus-watch-2026-08-17.pdf",
   "md": "/research/reports/consensus-watch-2026-08-17.md",
   "json": "/research/reports/consensus-watch-2026-08-17.json"
  }
 },
 "social": {
  "telegram": {
   "photo_caption_html": "🧭 <b>Consensus Watch #3</b> — weekly series (window Aug 17–23)\n\nResearch question: if a model's forecast matches the consensus of the other models, is it more reliable?\n\n4,357 leave-one-out scored calls across 7 models:\n• Herding eased to <b>97.1%</b> (was 99.8%) — and disagreeing got punished: agree-hit 54.8% vs <b>37.6%</b> when a model left the crowd (field base 54.1%)\n• Margin curve reshuffled again: 3-peer 56.4%, 4-peer 48.9% (below base), unanimity 56.6% — third shape in three weeks\n• Daily-horizon unanimity flipped: 81.0% hit on 1d/1d, 61.9% on 1d/4h (was 0 for 28)\n• Market check: field accuracy 43.8% -> 54.1% (+10.3pp), 7 of 7 models improved; avg |1d| move 0.73% -> 4.28%\n\n📄 <a href=\"https://marketmania.ai/research/reports/consensus-watch-2026-08-17.pdf\">Full report (PDF)</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #consensus #research",
   "document_caption_html": "📄 Consensus Watch #3 — full report (PDF).\nWeb copy: <a href=\"https://marketmania.ai/research/reports/consensus-watch-2026-08-17.pdf\">consensus-watch-2026-08-17.pdf</a> · machine-readable: <a href=\"https://marketmania.ai/research/reports/consensus-watch-2026-08-17.json\">JSON</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #consensus #research",
   "caption_len": 918,
   "document_caption_len": 402,
   "limit": 1024
  },
  "x": {
   "main": "Consensus Watch #3 is out.\n\n4,357 peer-judged LLM market calls: herding eased to 97.1% — and this week leaving the crowd was the mistake: 54.8% with the herd vs 37.6% against it (+17.1pp).\n\nFull report: https://marketmania.ai/research/reports/consensus-watch-2026-08-17.pdf\n\n#AI #Crypto #Research #Consensus",
   "main_tco_len": 260,
   "reply": "Margin curve, w/w (hit):\n3 peers: 50.2 -> 56.4\n4 peers: 39.4 -> 48.9\n5 peers: 41.0 -> 54.4\nunanimity: 43.3 -> 56.6\nField base: 43.8 -> 54.1. Daily unanimity went from 0-for-28 to 81% on 1d/1d. LOO 3+ peers, methodology v1.1.\nData: https://marketmania.ai/research/reports/consensus-watch-2026-08-17.json",
   "reply_tco_len": 254,
   "limit_tco": 280,
   "replies_max": 1
  }
 }
}
