{
 "slug": "consensus-watch-2026-09-21",
 "series": "weekly",
 "series_label": "WEEKLY · CONSENSUS",
 "issue": 8,
 "title": "Consensus Watch #8",
 "language": "en",
 "window_start": "2026-09-21T00:00:00+00:00",
 "window_end": "2026-09-28T00:00:00+00:00",
 "window_text": "Sep 21-27, 2026 UTC",
 "cutoff": "2026-09-28T16:00:00+00:00",
 "generated_at": "2026-09-30T05:45:53+00:00",
 "methodology": "v1.1 (2026-08-10)",
 "methodology_hash": "e66c7e8c864a2233",
 "source": "weekly_metrics_2026-09-21.json",
 "research_question": "if a model's forecast matches the consensus of the other models, is the forecast more reliable?",
 "method": "leave-one-out strict majority, >=3 directional peers",
 "base_field_hit": {
  "n": 4958,
  "hit_rate": 0.4683
 },
 "market_state": {
  "panel": "market_state_6p2",
  "week": {
   "start": "2026-09-21",
   "end": "2026-09-28"
  },
  "prev_week": {
   "start": "2026-09-14",
   "end": "2026-09-21"
  },
  "current": {
   "btc_net_pct": 4.06,
   "realized_vol_ann_pct": 52.25,
   "volume_top5_usd_bn": 22.4265,
   "volume_top5_wow_pct": 19.7,
   "avg_pairwise_corr": 0.71
  },
  "previous_as_published_issue7": {
   "btc_net_pct": 5.64,
   "realized_vol_ann_pct": 51.01,
   "volume_top5_usd_bn": 18.7386,
   "avg_pairwise_corr": 0.93
  },
  "snapshot_line": "BTC net +4.06% (prior +5.64%) - ann vol 52.2% (was 51.0%) - TOP5 volume $22.43B, +20% w/w - pairwise corr 0.71 (was 0.93)",
  "exchange": "binance",
  "audit_filter_applied": true,
  "source": "single-exchange (binance) daily candle recompute, market_state_2026-09-21.json; prior-week values as published in issue #7",
  "audit_note": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #7 after the audit that found the raw table mixes two exchanges; issue #7 values are as published.",
  "prev_week_control": {
   "published_issue3": {
    "btc_net_pct": 5.64,
    "realized_vol_ann_pct": 51.01,
    "volume_top5_usd_bn": 18.7386,
    "avg_pairwise_corr": 0.93
   },
   "published_issue7": {
    "btc_net_pct": 5.64,
    "realized_vol_ann_pct": 51.01,
    "volume_top5_usd_bn": 18.7386,
    "avg_pairwise_corr": 0.93
   },
   "got": {
    "btc_net_pct": 5.64,
    "realized_vol_ann_pct": 51.01,
    "volume_top5_usd_bn": 18.7386,
    "avg_pairwise_corr": 0.93
   },
   "match": {
    "btc_net_pct": true,
    "realized_vol_ann_pct": true,
    "volume_top5_usd_bn": true,
    "avg_pairwise_corr": true
   },
   "tolerance": {
    "btc_net_pct": 0.05,
    "realized_vol_ann_pct": 0.5,
    "volume_top5_usd_bn": 0.05,
    "avg_pairwise_corr": 0.02
   },
   "all_match": true,
   "note": "previous week of this run == published issue-#7 week; the four Snapshot pins must reproduce (key published_issue3 kept for the builder schema, published_issue7 is the add-only alias)"
  },
  "regime_days": {
   "2026-09-21": {
    "move_pct": 6.39,
    "regime": "trend"
   },
   "2026-09-22": {
    "move_pct": -0.4,
    "regime": "flat"
   },
   "2026-09-23": {
    "move_pct": -2.04,
    "regime": "trend"
   },
   "2026-09-24": {
    "move_pct": 0.02,
    "regime": "flat"
   },
   "2026-09-25": {
    "move_pct": -0.46,
    "regime": "flat"
   },
   "2026-09-26": {
    "move_pct": 0.36,
    "regime": "flat"
   },
   "2026-09-27": {
    "move_pct": 0.2,
    "regime": "flat"
   }
  },
  "alive_slice_context": {
   "abs_move_1d_pct": {
    "A": 2.693,
    "B": 1.835
   },
   "btc_realized_vol_ann_pct_hourly": {
    "A": 35.07,
    "B": 35.8
   }
  }
 },
 "tables": {
  "pooled": {
   "agree": {
    "n": 4215,
    "hit_rate": 0.4598,
    "wilson_95": [
     0.4448,
     0.4749
    ]
   },
   "disagree": {
    "n": 30,
    "hit_rate": 0.6,
    "wilson_95": [
     0.4232,
     0.7541
    ]
   },
   "lift_pp": -14.0,
   "thin_or_tied": 713,
   "by_margin": {
    "2": {
     "n": 9,
     "hit_rate": 0.6667
    },
    "3": {
     "n": 304,
     "hit_rate": 0.3914
    },
    "4": {
     "n": 485,
     "hit_rate": 0.3897
    },
    "5": {
     "n": 876,
     "hit_rate": 0.468
    },
    "6": {
     "n": 2541,
     "hit_rate": 0.4778
    }
   },
   "total_scoreable": 4245,
   "caveat": "Read with context. The disagree bucket is n=30 this week against 52 in issue #7 and 26 in issue #6; the two Wilson intervals overlap. This is a single week of dependent observations: not evidence about herding in general, and not a claim that the sign will hold next week."
  },
  "by_cell": [
   {
    "agree": {
     "n": 2565,
     "hit_rate": 0.4651,
     "wilson_95": [
      0.4459,
      0.4844
     ]
    },
    "disagree": {
     "n": 21,
     "hit_rate": 0.5714,
     "wilson_95": [
      0.3655,
      0.7553
     ]
    },
    "lift_pp": -10.6,
    "thin_or_tied": 460,
    "by_margin": {
     "2": {
      "n": 6,
      "hit_rate": 0.5
     },
     "3": {
      "n": 232,
      "hit_rate": 0.3922
     },
     "4": {
      "n": 335,
      "hit_rate": 0.3791
     },
     "5": {
      "n": 564,
      "hit_rate": 0.4362
     },
     "6": {
      "n": 1428,
      "hit_rate": 0.5084
     }
    },
    "n_subjects": 3046,
    "cell": "fh_1h_tf_1h",
    "fh": "1h",
    "tf": "1h"
   },
   {
    "agree": {
     "n": 677,
     "hit_rate": 0.4815,
     "wilson_95": [
      0.4441,
      0.5192
     ]
    },
    "disagree": {
     "n": 2,
     "hit_rate": 0.5,
     "wilson_95": [
      0.0945,
      0.9055
     ]
    },
    "lift_pp": -1.8,
    "thin_or_tied": 94,
    "by_margin": {
     "2": {
      "n": 3,
      "hit_rate": 1.0
     },
     "3": {
      "n": 36,
      "hit_rate": 0.3889
     },
     "4": {
      "n": 65,
      "hit_rate": 0.6462
     },
     "5": {
      "n": 132,
      "hit_rate": 0.5758
     },
     "6": {
      "n": 441,
      "hit_rate": 0.4331
     }
    },
    "n_subjects": 773,
    "cell": "fh_4h_tf_4h",
    "fh": "4h",
    "tf": "4h"
   },
   {
    "agree": {
     "n": 641,
     "hit_rate": 0.4212,
     "wilson_95": [
      0.3836,
      0.4598
     ]
    },
    "disagree": {
     "n": 3,
     "hit_rate": 1.0,
     "wilson_95": [
      0.4385,
      1.0
     ]
    },
    "lift_pp": -57.9,
    "thin_or_tied": 138,
    "by_margin": {
     "3": {
      "n": 28,
      "hit_rate": 0.4643
     },
     "4": {
      "n": 65,
      "hit_rate": 0.3077
     },
     "5": {
      "n": 114,
      "hit_rate": 0.4561
     },
     "6": {
      "n": 434,
      "hit_rate": 0.4263
     }
    },
    "n_subjects": 782,
    "cell": "fh_4h_tf_1h",
    "fh": "4h",
    "tf": "1h"
   },
   {
    "agree": {
     "n": 192,
     "hit_rate": 0.4583,
     "wilson_95": [
      0.3894,
      0.5289
     ]
    },
    "disagree": {
     "n": 4,
     "hit_rate": 0.5,
     "wilson_95": [
      0.15,
      0.85
     ]
    },
    "lift_pp": -4.2,
    "thin_or_tied": 10,
    "by_margin": {
     "4": {
      "n": 15,
      "hit_rate": 0.0
     },
     "5": {
      "n": 30,
      "hit_rate": 0.4
     },
     "6": {
      "n": 147,
      "hit_rate": 0.517
     }
    },
    "n_subjects": 206,
    "cell": "fh_1d_tf_1d",
    "fh": "1d",
    "tf": "1d"
   },
   {
    "agree": {
     "n": 140,
     "hit_rate": 0.4357,
     "wilson_95": [
      0.3564,
      0.5185
     ]
    },
    "disagree": {
     "n": 0,
     "hit_rate": null,
     "wilson_95": [
      null,
      null
     ]
    },
    "lift_pp": null,
    "thin_or_tied": 11,
    "by_margin": {
     "3": {
      "n": 8,
      "hit_rate": 0.125
     },
     "4": {
      "n": 5,
      "hit_rate": 0.0
     },
     "5": {
      "n": 36,
      "hit_rate": 0.6667
     },
     "6": {
      "n": 91,
      "hit_rate": 0.3956
     }
    },
    "n_subjects": 151,
    "cell": "fh_1d_tf_4h",
    "fh": "1d",
    "tf": "4h"
   }
  ],
  "by_margin": [
   {
    "margin": 2,
    "n": 9,
    "hit_rate": 0.6667
   },
   {
    "margin": 3,
    "n": 304,
    "hit_rate": 0.3914
   },
   {
    "margin": 4,
    "n": 485,
    "hit_rate": 0.3897
   },
   {
    "margin": 5,
    "n": 876,
    "hit_rate": 0.468
   },
   {
    "margin": 6,
    "n": 2541,
    "hit_rate": 0.4778
   }
  ],
  "by_margin_week_over_week": [
   {
    "margin": 3,
    "issue7_hit_rate_pct": 52.1,
    "issue8_hit_rate": 0.3914,
    "issue8_n": 304
   },
   {
    "margin": 4,
    "issue7_hit_rate_pct": 55.5,
    "issue8_hit_rate": 0.3897,
    "issue8_n": 485
   },
   {
    "margin": 5,
    "issue7_hit_rate_pct": 44.2,
    "issue8_hit_rate": 0.468,
    "issue8_n": 876
   },
   {
    "margin": 6,
    "issue7_hit_rate_pct": 51.6,
    "issue8_hit_rate": 0.4778,
    "issue8_n": 2541
   }
  ],
  "per_model": [
   {
    "agree": {
     "n": 620,
     "hit_rate": 0.4726,
     "wilson_95": [
      0.4336,
      0.5119
     ]
    },
    "disagree": {
     "n": 2,
     "hit_rate": 0.0,
     "wilson_95": [
      0.0,
      0.6576
     ]
    },
    "lift_pp": 47.3,
    "thin_or_tied": 114,
    "by_margin": {
     "2": {
      "n": 1,
      "hit_rate": 1.0
     },
     "3": {
      "n": 38,
      "hit_rate": 0.4211
     },
     "4": {
      "n": 77,
      "hit_rate": 0.4156
     },
     "5": {
      "n": 141,
      "hit_rate": 0.4823
     },
     "6": {
      "n": 363,
      "hit_rate": 0.4848
     }
    },
    "model": "claude-fable-5"
   },
   {
    "agree": {
     "n": 585,
     "hit_rate": 0.4752,
     "wilson_95": [
      0.435,
      0.5157
     ]
    },
    "disagree": {
     "n": 1,
     "hit_rate": 0.0,
     "wilson_95": [
      0.0,
      0.7935
     ]
    },
    "lift_pp": 47.5,
    "thin_or_tied": 34,
    "by_margin": {
     "2": {
      "n": 3,
      "hit_rate": 0.6667
     },
     "3": {
      "n": 30,
      "hit_rate": 0.4
     },
     "4": {
      "n": 57,
      "hit_rate": 0.3684
     },
     "5": {
      "n": 132,
      "hit_rate": 0.4773
     },
     "6": {
      "n": 363,
      "hit_rate": 0.4959
     }
    },
    "model": "claude-opus-5"
   },
   {
    "agree": {
     "n": 618,
     "hit_rate": 0.4417,
     "wilson_95": [
      0.4031,
      0.4811
     ]
    },
    "disagree": {
     "n": 5,
     "hit_rate": 0.4,
     "wilson_95": [
      0.1176,
      0.7693
     ]
    },
    "lift_pp": 4.2,
    "thin_or_tied": 154,
    "by_margin": {
     "2": {
      "n": 1,
      "hit_rate": 0.0
     },
     "3": {
      "n": 53,
      "hit_rate": 0.3962
     },
     "4": {
      "n": 80,
      "hit_rate": 0.4125
     },
     "5": {
      "n": 121,
      "hit_rate": 0.4628
     },
     "6": {
      "n": 363,
      "hit_rate": 0.449
     }
    },
    "model": "deepseek-v4-pro"
   },
   {
    "agree": {
     "n": 625,
     "hit_rate": 0.4832,
     "wilson_95": [
      0.4442,
      0.5224
     ]
    },
    "disagree": {
     "n": 16,
     "hit_rate": 0.75,
     "wilson_95": [
      0.505,
      0.8982
     ]
    },
    "lift_pp": -26.7,
    "thin_or_tied": 153,
    "by_margin": {
     "2": {
      "n": 2,
      "hit_rate": 0.5
     },
     "3": {
      "n": 53,
      "hit_rate": 0.4528
     },
     "4": {
      "n": 82,
      "hit_rate": 0.4268
     },
     "5": {
      "n": 125,
      "hit_rate": 0.488
     },
     "6": {
      "n": 363,
      "hit_rate": 0.4986
     }
    },
    "model": "gemini-3.1-pro"
   },
   {
    "agree": {
     "n": 636,
     "hit_rate": 0.4638,
     "wilson_95": [
      0.4254,
      0.5027
     ]
    },
    "disagree": {
     "n": 3,
     "hit_rate": 0.6667,
     "wilson_95": [
      0.2077,
      0.9385
     ]
    },
    "lift_pp": -20.3,
    "thin_or_tied": 105,
    "by_margin": {
     "2": {
      "n": 2,
      "hit_rate": 1.0
     },
     "3": {
      "n": 54,
      "hit_rate": 0.2963
     },
     "4": {
      "n": 78,
      "hit_rate": 0.3846
     },
     "5": {
      "n": 139,
      "hit_rate": 0.4748
     },
     "6": {
      "n": 363,
      "hit_rate": 0.4986
     }
    },
    "model": "gpt-5.6-sol"
   },
   {
    "agree": {
     "n": 495,
     "hit_rate": 0.4485,
     "wilson_95": [
      0.4052,
      0.4925
     ]
    },
    "disagree": {
     "n": 3,
     "hit_rate": 0.6667,
     "wilson_95": [
      0.2077,
      0.9385
     ]
    },
    "lift_pp": -21.8,
    "thin_or_tied": 39,
    "by_margin": {
     "3": {
      "n": 20,
      "hit_rate": 0.35
     },
     "4": {
      "n": 35,
      "hit_rate": 0.3429
     },
     "5": {
      "n": 77,
      "hit_rate": 0.4286
     },
     "6": {
      "n": 363,
      "hit_rate": 0.4683
     }
    },
    "model": "grok-4.6"
   },
   {
    "agree": {
     "n": 636,
     "hit_rate": 0.4324,
     "wilson_95": [
      0.3944,
      0.4712
     ]
    },
    "disagree": {
     "n": 0,
     "hit_rate": null,
     "wilson_95": [
      null,
      null
     ]
    },
    "lift_pp": null,
    "thin_or_tied": 114,
    "by_margin": {
     "3": {
      "n": 56,
      "hit_rate": 0.4107
     },
     "4": {
      "n": 76,
      "hit_rate": 0.3421
     },
     "5": {
      "n": 141,
      "hit_rate": 0.4468
     },
     "6": {
      "n": 363,
      "hit_rate": 0.449
     }
    },
    "model": "qwen-3.8-max"
   }
  ],
  "herd_worst_cards": [
   {
    "slot": "2026-09-26T00:01:00+00:00",
    "symbol": "SOL",
    "fh": "1d",
    "tf": "1d",
    "side": "long",
    "n_models": 7,
    "mean_conf": 70.4,
    "hit_rate": 0.0
   },
   {
    "slot": "2026-09-22T00:01:00+00:00",
    "symbol": "SOL",
    "fh": "1d",
    "tf": "1d",
    "side": "long",
    "n_models": 7,
    "mean_conf": 70.4,
    "hit_rate": 0.0
   },
   {
    "slot": "2026-09-21T01:01:00+00:00",
    "symbol": "SOL",
    "fh": "1h",
    "tf": "1h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 69.7,
    "hit_rate": 0.1429
   },
   {
    "slot": "2026-09-21T01:01:00+00:00",
    "symbol": "XRP",
    "fh": "1h",
    "tf": "1h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 69.1,
    "hit_rate": 0.0
   },
   {
    "slot": "2026-09-26T16:01:00+00:00",
    "symbol": "SOL",
    "fh": "4h",
    "tf": "4h",
    "side": "long",
    "n_models": 7,
    "mean_conf": 69,
    "hit_rate": 0.0
   }
  ],
  "counters": {
   "forecasts_total": 10290,
   "ok_in_gate": 9273,
   "out_of_gate": 973,
   "out_of_gate_by_fh": {
    "1w": 488,
    "1m": 485
   },
   "invalid": 44,
   "mature": 9273,
   "mature_directional": 4958,
   "mature_sideways": 4315,
   "pending_next_issue": 0,
   "pending_by_fh": {},
   "late_closes": 0,
   "uptime": [
    {
     "fh": "1h",
     "tf": "1h",
     "slots_seen": 168,
     "slots_expected": 168
    },
    {
     "fh": "4h",
     "tf": "4h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "4h",
     "tf": "1h",
     "slots_seen": 42,
     "slots_expected": 42
    },
    {
     "fh": "1d",
     "tf": "1d",
     "slots_seen": 7,
     "slots_expected": 7
    },
    {
     "fh": "1d",
     "tf": "4h",
     "slots_seen": 7,
     "slots_expected": 7
    }
   ],
   "regime_days": {
    "2026-09-21": {
     "move_pct": 6.39,
     "regime": "trend"
    },
    "2026-09-22": {
     "move_pct": -0.4,
     "regime": "flat"
    },
    "2026-09-23": {
     "move_pct": -2.04,
     "regime": "trend"
    },
    "2026-09-24": {
     "move_pct": 0.02,
     "regime": "flat"
    },
    "2026-09-25": {
     "move_pct": -0.46,
     "regime": "flat"
    },
    "2026-09-26": {
     "move_pct": 0.36,
     "regime": "flat"
    },
    "2026-09-27": {
     "move_pct": 0.2,
     "regime": "flat"
    }
   },
   "grok_era": {
    "ids_in_window": [
     "grok-4.6"
    ],
    "n_grok45": 0,
    "n_grok46": 1470,
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470,
    "first_grok46_slot": "2026-09-21T00:01:00+00:00",
    "last_grok45_slot": null,
    "first_grok45_slot": null,
    "flip_at": "2026-08-24T09:22:00+00:00",
    "flip_inside_window": false,
    "prev_window": {
     "n_grok45_rows": 0,
     "n_grok46_rows": 1470
    },
    "merged_model_id": "grok-4.6",
    "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "lineage_ids": [
     "grok-4.5",
     "grok-4.6"
    ],
    "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
    "lineage_warnings": [],
    "rows_remapped_a_plus_b": 0,
    "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- four windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #7) and week B (this window) are pure grok-4.6.",
    "cells_with_two_grok_rows": 0
   },
   "series_density": {
    "1d": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "1h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 168,
     "n_rows": 5880,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "1m": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "1w": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": false,
     "n_slots": 7,
     "n_rows": 490,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    },
    "4h": {
     "days_with_slots": 7,
     "expected_days": 7,
     "full_daily_coverage": true,
     "in_fh_gate": true,
     "n_slots": 42,
     "n_rows": 2940,
     "days": [
      "2026-09-21",
      "2026-09-22",
      "2026-09-23",
      "2026-09-24",
      "2026-09-25",
      "2026-09-26",
      "2026-09-27"
     ],
     "missing_days": []
    }
   },
   "model_roster": {
    "current": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "current_count": 7,
    "current_raw_ids_in_window": [
     "claude-fable-5",
     "claude-opus-5",
     "deepseek-v4-pro",
     "gemini-3.1-pro",
     "gpt-5.6-sol",
     "grok-4.6",
     "qwen-3.8-max"
    ],
    "archived_legacy": [
     "grok-4.5",
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "archived_legacy_count": 3,
    "archived_legacy_detail": [
     {
      "model": "grok-4.5",
      "n_rows_since": 11128,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-24T09:01:00+00:00"
     },
     {
      "model": "qwen-3.7-max",
      "n_rows_since": 7479,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-08-05T09:01:00+00:00"
     },
     {
      "model": "claude-opus-4.8",
      "n_rows_since": 6315,
      "first_slot": "2026-07-15T18:01:00+00:00",
      "last_slot": "2026-07-30T08:01:00+00:00"
     }
    ],
    "archived_excluding_lineage_merged": [
     "qwen-3.7-max",
     "claude-opus-4.8"
    ],
    "tracked_total": 10,
    "canon_expected": {
     "current": 7,
     "archived_legacy": 4,
     "tracked_total": 11
    },
    "matches_canon_11_7_4": false,
    "roster_since": "2026-07-11T00:00:00+00:00",
    "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
   }
  }
 },
 "herd_note": "505 cells had 6+ directional models on one side; unanimous_cells_ge6 per weekly metrics",
 "smart_note": "Smart-vs-Outlook deferred: as-of weights pipeline not yet recording; will appear once weights are logged per-slot (methodology 3.2.3 fallback)",
 "key_finding": {
  "label": "KEY FINDING · OBSERVATION (one weekly window)",
  "text": "Herding ran at 99.3% of leave-one-out scoreable calls and the pooled lift came in at -14.0pp: agree-hit 46.0% against 60.0% on disagreement, on a disagree bucket of n=30.",
  "evidence_level": "observation"
 },
 "key_findings": [
  "**OBSERVATION -- 99.3% of the field's leave-one-out calls ran with the herd.** Of 4,245 LOO-scoreable directional calls, 4,215 (99.3%) sided with peers' majority (issue #7: 98.9%); agree-hit was 46.0% [44.5%, 47.5%] vs. 60.0% [42.3%, 75.4%] for the 30 disagreers, a pooled lift of -14.0pp. Descriptive, one week, not a claim of a durable edge.",
  "**The margin curve took an eighth reading.** 3 peers 39.1% (was 52.1%), 4 peers 39.0% (was 55.5%), 5 peers 46.8% (was 44.2%), full unanimity 47.8% (was 51.6%). 1 of the 4 tiers sit above the 46.8% field base, and 2 of the 4 kept the side of the base they took in issue #7.",
  "**Daily-horizon unanimity split around the field base.** 1d/1d margin-6 hit 51.7% (n=147) and 1d/4h margin-6 hit 39.6% (n=91); issue #7 printed 38.4% and 53.8% on the same two cells.",
  "**Herd failures did not disappear.** 505 cells had 6+ directional models on one side (issue #7: 539); the 5 worst sat at 69.0-70.4 mean confidence and resolved 0.0%-14.3%."
 ],
 "market_check": {
  "label": "MARKET CHECK · OBSERVATION (two adjacent calendar weeks)",
  "text": "OBSERVATION — mean |1d move| 2.69% -> 1.83% (-32% rel), BTC realized vol 35.1% -> 35.8% (ann., hourly); field directional accuracy 50.7% -> 46.8% (-3.9pp), 0 of 7 models improved, sim win-rate up for 0 of 7, field sim PnL -$337.72 -> -$315.68 (adjacent calendar weeks Sep 14-20 vs Sep 21-27; descriptive, one pair of weeks, not a claim).",
  "robustness": "Robustness: raw price-sign accuracy 54.2% -> 49.2% — the same direction as the trade-based hit rule.",
  "weeks": {
   "A": {
    "days": "2026-09-14..2026-09-20",
    "slots": [
     "2026-09-14T00:00:00+00:00",
     "2026-09-21T00:00:00+00:00"
    ],
    "cutoff": "2026-09-21T16:00:00+00:00"
   },
   "B": {
    "days": "2026-09-21..2026-09-27",
    "slots": [
     "2026-09-21T00:00:00+00:00",
     "2026-09-28T00:00:00+00:00"
    ],
    "cutoff": "2026-09-28T16:00:00+00:00"
   }
  },
  "price_source": "crypto_spot/1h@:00",
  "accuracy_field": {
   "A": {
    "n": 5598,
    "hit_rate": 0.5075,
    "wilson_95": [
     0.4944,
     0.5206
    ]
   },
   "B": {
    "n": 4958,
    "hit_rate": 0.4683,
    "wilson_95": [
     0.4545,
     0.4822
    ]
   },
   "delta_pp": -3.9
  },
  "accuracy_per_model": [
   {
    "model": "gemini-3.1-pro",
    "model_label": "gemini-3.1-pro",
    "A": {
     "n": 883,
     "hit_rate": 0.5289,
     "wilson_95": [
      0.4959,
      0.5616
     ]
    },
    "B": {
     "n": 794,
     "hit_rate": 0.5139,
     "wilson_95": [
      0.4791,
      0.5485
     ]
    },
    "delta_pp": -1.5
   },
   {
    "model": "grok-4.6",
    "model_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
    "A": {
     "n": 629,
     "hit_rate": 0.496,
     "wilson_95": [
      0.4571,
      0.535
     ]
    },
    "B": {
     "n": 537,
     "hit_rate": 0.4693,
     "wilson_95": [
      0.4274,
      0.5116
     ]
    },
    "delta_pp": -2.7
   },
   {
    "model": "gpt-5.6-sol",
    "model_label": "gpt-5.6-sol",
    "A": {
     "n": 821,
     "hit_rate": 0.497,
     "wilson_95": [
      0.4628,
      0.5311
     ]
    },
    "B": {
     "n": 744,
     "hit_rate": 0.4677,
     "wilson_95": [
      0.4321,
      0.5037
     ]
    },
    "delta_pp": -2.9
   },
   {
    "model": "claude-opus-5",
    "model_label": "claude-opus-5",
    "A": {
     "n": 711,
     "hit_rate": 0.5162,
     "wilson_95": [
      0.4795,
      0.5527
     ]
    },
    "B": {
     "n": 620,
     "hit_rate": 0.4758,
     "wilson_95": [
      0.4368,
      0.5151
     ]
    },
    "delta_pp": -4.0
   },
   {
    "model": "deepseek-v4-pro",
    "model_label": "deepseek-v4-pro",
    "A": {
     "n": 936,
     "hit_rate": 0.484,
     "wilson_95": [
      0.4521,
      0.516
     ]
    },
    "B": {
     "n": 777,
     "hit_rate": 0.4389,
     "wilson_95": [
      0.4044,
      0.474
     ]
    },
    "delta_pp": -4.5
   },
   {
    "model": "claude-fable-5",
    "model_label": "claude-fable-5",
    "A": {
     "n": 799,
     "hit_rate": 0.5232,
     "wilson_95": [
      0.4885,
      0.5576
     ]
    },
    "B": {
     "n": 736,
     "hit_rate": 0.4674,
     "wilson_95": [
      0.4316,
      0.5035
     ]
    },
    "delta_pp": -5.6
   },
   {
    "model": "qwen-3.8-max",
    "model_label": "qwen-3.8-max",
    "A": {
     "n": 819,
     "hit_rate": 0.5079,
     "wilson_95": [
      0.4737,
      0.5421
     ]
    },
    "B": {
     "n": 750,
     "hit_rate": 0.4453,
     "wilson_95": [
      0.4101,
      0.4811
     ]
    },
    "delta_pp": -6.3
   }
  ],
  "raw_sign_field": {
   "A": {
    "n_scored": 5583,
    "hit_rate": 0.5416
   },
   "B": {
    "n_scored": 4954,
    "hit_rate": 0.4923
   }
  },
  "abs_move_1d": {
   "A": 2.693,
   "B": 1.835,
   "delta_pct_rel": -31.9
  },
  "btc_realized_vol_ann_pct_hourly": {
   "A": 35.07,
   "B": 35.8
  },
  "trading_field": {
   "A": {
    "n_trades": 5598,
    "wr": 0.4337,
    "pnl_net_usd": -337.72,
    "pnl_gross_usd": 222.08
   },
   "B": {
    "n_trades": 4958,
    "wr": 0.3998,
    "pnl_net_usd": -315.68,
    "pnl_gross_usd": 180.12
   }
  },
  "verdict": {
   "models_with_hit_improved": "0 of 7",
   "models_with_rawsign_improved": "0 of 7",
   "models_with_wr_improved": "0 of 7",
   "field_hit_delta_pp": -3.9,
   "field_pnl_net": {
    "A": -337.72,
    "B": -315.68
   },
   "abs_move_delta": {
    "1h": {
     "A": 0.369,
     "B": 0.358,
     "delta_pct_rel": -3.0
    },
    "4h": {
     "A": 0.835,
     "B": 0.766,
     "delta_pct_rel": -8.3
    },
    "1d": {
     "A": 2.693,
     "B": 1.835,
     "delta_pct_rel": -31.9
    }
   },
   "btc_vol_ann_pct": {
    "A": 35.07,
    "B": 35.8
   },
   "note": "descriptive, two adjacent calendar weeks; hit rule = methodology v1.1 trade-based; raw_sign = price-sign robustness check (closes at :00 vs slots at :01)"
  },
  "lineage_note": "lineage splice ACTIVE: grok-4.5 rows are aggregated into grok-4.6 (flip 2026-08-24T09:22:00+00:00, four windows back, inside issue #4's window). Week A (2026-09-14..2026-09-20) is pure grok-4.6 (it reproduces the published issue #7); week B is pure grok-4.6."
 },
 "why_it_matters": "MarketMania publishes forecasts from 7 frontier model lines into the same slots every week; a natural question for anyone reading the feed is whether a model's call is worth more when it agrees with what everyone else is saying, or whether that agreement is just everyone making the same mistake together. Consensus Watch answers that question empirically, one week at a time, using a leave-one-out design built specifically to avoid the obvious trap of comparing a model to an index that includes its own vote.",
 "how_to_read_this": "For every model M and every forecast it made, we rebuild that slot's consensus using only the OTHER active models in the same symbol / forecast-horizon (FH) / timeframe (TF) cell -- M's own call never counts toward its own consensus. M is scored agree if its side (long or short) matches the strict majority of those peers, and disagree if it sits alone against that majority; a cell needs at least 3 directional peers to count at all, and tied peer splits are excluded rather than forced either way. LOO peers are genuinely external to the model being scored, so the comparison below is not comparing a model to itself. The grok rows are one line, so grok never counts as two peers in a cell. The grok 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits four windows back (issue #4): both weeks of this issue are pure grok-4.6, with no 4.5-era rows in either window, so the grok week-over-week row is the third pure-4.6 against pure-4.6 comparison of the series.",
 "practical_implications": [
  "Eight issues, eight margin-curve readings. The tier ordering moved again -- full unanimity took the top and the 4-peer tier fell from the top to the bottom -- and three of the four tiers printed a lower rate than in issue #7 (4 peers 55.5% -> 39.0%, full unanimity 51.6% -> 47.8%), so a consensus filter still should not carry weight in a config.",
  "The pooled lift is -14.0pp this week against +12.1pp in issue #7 and -30.4pp in issue #6. The sign flipped back below zero after one issue above it: treat the number as this week's weather, not an edge.",
  "Read the consensus number together with the market check on the same page. The field base moved -3.9pp week over week; 1 of the 4 consensus tiers moved up and 3 down against issue #7 (3 peers -13.0pp, 4 peers -16.5pp, 5 peers +2.6pp, unanimity -3.8pp)."
 ],
 "limitations": [
  "Single week (Sep 21-27, 2026 UTC). Every finding is descriptive for this window only.",
  "Observations are not independent: the same models watch overlapping symbol / FH / TF cells hour after hour. Wilson intervals here are descriptive, not inferential.",
  "Correlational, not causal: all models see the same market data, so 'agreement' and 'hit' can rise together simply because a slot was easy to read. LOO removes self-match bias, not this confound.",
  "The pooled lift rests on 30 disagreeing calls; per-model counts run 0-16. 713 calls (14.4% of the 4,958-call mature directional universe) were thin or tied and are excluded from every consensus table.",
  "This window ran 2 trend days of 7 (issue #7: 3 of 7); every comparison with issue #7 is a comparison across regimes as well as across weeks. Series density and lineage notes (wave 8): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 21-27) daily coverage is FULL, as it was in issues #4, #5, #6 and #7: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 488, 1M 485). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, four windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#7 window), so every grok number in this issue and in issue #7 is a pure grok-4.6 line -- the grok week-over-week row is the third pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #7 after the audit that found the raw table mixes two exchanges; issue #7 values are as published.",
  "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 28 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6 and #7, which printed 52,971 at the Sep 21 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 57,929 is an additive step under one definition; counter deltas against issue #6 and earlier remain definitional."
 ],
 "notes": {
  "series_density_and_lineage_wave8": "Series density and lineage notes (wave 8): (1) rolling-1w forecasts emit daily since Aug 22, 2026 and rolling-1M daily since Aug 23, 2026, so inside this window (Sep 21-27) daily coverage is FULL, as it was in issues #4, #5, #6 and #7: the 1w series has slots on 7 of 7 days and the 1M series on 7 of 7 days; both sit outside this report's FH gate (1h/4h/1d) and appear only in the exclusions counter (1w 488, 1M 485). (2) The grok line flipped 4.5 -> 4.6 at Aug 24, 2026 09:22 UTC, four windows back (inside the issue-#4 window): this window carries 1,470 grok rows and none from the 4.5 era, and neither does week A (the issue-#7 window), so every grok number in this issue and in issue #7 is a pure grok-4.6 line -- the grok week-over-week row is the third pure-4.6 against pure-4.6 comparison of the series. (3) qwen-3.8-max succeeded qwen-3.7-max on Aug 5, 2026 (fully before this window).",
  "grok_era": {
   "ids_in_window": [
    "grok-4.6"
   ],
   "n_grok45": 0,
   "n_grok46": 1470,
   "n_grok45_rows": 0,
   "n_grok46_rows": 1470,
   "first_grok46_slot": "2026-09-21T00:01:00+00:00",
   "last_grok45_slot": null,
   "first_grok45_slot": null,
   "flip_at": "2026-08-24T09:22:00+00:00",
   "flip_inside_window": false,
   "prev_window": {
    "n_grok45_rows": 0,
    "n_grok46_rows": 1470
   },
   "merged_model_id": "grok-4.6",
   "merged_label": "grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)",
   "lineage_ids": [
    "grok-4.5",
    "grok-4.6"
   ],
   "lineage_source": "benchmarks.config.lineage_ids('grok-4.6') + MODEL_SUCCESSION[grok-4.6]=('grok-4.5',)",
   "lineage_warnings": [],
   "rows_remapped_a_plus_b": 0,
   "note": "grok-4.6 went live 2026-08-24T09:22:00+00:00 -- four windows back (inside issue #4's window); all per-model aggregates use the lineage splice grok-4.5 -> grok-4.6 (one row, labelled 'grok-4.6 (incl. 4.5-era, Aug 24 00:00-09:22 UTC)'). Both week A (prev window, issue #7) and week B (this window) are pure grok-4.6.",
   "cells_with_two_grok_rows": 0
  },
  "model_ids_line": [
   "claude-fable-5",
   "claude-opus-5",
   "deepseek-v4-pro",
   "gemini-3.1-pro",
   "gpt-5.6-sol",
   "grok-4.6",
   "qwen-3.8-max"
  ],
  "model_roster": {
   "current": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "current_count": 7,
   "current_raw_ids_in_window": [
    "claude-fable-5",
    "claude-opus-5",
    "deepseek-v4-pro",
    "gemini-3.1-pro",
    "gpt-5.6-sol",
    "grok-4.6",
    "qwen-3.8-max"
   ],
   "archived_legacy": [
    "grok-4.5",
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "archived_legacy_count": 3,
   "archived_legacy_detail": [
    {
     "model": "grok-4.5",
     "n_rows_since": 11128,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-24T09:01:00+00:00"
    },
    {
     "model": "qwen-3.7-max",
     "n_rows_since": 7479,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-08-05T09:01:00+00:00"
    },
    {
     "model": "claude-opus-4.8",
     "n_rows_since": 6315,
     "first_slot": "2026-07-15T18:01:00+00:00",
     "last_slot": "2026-07-30T08:01:00+00:00"
    }
   ],
   "archived_excluding_lineage_merged": [
    "qwen-3.7-max",
    "claude-opus-4.8"
   ],
   "tracked_total": 10,
   "canon_expected": {
    "current": 7,
    "archived_legacy": 4,
    "tracked_total": 11
   },
   "matches_canon_11_7_4": false,
   "roster_since": "2026-07-11T00:00:00+00:00",
   "note": "current = merged-lineage model ids emitting inside the window (grok-4.5 rows are spliced into grok-4.6); archived legacy = raw ids seen since 2026-07-11 that no longer emit in the window -- grok-4.5 is one of them by raw id even though its rows are spliced. Models retired before 2026-07-11 are not visible to this query."
  },
  "series_density": {
   "1d": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "1h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 168,
    "n_rows": 5880,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "1m": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "1w": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": false,
    "n_slots": 7,
    "n_rows": 490,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   },
   "4h": {
    "days_with_slots": 7,
    "expected_days": 7,
    "full_daily_coverage": true,
    "in_fh_gate": true,
    "n_slots": 42,
    "n_rows": 2940,
    "days": [
     "2026-09-21",
     "2026-09-22",
     "2026-09-23",
     "2026-09-24",
     "2026-09-25",
     "2026-09-26",
     "2026-09-27"
    ],
    "missing_days": []
   }
  },
  "week_over_week_note": "Week-over-week columns compare back-to-back windows: Sep 14-20 (issue #7, and week A of this issue's alive slice) vs Sep 21-27 (this issue). \"Was\" values are the numbers published in issue #7; deltas are computed on unrounded rates.",
  "research_to_date_pin": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 28 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6 and #7, which printed 52,971 at the Sep 21 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 57,929 is an additive step under one definition; counter deltas against issue #6 and earlier remain definitional.",
  "market_state_audit": "Market-state row is computed on a single-exchange (binance) daily candle series, as in issue #7 after the audit that found the raw table mixes two exchanges; issue #7 values are as published.",
  "engine": "Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
  "report_template": "PDF built with the canonical mm_report.py template module; preview card with report_preview.render_preview"
 },
 "bibtex_key": "mm_consensus_2026w39",
 "citation": {
  "bibtex_key": "mm_consensus_2026w39",
  "title": "Consensus Watch #8: does agreeing with the crowd make an LLM's market call safer?",
  "author": "MarketMania Research",
  "year": 2026,
  "month": "September",
  "day": 30,
  "url": "https://marketmania.ai/research/reports/consensus-watch-2026-09-21.pdf",
  "note": "Methodology v1.1; window Sep 21-27, 2026 UTC; source weekly_metrics_2026-09-21.json"
 },
 "living_series_note": "**Issue #8.** Consensus Watch is a living weekly comparison; each issue appends one more week of leave-one-out agreement-vs-hit data. Issue #7 asked \"the pooled agree-vs-disagree lift flipped to +12.1pp after three issues below zero -- does the positive sign hold for a second issue?\" -- it reads -14.0pp this week after +12.1pp in issue #7 (agree 46.0% vs disagree 60.0% on n=30), back below zero; and \"The 4-peer tier has sat above the field base two issues running -- does it hold for a third?\" -- 2 of the 4 tiers kept their issue-#7 side of the base, and the 4-peer tier came in below it. The grok row is pure grok-4.6 this issue and was pure grok-4.6 in issue #7; the 4.5 -> 4.6 flip (Aug 24, 2026 09:22 UTC) sits four windows back. Engine 1.1 has powered the sandbox since Aug 18, i.e. before this window; no cross-engine PnL comparisons are claimed.",
 "testing_next": [
  "Next issue: the pooled agree-vs-disagree lift went back below zero at -14.0pp after one issue above it -- does the negative sign hold for a second issue?",
  "The 4-peer tier fell below the field base after two issues above it -- does it stay below for a second issue?",
  "Monthly series: the agreement curve across regimes at monthly n."
 ],
 "related_research": [
  {
   "title": "Weekly Calibration #8",
   "url": "https://marketmania.ai/research/reports/weekly-calibration-2026-09-21.pdf"
  },
  {
   "title": "Weekly Model Watch #8",
   "url": "https://marketmania.ai/research/reports/model-watch-2026-09-21.pdf"
  },
  {
   "title": "Consensus Watch #7",
   "url": "https://marketmania.ai/research/reports/consensus-watch-2026-09-14.pdf"
  }
 ],
 "related_research_note": "The three weekly reports publish together as one issue each week; each links straight to the others' PDF and to its own previous issue. Direct links are the posting rule from wave 2 on.",
 "research_to_date": {
  "this_report": {
   "loo_scored_observations": 4245
  },
  "platform": {
   "as_of_cutoff": "2026-09-28",
   "resolved_forecasts": 57929,
   "resolved_forecasts_definition": "directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, at the issue cutoff",
   "resolved_forecasts_since": "2026-07-11",
   "models_tracked": 10,
   "models_tracked_detail": "7 models tracked (current line-up; earlier versions folded into their successors' lineage)",
   "assets": 5,
   "forecast_horizons": 5,
   "cadence": "hourly",
   "published_research_reports": 35
  },
  "source": "weekly_metrics_2026-09-21.json",
  "control_reproduces_wave3_pin": true,
  "pin_note": "Research-to-date counter: directional forecasts resolved inside the FH gate (1h/4h/1d) since Jul 11, counted at the issue cutoff (Sep 28 16:00 UTC) -- the definition pinned in issue #3 and carried by issues #4, #5, #6 and #7, which printed 52,971 at the Sep 21 16:00 cutoff. The pack reproduces that pin at the previous cutoff (control OK), so this issue's 57,929 is an additive step under one definition; counter deltas against issue #6 and earlier remain definitional."
 },
 "site_entry": {
  "slug": "consensus-watch-2026-09-21",
  "series": "weekly",
  "seriesLabel": "Weekly · Consensus",
  "date": "September 30, 2026",
  "title": "Consensus Watch #8",
  "summary": "9,273 mature forecasts asked whether agreeing with the crowd makes an LLM market call safer. Herding ran at 99.3% of leave-one-out scoreable calls (issue #7: 98.9%), and the pooled lift came in at -14.0pp: agree-hit 46.0% against 60.0% on disagreement, on a disagree bucket of n=30 and a 46.8% field base. The margin curve read 39.1% / 39.0% / 46.8% / 47.8% from 3 peers to full unanimity. Field accuracy moved 50.7% -> 46.8% (-3.9pp) week over week, with 0 of 7 models improving.",
  "stats": [
   "4,245 peer-judged calls",
   "99.3% ran with the herd",
   "lift -14.0pp vs disagree"
  ],
  "manifest_key_finding": "Herding ran at 99.3% of leave-one-out scoreable calls and the pooled lift came in at -14.0pp: agree-hit 46.0% against 60.0% on disagreement (disagree bucket n=30), against a 46.8% field base.",
  "files": {
   "pdf": "/research/reports/consensus-watch-2026-09-21.pdf",
   "md": "/research/reports/consensus-watch-2026-09-21.md",
   "json": "/research/reports/consensus-watch-2026-09-21.json"
  }
 },
 "social": {
  "telegram": {
   "photo_caption_html": "🧭 <b>Consensus Watch #8</b> — weekly series (window Sep 21–27)\n\nResearch question: if a model's forecast matches the consensus of the other models, is it more reliable?\n\n4,245 leave-one-out scored calls, 7 model lines:\n• Herding ran at <b>99.3%</b> (issue #7: 98.9%) — agree-hit 46.0% vs <b>60.0%</b> when a model left the crowd (field base 46.8%, disagree n=30)\n• Margin curve: 3-peer 39.1%, 4-peer 39.0%, 5-peer 46.8%, unanimity 47.8%\n• Daily horizon: 1d/1d unanimity 51.7%, 1d/4h unanimity 39.6%\n• Market check: field accuracy 50.7% -> 46.8% (-3.9pp), 0 of 7 models improved; avg |1d| move 2.69% -> 1.83%\n\nNote: grok is a pure 4.6 line in both weeks of this issue; the 4.5 -> 4.6 flip (Aug 24 09:22 UTC) sits four windows back\n\n📄 <a href=\"https://marketmania.ai/research/reports/consensus-watch-2026-09-21.pdf\">Full report (PDF)</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #consensus #research",
   "document_caption_html": "📄 Consensus Watch #8 — full report (PDF).\nWeb copy: <a href=\"https://marketmania.ai/research/reports/consensus-watch-2026-09-21.pdf\">consensus-watch-2026-09-21.pdf</a> · machine-readable: <a href=\"https://marketmania.ai/research/reports/consensus-watch-2026-09-21.json\">JSON</a> · <a href=\"https://marketmania.ai/research\">all research</a>\n\n#cryptotrading #LLM #AI #crypto #trading #consensus #research",
   "caption_len": 959,
   "document_caption_len": 402,
   "limit": 1024
  },
  "x": {
   "main": "Consensus Watch #8 is out.\n\n4,245 peer-judged LLM market calls: 99.3% ran with the herd, and agreeing scored 46.0% against 60.0% for the 30 calls that left it (-14.0pp).\n\nFull report: https://marketmania.ai/research/reports/consensus-watch-2026-09-21.pdf\n\n#AI #Crypto #Research #Consensus",
   "main_tco_len": 241,
   "reply": "Margin curve, w/w (hit):\n3 peers: 52.1 -> 39.1\n4 peers: 55.5 -> 39.0\n5 peers: 44.2 -> 46.8\nunanimity: 51.6 -> 47.8\nField base: 50.7 -> 46.8. LOO 3+ peers, methodology v1.1.\nData: https://marketmania.ai/research/reports/consensus-watch-2026-09-21.json",
   "reply_tco_len": 202,
   "limit_tco": 280,
   "replies_max": 1
  }
 }
}
