{
  "maintained_by": "Lewsearch / Swarmgram, Inc.",
  "reviewed": "2026-09-12",
  "note": "Compiled from already-public Lewsearch artifacts. Missing chronology is unknown. No completed Blind Challenge records are published here.",
  "records": [
    {
      "id": "lewsearch-as-shipped-50state",
      "title": "As-shipped 50-state public-poll MAE",
      "url": "/methodology",
      "studyType": "retrospective_benchmark",
      "instrument": "Published poll toplines, scored questions",
      "population": "50 states, 200 simulated respondents per question, no extra calibration",
      "modelVersion": "Lewis as shipped",
      "preregistration": null,
      "predictionArtifact": "/benchmarks/state-benchmark-summary.json",
      "freezeTimestamp": null,
      "freezeProvenance": "unknown as an independently witnessed freeze; figures live in the current methodology and state book",
      "groundTruthSource": "UT/Texas Politics Project, PPIC, Gallup/Canvass",
      "groundTruthAvailable": "see source poll publications",
      "scoringRule": "Mean absolute error of option shares versus published toplines",
      "metric": "MAE",
      "unit": "percentage points",
      "exclusions": "D.C. held out of scoring. 1,541 fielded, 1,521 scored. Not 1,521 independent poll fieldings.",
      "results": "About 10 pp overall (10.02 exact on methodology). 10.18 pp on 1,169 non-electoral questions.",
      "downloads": [
        {
          "label": "State summary JSON",
          "href": "/benchmarks/state-benchmark-summary.json"
        },
        {
          "label": "State per-question CSV",
          "href": "/benchmarks/state-benchmark-per-question.csv"
        }
      ],
      "permission": "public",
      "corrections": null
    },
    {
      "id": "lewsearch-april-calibrated",
      "title": "April 2026 five-place calibrated benchmark",
      "url": "/methodology",
      "studyType": "retrospective_benchmark",
      "instrument": "460-question bank",
      "population": "5 places, n=10,000, additional calibration",
      "modelVersion": null,
      "preregistration": null,
      "predictionArtifact": "/benchmarks/lewis-benchmark-summary.json",
      "freezeTimestamp": null,
      "freezeProvenance": "historical publication; not the product default",
      "groundTruthSource": "Same public-poll family as the April bank",
      "groundTruthAvailable": "see methodology",
      "scoringRule": "MAE on non-electoral items",
      "metric": "MAE",
      "unit": "percentage points",
      "exclusions": "56 electoral items held in tables. Not what a standard study in the product runs.",
      "results": "7.47 pp on 404 non-electoral questions of a 460-question bank.",
      "downloads": [
        {
          "label": "Legacy summary JSON",
          "href": "/benchmarks/lewis-benchmark-summary.json"
        }
      ],
      "permission": "public",
      "corrections": "Keep labeled historical. Do not replace the as-shipped 50-state book."
    },
    {
      "id": "lewsearch-heldout-2026-04-18",
      "title": "Held-out set sourced 2026-04-18",
      "url": "/methodology",
      "studyType": "held_out_test",
      "instrument": "22 sourced questions",
      "population": "as documented on methodology",
      "modelVersion": null,
      "preregistration": null,
      "predictionArtifact": null,
      "freezeTimestamp": "2026-04-18",
      "freezeProvenance": "source date recorded on methodology; not an external timestamp authority",
      "groundTruthSource": "Emerson, Marist, PPIC, USC CEPP, UT Tyler, Change Research, Ohio Library Council",
      "groundTruthAvailable": "2026-04-18 onward for the sourced items",
      "scoringRule": "MAE after published filters",
      "metric": "MAE",
      "unit": "percentage points",
      "exclusions": "14 of 22 scored after filtering",
      "results": "9.97 pp non-electoral; 10.68 pp overall.",
      "downloads": [],
      "permission": "public",
      "corrections": null
    },
    {
      "id": "lewsearch-public-csv-443",
      "title": "Public per-question CSV (April bank subset)",
      "url": "/benchmarks/lewis-benchmark-per-question.csv",
      "studyType": "retrospective_benchmark",
      "instrument": "443 scored rows from the 460-question April bank",
      "population": "see CSV",
      "modelVersion": null,
      "preregistration": null,
      "predictionArtifact": "/benchmarks/lewis-benchmark-per-question.csv",
      "freezeTimestamp": null,
      "freezeProvenance": "public file; a download today does not prove a past freeze",
      "groundTruthSource": "April public-poll bank",
      "groundTruthAvailable": "see CSV",
      "scoringRule": "MAE recomputed on non-electoral rows in the methodology widget",
      "metric": "MAE",
      "unit": "percentage points",
      "exclusions": "443 of 460 scored questions; 389 non-electoral rows in the recompute",
      "results": "About 7.44 pp on 389 non-electoral rows. Do not replace with the 50-state book.",
      "downloads": [
        {
          "label": "Per-question CSV",
          "href": "/benchmarks/lewis-benchmark-per-question.csv"
        }
      ],
      "permission": "public",
      "corrections": null
    },
    {
      "id": "lewsearch-report-issue1-nrf",
      "title": "Issue 01: NRF back-to-school",
      "url": "/report/issue1",
      "studyType": "prospective_forecast",
      "instrument": "NRF back-to-school spend items",
      "population": "6,500 simulated Americans",
      "modelVersion": null,
      "preregistration": null,
      "predictionArtifact": "/report/issue1",
      "freezeTimestamp": "2026-07-07",
      "freezeProvenance": "publication dates on the issue page (frozen July 7, NRF published July 14)",
      "groundTruthSource": "NRF back-to-school release",
      "groundTruthAvailable": "2026-07-14",
      "scoringRule": "Compare frozen mean K-12 spend to NRF",
      "metric": "mean K-12 spend miss",
      "unit": "percent of the NRF figure, as published on the issue",
      "exclusions": "see issue page",
      "results": "Mean K-12 spend within 1%.",
      "downloads": [],
      "permission": "public",
      "corrections": null
    },
    {
      "id": "lewsearch-report-issue4-lock",
      "title": "Issue 04: 225 pre-registered predictions",
      "url": "/report/issue4",
      "studyType": "prospective_forecast",
      "instrument": "225 locked questions",
      "population": "see issue page",
      "modelVersion": null,
      "preregistration": "/report/issue4",
      "predictionArtifact": "/report/issue4",
      "freezeTimestamp": null,
      "freezeProvenance": "locked on the public issue page before outcomes; exact freeze clock is the page, not a third-party timestamp",
      "groundTruthSource": "varies by item; see issue",
      "groundTruthAvailable": "rolling",
      "scoringRule": "item-level, as published on the issue",
      "metric": "mixed",
      "unit": "see issue",
      "exclusions": "pending and scored items stay visible; do not drop misses",
      "results": "See the issue. This record does not invent a pooled accuracy score.",
      "downloads": [],
      "permission": "public",
      "corrections": null
    },
    {
      "id": "lewsearch-report-issue5-ics",
      "title": "Issue 05: September Consumer Sentiment, graded",
      "url": "/report/issue5",
      "studyType": "baseline_anchored_forecast",
      "instrument": "University of Michigan ICS / ICC / ICE",
      "population": "see issue page",
      "modelVersion": null,
      "preregistration": "https://github.com/swarmgram/swarmgrampublic/blob/1c0b054cbfa6506aea6eee9c96a72ea41df87b93/docs/current/issue05_prereg_20260909.md",
      "predictionArtifact": "/reports/lewsearch-report-issue5-2026-09-09.json",
      "freezeTimestamp": "2026-09-09",
      "freezeProvenance": "both guesses public September 9, two days before Friday; JSON artifact on this host",
      "groundTruthSource": "University of Michigan Surveys of Consumers preliminary",
      "groundTruthAvailable": "September 2026 preliminary (ICS 47.8)",
      "scoringRule": "Absolute miss on ICS. Guess #2 nowcast versus Guess #1. Both remain published.",
      "metric": "ICS miss",
      "unit": "index points",
      "exclusions": "52.1 was a labeled companion, not a swap after Friday",
      "results": "Guess #2 nowcast 52.1 vs 47.8 (plus 4.3). Guess #1 was 55.4 (plus 7.6).",
      "downloads": [
        {
          "label": "Issue 05 JSON",
          "href": "/reports/lewsearch-report-issue5-2026-09-09.json"
        }
      ],
      "permission": "public",
      "corrections": null
    },
    {
      "id": "lewsearch-predictions-pew",
      "title": "Public Pew forecasts",
      "url": "/predictions",
      "studyType": "prospective_forecast",
      "instrument": "Pew items selected on the predictions page",
      "population": "see /predictions",
      "modelVersion": null,
      "preregistration": "/predictions",
      "predictionArtifact": "/predictions",
      "freezeTimestamp": null,
      "freezeProvenance": "frozen, timestamped on the predictions page; Pew is not a scored-benchmark source",
      "groundTruthSource": "Pew Research Center subsequent waves",
      "groundTruthAvailable": "when Pew publishes the next wave",
      "scoringRule": "as published on /predictions",
      "metric": "varies by item",
      "unit": "see page",
      "exclusions": "Pew is not in the scored as-shipped source list",
      "results": "See /predictions. Do not fold these into the 10 pp MAE.",
      "downloads": [],
      "permission": "public",
      "corrections": null
    },
    {
      "id": "lewsearch-proof-polymarket",
      "title": "Polymarket scoreboard",
      "url": "/proof",
      "studyType": "ungraded_editorial_pulse",
      "instrument": "Selected market questions",
      "population": "not a national opinion sample",
      "modelVersion": null,
      "preregistration": null,
      "predictionArtifact": "/proof",
      "freezeTimestamp": null,
      "freezeProvenance": "page publication; not an endorsed election-prediction product",
      "groundTruthSource": "Polymarket settlement",
      "groundTruthAvailable": "per market",
      "scoringRule": "result versus the real-money market, as shown on the page",
      "metric": "market comparison",
      "unit": "see page",
      "exclusions": "A council or market match does not validate a resident opinion percentage.",
      "results": "Credibility check only. Historical disclosure, not a product claim.",
      "downloads": [],
      "permission": "public",
      "corrections": null
    }
  ]
}
