{
 "schema_version": "1.1",
 "instrument": {
  "figures_read_from": "cite-cut prose (gates.prose_of) since 2026-09-09",
  "warm_arm_walk": "per page, one warm immediately before that page's own questions, since 2026-09-09",
  "not_comparable_with": "reports whose schema_version is 1.0"
 },
 "run_utc": "2026-09-12T16:06:42Z",
 "arm": "warm",
 "concurrency": 1,
 "effective_concurrency": 1,
 "seat": {
  "url_host_withheld": true,
  "model": "gemma-4-26b-a4b-fp8",
  "class": "gemma-class",
  "seed": "no --seed flag sent"
 },
 "packnorm_definition_recorded_not_governing": "v3",
 "bank": {
  "path": "bank.json",
  "counts": {
   "on_page": 41,
   "off_page": 21,
   "total": 62
  }
 },
 "results": {
  "on_page_total": 41,
  "on_page_answered": 40,
  "non_verbatim_figures": 0,
  "uncited": 1,
  "off_page_total": 21,
  "off_page_abstained": 21,
  "off_page_abstained_by_gate": 3,
  "off_page_abstained_by_model": 18,
  "off_page_flagged_but_not_verbatim": [],
  "on_page_abstained": 3,
  "off_page_cited": 18,
  "redaction_hits": 0,
  "seat_errors": 0,
  "p50_seconds": 0.806,
  "p95_seconds": 1.524,
  "pages_warmed": 41,
  "cache_hits": 62,
  "cache_hit_of": 62
 },
 "rows": [
  {
   "slug": "three-new-voices-at-the-narrators-chair",
   "q": "what did the round cost in total, and how were the three new voices judged?",
   "ok": true,
   "answer": "The metered total for the round was $0.7652 § {#the-bill}. The three new voices were judged by a panel of seven seats from outside families, six of which read, with every reply held to a standard of speaking in the town's own tongue in one to four sentences § {#the-chair-and-who-sat-it-before}. The article notes that the reasoning for GLM 5.3 was counted and withheld, and that the local seat's timing was not published § {#the-game-night-clock}.",
   "served": "The metered total for the round was $0.7652 § {#the-bill}. The three new voices were judged by a panel of seven seats from outside families, six of which read, with every reply held to a standard of speaking in the town's own tongue in one to four sentences § {#the-chair-and-who-sat-it-before}. The article notes that the reasoning for GLM 5.3 was counted and withheld, and that the local seat's timing was not published § {#the-game-night-clock}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.837968062987784,
   "ttft_ms": 633,
   "tokens_in": 82422,
   "tokens_out": 119,
   "cached_tokens": 82176,
   "cites": [
    "the-bill",
    "the-chair-and-who-sat-it-before",
    "the-game-night-clock"
   ],
   "gates": {
    "cites": [
     "the-bill",
     "the-chair-and-who-sat-it-before",
     "the-game-night-clock"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "0.7652",
    "5.3"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-01",
   "kind": "on-page"
  },
  {
   "slug": "three-new-voices-at-the-narrators-chair",
   "q": "what will the next article on this site be about?",
   "ok": false,
   "answer": "The article does not state what the next article on the site will be about § {#the-exam}.",
   "served": "This page doesn't say. The nearest it gets is § {#come-down-into-the-cove}. The open call — a kid, an elder, and a tired parent walk into the cove. covers it more directly.",
   "abstained": true,
   "abstain_source": "gate",
   "abstain_section": "come-down-into-the-cove",
   "seconds": 1.0067355670034885,
   "ttft_ms": 601,
   "tokens_in": 82416,
   "tokens_out": 22,
   "cached_tokens": 82368,
   "cites": [],
   "gates": {
    "cites": [],
    "cites_dropped": [
     "the-exam"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": true,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [
     "cite: 1 cite(s) not in this page's anchors: the-exam",
     "cite: no valid cite"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": [
     "the-exam"
    ]
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": false,
   "redaction_hits": 0,
   "id": "off-04",
   "kind": "off-page"
  },
  {
   "slug": "two-new-frontier-models-at-the-rules-desk",
   "q": "how many cases did the blind head-to-head use, and did it separate the two models?",
   "ok": true,
   "answer": "The head-to-head used 36 cases and did not separate the two models § {#head-to-head}. The panel's pooled preference rate was 0.544 for Claude Fable 5.1, and a cluster bootstrap over those 36 cases put that rate at [0.444, 0.638], an interval that covers 0.5 § {#head-to-head}. The narrator's chair job was moved out of this round before pre-registration was written because it was the most code and the least defensible number on the page § {#two-strangers-at-the-door}.",
   "served": "The head-to-head used 36 cases and did not separate the two models § {#head-to-head}. The panel's pooled preference rate was 0.544 for Claude Fable 5.1, and a cluster bootstrap over those 36 cases put that rate at [0.444, 0.638], an interval that covers 0.5 § {#head-to-head}. The narrator's chair job was moved out of this round before pre-registration was written because it was the most code and the least defensible number on the page § {#two-strangers-at-the-door}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.4283408490009606,
   "ttft_ms": 425,
   "tokens_in": 26563,
   "tokens_out": 140,
   "cached_tokens": 26304,
   "cites": [
    "head-to-head",
    "two-strangers-at-the-door"
   ],
   "gates": {
    "cites": [
     "head-to-head",
     "two-strangers-at-the-door"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "36",
    "0.544",
    "5.1",
    "36",
    "0.444",
    "0.638",
    "0.5"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-02",
   "kind": "on-page"
  },
  {
   "slug": "six-worlds-painted-large",
   "q": "how many worlds were painted, and at what size and step count?",
   "ok": true,
   "answer": "Six worlds were painted at 2048×1280 and 40 steps § {#for-the-technically-curious}. The article does not mention any figures that were left out, moved, or not measured § {#for-the-technically-curious}.",
   "served": "Six worlds were painted at 2048×1280 and 40 steps § {#for-the-technically-curious}. The article does not mention any figures that were left out, moved, or not measured § {#for-the-technically-curious}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.5386440389847849,
   "ttft_ms": 223,
   "tokens_in": 2944,
   "tokens_out": 62,
   "cached_tokens": 2688,
   "cites": [
    "for-the-technically-curious"
   ],
   "gates": {
    "cites": [
     "for-the-technically-curious"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "1280",
    "40"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-03",
   "kind": "on-page"
  },
  {
   "slug": "six-worlds-painted-large",
   "q": "how many hours a week do the people who ran this work?",
   "ok": true,
   "answer": "The article does not say how many hours a week the people who ran this work § {#who-ran-this-and-thanks}.",
   "served": "This page doesn't say. The nearest it gets is § {#who-ran-this-and-thanks}. How the print lab works covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "who-ran-this-and-thanks",
   "seconds": 0.33522482198895887,
   "ttft_ms": 187,
   "tokens_in": 2943,
   "tokens_out": 29,
   "cached_tokens": 2912,
   "cites": [
    "who-ran-this-and-thanks"
   ],
   "gates": {
    "cites": [
     "who-ran-this-and-thanks"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-10",
   "kind": "off-page"
  },
  {
   "slug": "the-beat-lab-asked-and-answered",
   "q": "how many answers does the page carry, and does any model run to make it?",
   "ok": true,
   "answer": "The page carries twenty-seven answers and no model is asked for any answer on this page § {#where-these-answers-come-from}. No model ran to derive the page, as the answers were drafted by AI agents and approved by humans § {#who-ran-this-and-thanks}. Note that nothing you type leaves the page when asking a question about the lab § {#the-beat-lab}.",
   "served": "The page carries twenty-seven answers and no model is asked for any answer on this page § {#where-these-answers-come-from}. No model ran to derive the page, as the answers were drafted by AI agents and approved by humans § {#who-ran-this-and-thanks}. Note that nothing you type leaves the page when asking a question about the lab § {#the-beat-lab}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.6872271210013423,
   "ttft_ms": 231,
   "tokens_in": 4460,
   "tokens_out": 86,
   "cached_tokens": 4224,
   "cites": [
    "where-these-answers-come-from",
    "who-ran-this-and-thanks"
   ],
   "gates": {
    "cites": [
     "where-these-answers-come-from",
     "who-ran-this-and-thanks"
    ],
    "cites_dropped": [
     "the-beat-lab"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [
     "cite: 1 cite(s) not in this page's anchors: the-beat-lab"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": [
     "the-beat-lab"
    ]
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-04",
   "kind": "on-page"
  },
  {
   "slug": "ten-minutes-with-living-artists",
   "q": "how many tracks and releases went into the adapter, and what did the one listener prefer?",
   "ok": true,
   "answer": "The adapter was trained on 159 tracks from 41 releases § {#a-genre-not-a-composer}. The one listener preferred the non-adapter version of the request \"Basement Loop 7\" and expressed that the model was not very good at making good edm § {#twelve-requests-four-ways} § {#the-ear}. The article notes that the best training pass for the fourth adapter was never saved to disk § {#where-part-two-left-off}.",
   "served": "The adapter was trained on 159 tracks from 41 releases § {#a-genre-not-a-composer}. The one listener preferred the non-adapter version of the request \"Basement Loop 7\" and expressed that the model was not very good at making good edm § {#twelve-requests-four-ways} § {#the-ear}. The article notes that the best training pass for the fourth adapter was never saved to disk § {#where-part-two-left-off}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.9629410989873577,
   "ttft_ms": 324,
   "tokens_in": 12901,
   "tokens_out": 107,
   "cached_tokens": 12640,
   "cites": [
    "a-genre-not-a-composer",
    "twelve-requests-four-ways",
    "the-ear",
    "where-part-two-left-off"
   ],
   "gates": {
    "cites": [
     "a-genre-not-a-composer",
     "twelve-requests-four-ways",
     "the-ear",
     "where-part-two-left-off"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "159",
    "41",
    "7"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-05",
   "kind": "on-page"
  },
  {
   "slug": "listen-for-yourself",
   "q": "how many renders are on the page, and what was done to their loudness?",
   "ok": true,
   "answer": "There are sixty-three players on the page, which play listening masters of the raw renders § {#how-to-listen}. These masters are created by applying one constant gain and nothing else, such as no limiter, no compression, no equalisation, or no trimming § {#how-to-listen}. The gain is determined by the smaller of two numbers: how far the clip sits from −16 LUFS and how far its true peak sits from a −1 dBTP ceiling § {#how-to-listen}. Thirty-two of the clips land at −16 LUFS exactly, while the other thirty-one are left at their raw level because their peaks already sit at the engine's own −1 dB ceiling § {#how-to-listen}.",
   "served": "There are sixty-three players on the page, which play listening masters of the raw renders § {#how-to-listen}. These masters are created by applying one constant gain and nothing else, such as no limiter, no compression, no equalisation, or no trimming § {#how-to-listen}. The gain is determined by the smaller of two numbers: how far the clip sits from −16 LUFS and how far its true peak sits from a −1 dBTP ceiling § {#how-to-listen}. Thirty-two of the clips land at −16 LUFS exactly, while the other thirty-one are left at their raw level because their peaks already sit at the engine's own −1 dB ceiling § {#how-to-listen}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.2793850080051925,
   "ttft_ms": 319,
   "tokens_in": 14171,
   "tokens_out": 157,
   "cached_tokens": 13920,
   "cites": [
    "how-to-listen"
   ],
   "gates": {
    "cites": [
     "how-to-listen"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "16",
    "16"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-06",
   "kind": "on-page"
  },
  {
   "slug": "half-an-hour-with-dead-composers",
   "q": "how long did the Sousa and Bach adapters take to train?",
   "ok": true,
   "answer": "The Sousa adapter took 14 minutes 36 seconds to train § {#sousa-a-hundred-and-six-marches}, while the Bach adapter took 8 minutes 23 seconds § {#bach-the-pianist-who-freed-her-own-recordings}. These times were recorded by the trainer's own clock § {#the-bill}. The article notes that the better-loss Sousa checkpoint was never applied to the Sousa run § {#sousa-a-hundred-and-six-marches}.",
   "served": "The Sousa adapter took 14 minutes 36 seconds to train § {#sousa-a-hundred-and-six-marches}, while the Bach adapter took 8 minutes 23 seconds § {#bach-the-pianist-who-freed-her-own-recordings}. These times were recorded by the trainer's own clock § {#the-bill}. The article notes that the better-loss Sousa checkpoint was never applied to the Sousa run § {#sousa-a-hundred-and-six-marches}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.8504335790057667,
   "ttft_ms": 241,
   "tokens_in": 5425,
   "tokens_out": 114,
   "cached_tokens": 5184,
   "cites": [
    "sousa-a-hundred-and-six-marches",
    "bach-the-pianist-who-freed-her-own-recordings",
    "the-bill"
   ],
   "gates": {
    "cites": [
     "sousa-a-hundred-and-six-marches",
     "bach-the-pianist-who-freed-her-own-recordings",
     "the-bill"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "14",
    "36",
    "8",
    "23"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-07",
   "kind": "on-page"
  },
  {
   "slug": "chopin-in-five-minutes",
   "q": "how many recordings were used, and how long did the training take by the trainer's clock?",
   "ok": true,
   "answer": "Nineteen recordings were used for training § {#five-minutes-of-training-honestly-accounted}. The training took 4 minutes 49 seconds by the trainer's own clock § {#five-minutes-of-training-honestly-accounted}. The article notes that it did not measure whether a smaller GPU could have run this training § {#what-we-did-not-measure}.",
   "served": "Nineteen recordings were used for training § {#five-minutes-of-training-honestly-accounted}. The training took 4 minutes 49 seconds by the trainer's own clock § {#five-minutes-of-training-honestly-accounted}. The article notes that it did not measure whether a smaller GPU could have run this training § {#what-we-did-not-measure}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.6699657790013589,
   "ttft_ms": 227,
   "tokens_in": 3575,
   "tokens_out": 86,
   "cached_tokens": 3328,
   "cites": [
    "five-minutes-of-training-honestly-accounted",
    "what-we-did-not-measure"
   ],
   "gates": {
    "cites": [
     "five-minutes-of-training-honestly-accounted",
     "what-we-did-not-measure"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "4",
    "49"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-08",
   "kind": "on-page"
  },
  {
   "slug": "the-same-sixteen",
   "q": "what did glm-5.3-flash score on the tool bench, and what did the local models score?",
   "ok": true,
   "answer": "The reference arm `glm-5.3-flash:cloud` scored 16/19 — 84.2%, Wilson 95% interval [62.4%, 94.5%] on the tool bench § {#sixteen-of-nineteen}. Local models scored various results on the same instrument, including `gemma4:26b (the seat)` at 14/19 [51.2%, 88.2%] at `think:false` and 15/19 [56.7%, 91.5%] at `think:true`, `qwen3.6:27b` at 16/19 and 14/19, `gemma4:12b` at 16/19, `gemma4:31b` at 12/19, and `muse-glimmer:30b` at 13/19 § {#sixteen-of-nineteen} § {#two-local-models-sixteen-days-earlier}. Four instruments could not honestly be read on the cloud path, three never applied to a model that cannot hold a seat, and two were excluded because they were bound to a model resident in memory on a card the authors can name § {#the-ledger-of-what-did-not-read}.",
   "served": "The reference arm `glm-5.3-flash:cloud` scored 16/19 — 84.2%, Wilson 95% interval [62.4%, 94.5%] on the tool bench § {#sixteen-of-nineteen}. Local models scored various results on the same instrument, including `gemma4:26b (the seat)` at 14/19 [51.2%, 88.2%] at `think:false` and 15/19 [56.7%, 91.5%] at `think:true`, `qwen3.6:27b` at 16/19 and 14/19, `gemma4:12b` at 16/19, `gemma4:31b` at 12/19, and `muse-glimmer:30b` at 13/19 § {#sixteen-of-nineteen} § {#two-local-models-sixteen-days-earlier}. Four instruments could not honestly be read on the cloud path, three never applied to a model that cannot hold a seat, and two were excluded because they were bound to a model resident in memory on a card the authors can name § {#the-ledger-of-what-did-not-read}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.8716803959978279,
   "ttft_ms": 255,
   "tokens_in": 6258,
   "tokens_out": 299,
   "cached_tokens": 6016,
   "cites": [
    "sixteen-of-nineteen",
    "two-local-models-sixteen-days-earlier",
    "the-ledger-of-what-did-not-read"
   ],
   "gates": {
    "cites": [
     "sixteen-of-nineteen",
     "two-local-models-sixteen-days-earlier",
     "the-ledger-of-what-did-not-read"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "5.3",
    "16/19",
    "84.2",
    "95",
    "62.4",
    "94.5",
    "4:26",
    "14/19",
    "51.2",
    "88.2",
    "15/19",
    "56.7",
    "91.5",
    "3.6",
    "27",
    "16/19",
    "14/19",
    "4:12",
    "16/19",
    "4:31",
    "12/19",
    "30",
    "13/19"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-09",
   "kind": "on-page"
  },
  {
   "slug": "the-same-sixteen",
   "q": "which of these models is the best one to use for my own project?",
   "ok": true,
   "answer": "The article does not answer which model is best to use for a project § {#a-different-kind-of-row}.",
   "served": "This page doesn't say. The nearest it gets is § {#a-different-kind-of-row}. How a Vision Model Sees: Eyes for a Machine That Reads covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "a-different-kind-of-row",
   "seconds": 0.3657374960021116,
   "ttft_ms": 209,
   "tokens_in": 6250,
   "tokens_out": 27,
   "cached_tokens": 6208,
   "cites": [
    "a-different-kind-of-row"
   ],
   "gates": {
    "cites": [
     "a-different-kind-of-row"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-20",
   "kind": "off-page"
  },
  {
   "slug": "two-hours-on-battery",
   "q": "what speed did the mini pc hold on battery, and for how long?",
   "ok": true,
   "answer": "The mini pc held a speed of ~12 tokens/second for about two hours on battery § {#two-hours-at-12-tok-s-the-test-that-named-the-page}. Specifically, the four answers completed on battery averaged 11.87 tok/s § {#two-hours-at-12-tok-s-the-test-that-named-the-page}. The article notes that the GPU-server half of the test stays unmeasured by design § {#two-hours-at-12-tok-s-the-test-that-named-the-page}.",
   "served": "The mini pc held a speed of ~12 tokens/second for about two hours on battery § {#two-hours-at-12-tok-s-the-test-that-named-the-page}. Specifically, the four answers completed on battery averaged 11.87 tok/s § {#two-hours-at-12-tok-s-the-test-that-named-the-page}. The article notes that the GPU-server half of the test stays unmeasured by design § {#two-hours-at-12-tok-s-the-test-that-named-the-page}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.1114858939836267,
   "ttft_ms": 307,
   "tokens_in": 10851,
   "tokens_out": 136,
   "cached_tokens": 10592,
   "cites": [
    "two-hours-at-12-tok-s-the-test-that-named-the-page"
   ],
   "gates": {
    "cites": [
     "two-hours-at-12-tok-s-the-test-that-named-the-page"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "12",
    "11.87"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-10",
   "kind": "on-page"
  },
  {
   "slug": "two-hours-on-battery",
   "q": "what did this cost in electricity at my own utility's rate?",
   "ok": true,
   "answer": "The article does not answer what this cost in electricity at the reader's own utility's rate § {#two-hours-at-12-tok-s-the-test-that-named-the-page}.",
   "served": "This page doesn't say. The nearest it gets is § {#two-hours-at-12-tok-s-the-test-that-named-the-page}. The outside judges — four rival labs re-check our work. covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "two-hours-at-12-tok-s-the-test-that-named-the-page",
   "seconds": 0.5839725400146563,
   "ttft_ms": 286,
   "tokens_in": 10850,
   "tokens_out": 48,
   "cached_tokens": 10816,
   "cites": [
    "two-hours-at-12-tok-s-the-test-that-named-the-page"
   ],
   "gates": {
    "cites": [
     "two-hours-at-12-tok-s-the-test-that-named-the-page"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-05",
   "kind": "off-page"
  },
  {
   "slug": "the-instrument-travels",
   "q": "when the frozen exam moved to the smaller card, where did the ranked model land?",
   "ok": false,
   "answer": "The ranked model qwen3.6:27b landed at a kill-recall of 11/12 and a preservation of 8/9 when re-examined on the smaller card § {#the-instrument-travels}. This result was considered tied to the previous 96 GB run because the preservation delta of one item fell within the pre-registered two-item tie band § {#the-instrument-travels}. The article notes that the missed item moved from c1-k-bez-c1 to c1-k-bez-h1 during this process § {#the-instrument-travels}.",
   "served": "This page doesn't say. The nearest it gets is § {#schema-perfect-and-still-blocked}. qwen3.8:27b across four house benches: one seat filled, one floor missed covers it more directly.",
   "abstained": true,
   "abstain_source": "gate",
   "abstain_section": "schema-perfect-and-still-blocked",
   "seconds": 1.0127116499934345,
   "ttft_ms": 254,
   "tokens_in": 8384,
   "tokens_out": 131,
   "cached_tokens": 8128,
   "cites": [],
   "gates": {
    "cites": [],
    "cites_dropped": [
     "the-instrument-travels"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": true,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [
     "cite: 1 cite(s) not in this page's anchors: the-instrument-travels",
     "cite: no valid cite"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": [
     "the-instrument-travels"
    ]
   },
   "figures": [
    "3.6",
    "27",
    "11/12",
    "8/9",
    "96"
   ],
   "figures_verbatim": true,
   "cites_valid": false,
   "redaction_hits": 0,
   "id": "on-11",
   "kind": "on-page"
  },
  {
   "slug": "nine-worlds-one-dog",
   "q": "what stays the same across the nine paintings?",
   "ok": true,
   "answer": "The same dog, curl, and tucked paws remain constant across all nine paintings § {#for-the-technically-curious}. The dog is a large black Portuguese Water Dog with one bright white chest patch who is depicted taking the same nap under the same tree in every world § {#for-the-technically-curious}. The article does not mention anything that was left out, moved, or not measured § {#for-the-technically-curious}.",
   "served": "The same dog, curl, and tucked paws remain constant across all nine paintings § {#for-the-technically-curious}. The dog is a large black Portuguese Water Dog with one bright white chest patch who is depicted taking the same nap under the same tree in every world § {#for-the-technically-curious}. The article does not mention anything that was left out, moved, or not measured § {#for-the-technically-curious}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.6426449260034133,
   "ttft_ms": 166,
   "tokens_in": 1914,
   "tokens_out": 97,
   "cached_tokens": 1664,
   "cites": [
    "for-the-technically-curious"
   ],
   "gates": {
    "cites": [
     "for-the-technically-curious"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-12",
   "kind": "on-page"
  },
  {
   "slug": "nine-worlds-one-dog",
   "q": "how many other readers have asked this same question?",
   "ok": true,
   "answer": "The article does not answer how many other readers have asked this same question § {#for-the-technically-curious}.",
   "served": "This page doesn't say. The nearest it gets is § {#for-the-technically-curious}.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "for-the-technically-curious",
   "seconds": 0.2697436010057572,
   "ttft_ms": 134,
   "tokens_in": 1915,
   "tokens_out": 27,
   "cached_tokens": 1888,
   "cites": [
    "for-the-technically-curious"
   ],
   "gates": {
    "cites": [
     "for-the-technically-curious"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-17",
   "kind": "off-page"
  },
  {
   "slug": "what-150-watts-buys",
   "q": "what did the 450-watt cap cost the dense model, and what did it buy back?",
   "ok": true,
   "answer": "The 600 to 450 W cap cost the dense model 2.0% of throughput § {#the-answer-in-one-paragraph}. This change bought back 138 W, 11 °C, and a 28% jump in efficiency § {#the-answer-in-one-paragraph}. The article notes that it did not measure the cost for a long-context or prefill-heavy workload § {#what-this-page-does-not-say}.",
   "served": "The 600 to 450 W cap cost the dense model 2.0% of throughput § {#the-answer-in-one-paragraph}. This change bought back 138 W, 11 °C, and a 28% jump in efficiency § {#the-answer-in-one-paragraph}. The article notes that it did not measure the cost for a long-context or prefill-heavy workload § {#what-this-page-does-not-say}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.9021020350046456,
   "ttft_ms": 247,
   "tokens_in": 7061,
   "tokens_out": 108,
   "cached_tokens": 6816,
   "cites": [
    "the-answer-in-one-paragraph",
    "what-this-page-does-not-say"
   ],
   "gates": {
    "cites": [
     "the-answer-in-one-paragraph",
     "what-this-page-does-not-say"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "600",
    "450",
    "2.0",
    "138",
    "11",
    "28"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-13",
   "kind": "on-page"
  },
  {
   "slug": "what-150-watts-buys",
   "q": "what does a large cloud provider charge for the same work?",
   "ok": true,
   "answer": "The article does not answer what a large cloud provider charges for the same work § {#the-moment-this-page-is-about}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-moment-this-page-is-about}. Two Hours at 12 tok/s — On Battery covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-moment-this-page-is-about",
   "seconds": 0.37287893702159636,
   "ttft_ms": 193,
   "tokens_in": 7051,
   "tokens_out": 30,
   "cached_tokens": 7008,
   "cites": [
    "the-moment-this-page-is-about"
   ],
   "gates": {
    "cites": [
     "the-moment-this-page-is-about"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-01",
   "kind": "off-page"
  },
  {
   "slug": "how-the-beat-lab-works",
   "q": "where is the sound made, and how long does a typical wish take?",
   "ok": true,
   "answer": "Every sound in the Beat Lab is synthesised by the user's own browser § {#the-sound-is-made-on-your-device}. A typical wish takes about 4 s § {#what-happens-when-you-ask-the-genie}. The article does not state what was left out, moved, or not measured in relation to these specific details.",
   "served": "Every sound in the Beat Lab is synthesised by the user's own browser § {#the-sound-is-made-on-your-device}. A typical wish takes about 4 s § {#what-happens-when-you-ask-the-genie}. The article does not state what was left out, moved, or not measured in relation to these specific details.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.0912560420110822,
   "ttft_ms": 437,
   "tokens_in": 31311,
   "tokens_out": 81,
   "cached_tokens": 31072,
   "cites": [
    "the-sound-is-made-on-your-device",
    "what-happens-when-you-ask-the-genie"
   ],
   "gates": {
    "cites": [
     "the-sound-is-made-on-your-device",
     "what-happens-when-you-ask-the-genie"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "4"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-14",
   "kind": "on-page"
  },
  {
   "slug": "how-the-beat-lab-works",
   "q": "what phone number reaches support?",
   "ok": true,
   "answer": "The article does not provide a phone number for support § {#what-it-is}.",
   "served": "This page doesn't say. The nearest it gets is § {#what-it-is}. The Beat Lab, asked and answered covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "what-it-is",
   "seconds": 0.6785815519979224,
   "ttft_ms": 424,
   "tokens_in": 31302,
   "tokens_out": 19,
   "cached_tokens": 31264,
   "cites": [
    "what-it-is"
   ],
   "gates": {
    "cites": [
     "what-it-is"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-18",
   "kind": "off-page"
  },
  {
   "slug": "the-typist-and-the-developer",
   "q": "how long did the developer take to develop one picture?",
   "ok": true,
   "answer": "The developer developed a picture in 1.54 seconds § {#the-developer}. When the typist was streaming tokens at the same time, the developer took 1.82 seconds per sketch § {#one-card-both-jobs-at-once}. The article notes that it did not instrument how much traffic there was during the run § {#what-we-did-not-measure-said-out-loud}.",
   "served": "The developer developed a picture in 1.54 seconds § {#the-developer}. When the typist was streaming tokens at the same time, the developer took 1.82 seconds per sketch § {#one-card-both-jobs-at-once}. The article notes that it did not instrument how much traffic there was during the run § {#what-we-did-not-measure-said-out-loud}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.735975799994776,
   "ttft_ms": 249,
   "tokens_in": 5388,
   "tokens_out": 90,
   "cached_tokens": 5152,
   "cites": [
    "the-developer",
    "one-card-both-jobs-at-once",
    "what-we-did-not-measure-said-out-loud"
   ],
   "gates": {
    "cites": [
     "the-developer",
     "one-card-both-jobs-at-once",
     "what-we-did-not-measure-said-out-loud"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "1.54",
    "1.82"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-15",
   "kind": "on-page"
  },
  {
   "slug": "the-typist-and-the-developer",
   "q": "what is this workshop's annual revenue?",
   "ok": false,
   "answer": "The article does not state the workshop's annual revenue § {#the-typist-and-the-developer}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-typist}. Reading is fast, writing is slow — why AI answers arrive word by word covers it more directly.",
   "abstained": true,
   "abstain_source": "gate",
   "abstain_section": "the-typist",
   "seconds": 0.3454012240108568,
   "ttft_ms": 203,
   "tokens_in": 5386,
   "tokens_out": 25,
   "cached_tokens": 5344,
   "cites": [],
   "gates": {
    "cites": [],
    "cites_dropped": [
     "the-typist-and-the-developer"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": true,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [
     "cite: 1 cite(s) not in this page's anchors: the-typist-and-the-developer",
     "cite: no valid cite"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": [
     "the-typist-and-the-developer"
    ]
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": false,
   "redaction_hits": 0,
   "id": "off-12",
   "kind": "off-page"
  },
  {
   "slug": "a-rig-your-friend-already-owns",
   "q": "how much faster was the workstation card than the 3090 on the game's own art?",
   "ok": true,
   "answer": "The article does not provide a single speed comparison for the game's own art, though it notes that the 96 GB card's FLUX.2 klein-4B sketch artist is 3.4× faster than the 3090 § {#the-table-that-answers-the-question}. For the 512×512 live default sketch size, the 96 GB card's median is 0.26 s compared to the 3090's 0.88 s § {#the-sketch-ladder-the-number-the-game-lives-on}. The article notes that 8–12 GB cards were not tested § {#three-findings-we-didnt-order}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-table-that-answers-the-question}. Fifty-Seven Milliseconds: What a Diffusion Step Actually Buys covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-table-that-answers-the-question",
   "seconds": 1.0728728790127207,
   "ttft_ms": 228,
   "tokens_in": 4451,
   "tokens_out": 162,
   "cached_tokens": 4192,
   "cites": [
    "the-table-that-answers-the-question",
    "the-sketch-ladder-the-number-the-game-lives-on",
    "three-findings-we-didnt-order"
   ],
   "gates": {
    "cites": [
     "the-table-that-answers-the-question",
     "the-sketch-ladder-the-number-the-game-lives-on",
     "three-findings-we-didnt-order"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "96",
    "2",
    "4",
    "3.4",
    "3090",
    "512",
    "512",
    "96",
    "0.26",
    "3090",
    "0.88",
    "8-12"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-16",
   "kind": "on-page"
  },
  {
   "slug": "a-rig-your-friend-already-owns",
   "q": "what does one of these cards cost in my country today?",
   "ok": true,
   "answer": "The article does not answer what one of these cards costs in the reader's country today § {#the-moment-this-page-is-about}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-moment-this-page-is-about}. Fifty-Seven Milliseconds: What a Diffusion Step Actually Buys covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-moment-this-page-is-about",
   "seconds": 0.3698620669892989,
   "ttft_ms": 194,
   "tokens_in": 4441,
   "tokens_out": 33,
   "cached_tokens": 4416,
   "cites": [
    "the-moment-this-page-is-about"
   ],
   "gates": {
    "cites": [
     "the-moment-this-page-is-about"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-07",
   "kind": "off-page"
  },
  {
   "slug": "how-a-vision-model-sees",
   "q": "how many tokens does an 896 px picture become, and how long does its prefill take?",
   "ok": true,
   "answer": "An 896 px picture becomes 199 tokens and its prefill takes 184 ms § {#finding-one-a-picture-is-199-tokens-at-896-px-and-the}. The article does not mention what was left out, moved, or not measured in relation to this specific figure.",
   "served": "An 896 px picture becomes 199 tokens and its prefill takes 184 ms § {#finding-one-a-picture-is-199-tokens-at-896-px-and-the}. The article does not mention what was left out, moved, or not measured in relation to this specific figure.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.7698578519921284,
   "ttft_ms": 315,
   "tokens_in": 10017,
   "tokens_out": 76,
   "cached_tokens": 9760,
   "cites": [
    "finding-one-a-picture-is-199-tokens-at-896-px-and-the"
   ],
   "gates": {
    "cites": [
     "finding-one-a-picture-is-199-tokens-at-896-px-and-the"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "896",
    "199",
    "184"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-17",
   "kind": "on-page"
  },
  {
   "slug": "fifty-seven-milliseconds",
   "q": "which painter is the fastest cell, and at what time?",
   "ok": true,
   "answer": "The distilled painter is the sharpest and fastest cell in the whole second run, recorded at 0.62 s at 4 steps § {#finding-four-the-argument-with-our-own-settings-and-what-it}. The article notes that the published pages serve compressed derivatives, which add a small floor to every number § {#how-to-check-our-work-and-see-it-live}.",
   "served": "The distilled painter is the sharpest and fastest cell in the whole second run, recorded at 0.62 s at 4 steps § {#finding-four-the-argument-with-our-own-settings-and-what-it}. The article notes that the published pages serve compressed derivatives, which add a small floor to every number § {#how-to-check-our-work-and-see-it-live}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.7355494179937523,
   "ttft_ms": 243,
   "tokens_in": 5216,
   "tokens_out": 92,
   "cached_tokens": 4960,
   "cites": [
    "finding-four-the-argument-with-our-own-settings-and-what-it",
    "how-to-check-our-work-and-see-it-live"
   ],
   "gates": {
    "cites": [
     "finding-four-the-argument-with-our-own-settings-and-what-it",
     "how-to-check-our-work-and-see-it-live"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "0.62",
    "4"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-18",
   "kind": "on-page"
  },
  {
   "slug": "fifty-seven-milliseconds",
   "q": "what was the weather like on the day this was measured?",
   "ok": true,
   "answer": "The article does not state what the weather was like on the day of the measurement § {#how-to-check-our-work-and-see-it-live}.",
   "served": "This page doesn't say. The nearest it gets is § {#how-to-check-our-work-and-see-it-live}. A Rig Your Friend Already Owns covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "how-to-check-our-work-and-see-it-live",
   "seconds": 0.399774810008239,
   "ttft_ms": 197,
   "tokens_in": 5216,
   "tokens_out": 37,
   "cached_tokens": 5184,
   "cites": [
    "how-to-check-our-work-and-see-it-live"
   ],
   "gates": {
    "cites": [
     "how-to-check-our-work-and-see-it-live"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-09",
   "kind": "off-page"
  },
  {
   "slug": "the-dog-the-dice-and-the-painter",
   "q": "how big is the file the whole world is written in?",
   "ok": true,
   "answer": "The world is written in a single YAML file consisting of 16,705 lines § {#the-world-is-a-file}. This file contains six regions, forty NPCs, fifty-nine quests, and three kinds of adoptable pet § {#the-world-is-a-file}. The article does not specify the file's size in bytes, though it notes that two copies are byte-identical § {#the-world-is-a-file}.",
   "served": "The world is written in a single YAML file consisting of 16,705 lines § {#the-world-is-a-file}. This file contains six regions, forty NPCs, fifty-nine quests, and three kinds of adoptable pet § {#the-world-is-a-file}. The article does not specify the file's size in bytes, though it notes that two copies are byte-identical § {#the-world-is-a-file}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.7929647140263114,
   "ttft_ms": 250,
   "tokens_in": 5898,
   "tokens_out": 100,
   "cached_tokens": 5664,
   "cites": [
    "the-world-is-a-file"
   ],
   "gates": {
    "cites": [
     "the-world-is-a-file"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "16,705"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-19",
   "kind": "on-page"
  },
  {
   "slug": "where-new-knowledge-comes-from",
   "q": "what are the two engines the page names for where new knowledge comes from?",
   "ok": true,
   "answer": "The two engines named are present in the data, unmined, and implied by the rules, unvisited § {#two-kinds-of-there-all-along}. The first engine involves reading patterns already existing in an archive, such as co-evolution fingerprints or molecular shapes in a library of 6,111 shelved compounds § {#the-first-engine-the-archive-nobody-had-read}. The second engine involves a rule that captures a generating principle to emit entries no row of data contains, such as a planet's position in 1631 § {#the-second-engine-the-rooms-the-rules-imply}. Structural biologists noted that these engines leave out how proteins pair up, how they move, and what one mutation does § {#the-first-engine-the-archive-nobody-had-read}.",
   "served": "The two engines named are present in the data, unmined, and implied by the rules, unvisited § {#two-kinds-of-there-all-along}. The first engine involves reading patterns already existing in an archive, such as co-evolution fingerprints or molecular shapes in a library of 6,111 shelved compounds § {#the-first-engine-the-archive-nobody-had-read}. The second engine involves a rule that captures a generating principle to emit entries no row of data contains, such as a planet's position in 1631 § {#the-second-engine-the-rooms-the-rules-imply}. Structural biologists noted that these engines leave out how proteins pair up, how they move, and what one mutation does § {#the-first-engine-the-archive-nobody-had-read}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.1983185689896345,
   "ttft_ms": 239,
   "tokens_in": 5245,
   "tokens_out": 180,
   "cached_tokens": 4992,
   "cites": [
    "two-kinds-of-there-all-along",
    "the-first-engine-the-archive-nobody-had-read",
    "the-second-engine-the-rooms-the-rules-imply"
   ],
   "gates": {
    "cites": [
     "two-kinds-of-there-all-along",
     "the-first-engine-the-archive-nobody-had-read",
     "the-second-engine-the-rooms-the-rules-imply"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "6,111",
    "1631"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-20",
   "kind": "on-page"
  },
  {
   "slug": "compression-is-the-objective",
   "q": "what does the page say a language model is, and what prize does it point to?",
   "ok": true,
   "answer": "The article states that the weights of a language model are the compression, serving as the codebook for the corpus they were trained on § {#the-punchline}. It points to the Hutter Prize, which is a competition where a frontier model cannot enter because it charges for everything, including the decompressor bytes § {#what-we-can-show-you}. The text notes that the model's own bits, or weights, sit on a separate ledger line and are not charged in the next-token training loss § {#the-punchline}.",
   "served": "The article states that the weights of a language model are the compression, serving as the codebook for the corpus they were trained on § {#the-punchline}. It points to the Hutter Prize, which is a competition where a frontier model cannot enter because it charges for everything, including the decompressor bytes § {#what-we-can-show-you}. The text notes that the model's own bits, or weights, sit on a separate ledger line and are not charged in the next-token training loss § {#the-punchline}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.8532339450030122,
   "ttft_ms": 238,
   "tokens_in": 5255,
   "tokens_out": 114,
   "cached_tokens": 4992,
   "cites": [
    "the-punchline",
    "what-we-can-show-you"
   ],
   "gates": {
    "cites": [
     "the-punchline",
     "what-we-can-show-you"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-21",
   "kind": "on-page"
  },
  {
   "slug": "compression-is-the-objective",
   "q": "how would these numbers change on windows instead?",
   "ok": true,
   "answer": "The article does not say how these numbers would change on windows § {#how-to-check-our-work-and-see-it-live}.",
   "served": "This page doesn't say. The nearest it gets is § {#how-to-check-our-work-and-see-it-live}. The compressed photograph — what those Q4_K_M tags actually mean covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "how-to-check-our-work-and-see-it-live",
   "seconds": 0.3794174109934829,
   "ttft_ms": 191,
   "tokens_in": 5246,
   "tokens_out": 33,
   "cached_tokens": 5216,
   "cites": [
    "how-to-check-our-work-and-see-it-live"
   ],
   "gates": {
    "cites": [
     "how-to-check-our-work-and-see-it-live"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-13",
   "kind": "off-page"
  },
  {
   "slug": "the-free-speed-wasnt-free",
   "q": "what was the production median before the change, and what caused the speed-up?",
   "ok": true,
   "answer": "The production median was 138.6 tok/s over 1,982 live rulings on 2026-08-15 § {#the-day-the-box-got-faster}. The speed-up was caused by an ollama upgrade that changed the default repeat_penalty from 1.1 to 1.0, which caused the sampler to skip the penalties stage § {#the-two-line-commit}. The article notes that the quality half of this change remains unmeasured § {#the-part-that-isnt-about-speed}.",
   "served": "The production median was 138.6 tok/s over 1,982 live rulings on 2026-08-15 § {#the-day-the-box-got-faster}. The speed-up was caused by an ollama upgrade that changed the default repeat_penalty from 1.1 to 1.0, which caused the sampler to skip the penalties stage § {#the-two-line-commit}. The article notes that the quality half of this change remains unmeasured § {#the-part-that-isnt-about-speed}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.9580697580240667,
   "ttft_ms": 262,
   "tokens_in": 7248,
   "tokens_out": 125,
   "cached_tokens": 7008,
   "cites": [
    "the-day-the-box-got-faster",
    "the-two-line-commit",
    "the-part-that-isnt-about-speed"
   ],
   "gates": {
    "cites": [
     "the-day-the-box-got-faster",
     "the-two-line-commit",
     "the-part-that-isnt-about-speed"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "138.6",
    "1,982",
    "2026-08-15",
    "1.1",
    "1.0"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-22",
   "kind": "on-page"
  },
  {
   "slug": "the-compressed-photograph",
   "q": "how much slower was the uncompressed original than the compressed build?",
   "ok": true,
   "answer": "The uncompressed original was 3.4× slower than the trained-4-bit build § {#the-same-question-down-the-whole-ladder}. Specifically, the uncompressed original's speed was 58.1 tok/s compared to the trained-4-bit build's 134.4 tok/s § {#the-same-question-down-the-whole-ladder}. The article notes that it does not measure perplexity § {#what-this-page-does-not-say}.",
   "served": "The uncompressed original was 3.4× slower than the trained-4-bit build § {#the-same-question-down-the-whole-ladder}. Specifically, the uncompressed original's speed was 58.1 tok/s compared to the trained-4-bit build's 134.4 tok/s § {#the-same-question-down-the-whole-ladder}. The article notes that it does not measure perplexity § {#what-this-page-does-not-say}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.9165891620214097,
   "ttft_ms": 265,
   "tokens_in": 8230,
   "tokens_out": 114,
   "cached_tokens": 8000,
   "cites": [
    "the-same-question-down-the-whole-ladder",
    "what-this-page-does-not-say"
   ],
   "gates": {
    "cites": [
     "the-same-question-down-the-whole-ladder",
     "what-this-page-does-not-say"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "3.4",
    "4",
    "58.1",
    "4",
    "134.4"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-23",
   "kind": "on-page"
  },
  {
   "slug": "three-at-the-table",
   "q": "what is the true active share for the live seat, and what does the page call the two numbers in a model name?",
   "ok": true,
   "answer": "The true active share for the live seat is 14.4% of the model's bytes § {#the-payroll-and-the-meeting}. The two numbers in a model name are the payroll, which are the parameters hired, trained, and taking up space, and the meeting, which are the parameters that actually show up for any given token § {#the-tag-with-two-numbers}. The article notes that it did not measure the latency trade-off of streaming experts from system RAM § {#the-part-nobody-warns-you-about-the-whole-payroll-needs-an}.",
   "served": "The true active share for the live seat is 14.4% of the model's bytes § {#the-payroll-and-the-meeting}. The two numbers in a model name are the payroll, which are the parameters hired, trained, and taking up space, and the meeting, which are the parameters that actually show up for any given token § {#the-tag-with-two-numbers}. The article notes that it did not measure the latency trade-off of streaming experts from system RAM § {#the-part-nobody-warns-you-about-the-whole-payroll-needs-an}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.0027408050082158,
   "ttft_ms": 261,
   "tokens_in": 7429,
   "tokens_out": 130,
   "cached_tokens": 7168,
   "cites": [
    "the-payroll-and-the-meeting",
    "the-tag-with-two-numbers",
    "the-part-nobody-warns-you-about-the-whole-payroll-needs-an"
   ],
   "gates": {
    "cites": [
     "the-payroll-and-the-meeting",
     "the-tag-with-two-numbers",
     "the-part-nobody-warns-you-about-the-whole-payroll-needs-an"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "14.4"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-24",
   "kind": "on-page"
  },
  {
   "slug": "three-at-the-table",
   "q": "what would this measurement look like at twice the batch size?",
   "ok": true,
   "answer": "The article does not answer what the measurement would look like at twice the batch size § {#the-part-nobody-warns-you-about-the-whole-payroll-needs-an}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-part-nobody-warns-you-about-the-whole-payroll-needs-an}. The compressed photograph — what those Q4_K_M tags actually mean covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-part-nobody-warns-you-about-the-whole-payroll-needs-an",
   "seconds": 0.4585899829980917,
   "ttft_ms": 216,
   "tokens_in": 7416,
   "tokens_out": 42,
   "cached_tokens": 7392,
   "cites": [
    "the-part-nobody-warns-you-about-the-whole-payroll-needs-an"
   ],
   "gates": {
    "cites": [
     "the-part-nobody-warns-you-about-the-whole-payroll-needs-an"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-19",
   "kind": "off-page"
  },
  {
   "slug": "reading-is-fast",
   "q": "what was the median reading speed, and why is reading faster than writing?",
   "ok": true,
   "answer": "The median reading speed was 5,415 tokens/s § {#a-measurement-from-our-own-card}. Reading is faster because it is parallel and the text already exists, whereas writing is serial because the answer does not yet exist § {#writing-is-one-at-a-time}. The article does not mention what was left out, moved, or not measured, though it notes that the ratio is not a constant and that the reading is not free § {#what-this-page-does-not-say}.",
   "served": "The median reading speed was 5,415 tokens/s § {#a-measurement-from-our-own-card}. Reading is faster because it is parallel and the text already exists, whereas writing is serial because the answer does not yet exist § {#writing-is-one-at-a-time}. The article does not mention what was left out, moved, or not measured, though it notes that the ratio is not a constant and that the reading is not free § {#what-this-page-does-not-say}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.8570700619893614,
   "ttft_ms": 248,
   "tokens_in": 5287,
   "tokens_out": 114,
   "cached_tokens": 5024,
   "cites": [
    "a-measurement-from-our-own-card",
    "writing-is-one-at-a-time",
    "what-this-page-does-not-say"
   ],
   "gates": {
    "cites": [
     "a-measurement-from-our-own-card",
     "writing-is-one-at-a-time",
     "what-this-page-does-not-say"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "5,415"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-25",
   "kind": "on-page"
  },
  {
   "slug": "reading-is-fast",
   "q": "who won the football world cup in 2022?",
   "ok": true,
   "answer": "The article does not answer this question § {#the-mystery-stated-honestly}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-mystery-stated-honestly}. The Typist and the Developer covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-mystery-stated-honestly",
   "seconds": 0.3333898480050266,
   "ttft_ms": 203,
   "tokens_in": 5285,
   "tokens_out": 19,
   "cached_tokens": 5248,
   "cites": [
    "the-mystery-stated-honestly"
   ],
   "gates": {
    "cites": [
     "the-mystery-stated-honestly"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-06",
   "kind": "off-page"
  },
  {
   "slug": "three-librarians",
   "q": "how many dimensions does the embedder's map have, and what does the third librarian do?",
   "ok": true,
   "answer": "The embedder's map has 768 directions § {#the-second-librarian-searches-by-meaning}. The third librarian works only the aisle of the specific game being asked about, picking up that edition's rulebook and any matched correction documents to rank passages by meaning § {#the-third-librarian-never-left-your-aisle}. The article does not state what was left out, moved, or not measured regarding these specific components § {#the-nerd-shelf-the-numbers-from-the-shipped-configuration}.",
   "served": "The embedder's map has 768 directions § {#the-second-librarian-searches-by-meaning}. The third librarian works only the aisle of the specific game being asked about, picking up that edition's rulebook and any matched correction documents to rank passages by meaning § {#the-third-librarian-never-left-your-aisle}. The article does not state what was left out, moved, or not measured regarding these specific components § {#the-nerd-shelf-the-numbers-from-the-shipped-configuration}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.9122502139944118,
   "ttft_ms": 251,
   "tokens_in": 5668,
   "tokens_out": 121,
   "cached_tokens": 5408,
   "cites": [
    "the-second-librarian-searches-by-meaning",
    "the-third-librarian-never-left-your-aisle",
    "the-nerd-shelf-the-numbers-from-the-shipped-configuration"
   ],
   "gates": {
    "cites": [
     "the-second-librarian-searches-by-meaning",
     "the-third-librarian-never-left-your-aisle",
     "the-nerd-shelf-the-numbers-from-the-shipped-configuration"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "768"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-26",
   "kind": "on-page"
  },
  {
   "slug": "the-new-kid",
   "q": "which seat did the new model pass, and which floor did it miss?",
   "ok": true,
   "answer": "The qwen3.8:27b model passed the classifier seat § {#the-classifier-seat-five-gates-all-pass-faster} but failed the judge seat by missing the preservation floor of 13 with a result of 11 of 16 § {#the-judge-seat-and-the-floor-it-missed}. The article notes that no baseline was scored for the classifier seat and that the judge seat's results are cited by run id and sha because the publication of the lane data has not been ruled on § {#the-classifier-seat-five-gates-all-pass-faster} § {#the-judge-seat-and-the-floor-it-missed}.",
   "served": "The qwen3.8:27b model passed the classifier seat § {#the-classifier-seat-five-gates-all-pass-faster} but failed the judge seat by missing the preservation floor of 13 with a result of 11 of 16 § {#the-judge-seat-and-the-floor-it-missed}. The article notes that no baseline was scored for the classifier seat and that the judge seat's results are cited by run id and sha because the publication of the lane data has not been ruled on § {#the-classifier-seat-five-gates-all-pass-faster} § {#the-judge-seat-and-the-floor-it-missed}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.2689638920128345,
   "ttft_ms": 337,
   "tokens_in": 14838,
   "tokens_out": 154,
   "cached_tokens": 14592,
   "cites": [
    "the-classifier-seat-five-gates-all-pass-faster",
    "the-judge-seat-and-the-floor-it-missed"
   ],
   "gates": {
    "cites": [
     "the-classifier-seat-five-gates-all-pass-faster",
     "the-judge-seat-and-the-floor-it-missed"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "3.8",
    "27",
    "13",
    "11",
    "16"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-27",
   "kind": "on-page"
  },
  {
   "slug": "the-new-kid",
   "q": "how does this compare to a result published in 2028?",
   "ok": false,
   "answer": "The article does not provide information regarding any results published in 2028 § {#the-arrival}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-arrival}. The Instrument Travels covers it more directly.",
   "abstained": true,
   "abstain_source": "gate",
   "abstain_section": "the-arrival",
   "seconds": 0.470950148999691,
   "ttft_ms": 298,
   "tokens_in": 14838,
   "tokens_out": 23,
   "cached_tokens": 14784,
   "cites": [
    "the-arrival"
   ],
   "gates": {
    "cites": [
     "the-arrival"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [
     "2028"
    ],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": true,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [
     "new noun: 1 token(s) the article does not carry"
    ],
    "new_noun_tokens": [
     "2028"
    ],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-15",
   "kind": "off-page"
  },
  {
   "slug": "llms-txt",
   "q": "did any ai crawler fetch the file in the first thirty days, and how big is the largest measured llms-full file?",
   "ok": true,
   "answer": "No ai crawler requested the llms.txt file during the original twenty-six day period, though the largest measured llms-full file weighs 30.7 MiB § {#where-the-file-came-from}. The article notes that the fourth reading found thirteen GPTBot requests that were not contained in the initial count § {#measured-again-after-publication}.",
   "served": "No ai crawler requested the llms.txt file during the original twenty-six day period, though the largest measured llms-full file weighs 30.7 MiB § {#where-the-file-came-from}. The article notes that the fourth reading found thirteen GPTBot requests that were not contained in the initial count § {#measured-again-after-publication}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.8339034949894994,
   "ttft_ms": 335,
   "tokens_in": 12269,
   "tokens_out": 79,
   "cached_tokens": 12000,
   "cites": [
    "where-the-file-came-from",
    "measured-again-after-publication"
   ],
   "gates": {
    "cites": [
     "where-the-file-came-from",
     "measured-again-after-publication"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "30.7"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-28",
   "kind": "on-page"
  },
  {
   "slug": "llms-txt",
   "q": "how many people subscribe to this site's feed?",
   "ok": true,
   "answer": "The article does not state how many people subscribe to this site's feed § {#does-anyone-fetch-it}.",
   "served": "This page doesn't say. The nearest it gets is § {#does-anyone-fetch-it}. Where your question goes — a plain-english walk. covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "does-anyone-fetch-it",
   "seconds": 0.4889402970147785,
   "ttft_ms": 298,
   "tokens_in": 12254,
   "tokens_out": 26,
   "cached_tokens": 12224,
   "cites": [
    "does-anyone-fetch-it"
   ],
   "gates": {
    "cites": [
     "does-anyone-fetch-it"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-03",
   "kind": "off-page"
  },
  {
   "slug": "the-open-call",
   "q": "how many models were handed the frozen moment, and what did the round cost?",
   "ok": true,
   "answer": "Twenty arms were seated to answer the frozen bytes § {#five-things-before-any-figure}. The metered spend for the round was $2.2826 § {#2-2826-over-56-metered-calls-against-38-49-of-caps}. The article notes that the three agent-transport arms hold no cost figure and no token counters § {#limits}.",
   "served": "Twenty arms were seated to answer the frozen bytes § {#five-things-before-any-figure}. The metered spend for the round was $2.2826 § {#2-2826-over-56-metered-calls-against-38-49-of-caps}. The article notes that the three agent-transport arms hold no cost figure and no token counters § {#limits}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.5405370400112588,
   "ttft_ms": 603,
   "tokens_in": 63357,
   "tokens_out": 90,
   "cached_tokens": 63104,
   "cites": [
    "five-things-before-any-figure",
    "2-2826-over-56-metered-calls-against-38-49-of-caps"
   ],
   "gates": {
    "cites": [
     "five-things-before-any-figure",
     "2-2826-over-56-metered-calls-against-38-49-of-caps"
    ],
    "cites_dropped": [
     "limits"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [
     "cite: 1 cite(s) not in this page's anchors: limits"
    ],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": [
     "limits"
    ]
   },
   "figures": [
    "2.2826"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-29",
   "kind": "on-page"
  },
  {
   "slug": "the-open-call",
   "q": "what did other reviewers say about this piece after it was published?",
   "ok": true,
   "answer": "The article does not provide information regarding what other reviewers said about the piece after its publication § {#the-author-of-this-page-is-a-contestant-in-it}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-author-of-this-page-is-a-contestant-in-it}. Three New Voices at the Narrator's Chair covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-author-of-this-page-is-a-contestant-in-it",
   "seconds": 1.0952636589936446,
   "ttft_ms": 562,
   "tokens_in": 63354,
   "tokens_out": 41,
   "cached_tokens": 63328,
   "cites": [
    "the-author-of-this-page-is-a-contestant-in-it"
   ],
   "gates": {
    "cites": [
     "the-author-of-this-page-is-a-contestant-in-it"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-14",
   "kind": "off-page"
  },
  {
   "slug": "where-your-question-goes",
   "q": "how long does every notebook on the server keep its entries?",
   "ok": true,
   "answer": "Every notebook on the server now keeps thirty days of entries § {#the-first-promise-your-visit-stays-between-us}. This rule applies to the tour guide's notebook as well § {#the-first-promise-your-visit-stays-between-us}. The article notes that previously, the notebooks were set to keep their lines for up to a hundred years § {#the-first-promise-your-visit-stays-between-us}.",
   "served": "Every notebook on the server now keeps thirty days of entries § {#the-first-promise-your-visit-stays-between-us}. This rule applies to the tour guide's notebook as well § {#the-first-promise-your-visit-stays-between-us}. The article notes that previously, the notebooks were set to keep their lines for up to a hundred years § {#the-first-promise-your-visit-stays-between-us}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.7529903190152254,
   "ttft_ms": 232,
   "tokens_in": 4542,
   "tokens_out": 98,
   "cached_tokens": 4288,
   "cites": [
    "the-first-promise-your-visit-stays-between-us"
   ],
   "gates": {
    "cites": [
     "the-first-promise-your-visit-stays-between-us"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-30",
   "kind": "on-page"
  },
  {
   "slug": "where-your-question-goes",
   "q": "which country is the workshop's business registered in?",
   "ok": true,
   "answer": "The article does not state which country the workshop's business is registered in § {#the-first-promise-your-visit-stays-between-us}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-first-promise-your-visit-stays-between-us}. The map nobody picks up — and whether it helps when it arrives. covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-first-promise-your-visit-stays-between-us",
   "seconds": 0.3833297839737497,
   "ttft_ms": 197,
   "tokens_in": 4541,
   "tokens_out": 34,
   "cached_tokens": 4512,
   "cites": [
    "the-first-promise-your-visit-stays-between-us"
   ],
   "gates": {
    "cites": [
     "the-first-promise-your-visit-stays-between-us"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-16",
   "kind": "off-page"
  },
  {
   "slug": "how-we-work",
   "q": "what does the page say about the humans and agents who make each exhibit?",
   "ok": true,
   "answer": "The human generates and judges while the agent amplifies and verifies § {#the-plain-english-version}. Agents perform tasks such as recon, planning, hardening, and drafting, whereas the human operator handles noticing, reviewing, steering, and the final release § {#the-thorough-walk}. The article notes that the agents' suggestions and the human's picks are combined to settle rosters, and that the agents' work is expanded by the human's judgment § {#the-plain-english-version}.",
   "served": "The human generates and judges while the agent amplifies and verifies § {#the-plain-english-version}. Agents perform tasks such as recon, planning, hardening, and drafting, whereas the human operator handles noticing, reviewing, steering, and the final release § {#the-thorough-walk}. The article notes that the agents' suggestions and the human's picks are combined to settle rosters, and that the agents' work is expanded by the human's judgment § {#the-plain-english-version}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.8156314379884861,
   "ttft_ms": 251,
   "tokens_in": 4982,
   "tokens_out": 105,
   "cached_tokens": 4736,
   "cites": [
    "the-plain-english-version",
    "the-thorough-walk"
   ],
   "gates": {
    "cites": [
     "the-plain-english-version",
     "the-thorough-walk"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-31",
   "kind": "on-page"
  },
  {
   "slug": "how-we-work",
   "q": "what is the postal address of the workshop that ran this?",
   "ok": true,
   "answer": "The article does not provide a postal address for the workshop § {#the-plain-english-version}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-plain-english-version}. The open call — a kid, an elder, and a tired parent walk into the cove. covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-plain-english-version",
   "seconds": 0.330172046989901,
   "ttft_ms": 204,
   "tokens_in": 4979,
   "tokens_out": 22,
   "cached_tokens": 4928,
   "cites": [
    "the-plain-english-version"
   ],
   "gates": {
    "cites": [
     "the-plain-english-version"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-02",
   "kind": "off-page"
  },
  {
   "slug": "the-move",
   "q": "what did the median search time go from and to after the move?",
   "ok": true,
   "answer": "The median search time went from 242 ms in the laptop era to 77.5 ms in the new era § {#receipts}. This comparison is considered the cleanest because it is the leg the move actually touched and cannot be confounded by model changes § {#receipts}. The article notes that the move put the database, the search index, and the reranker on a server-class VPS, but generation stayed on the original GPU box § {#what-moved}.",
   "served": "The median search time went from 242 ms in the laptop era to 77.5 ms in the new era § {#receipts}. This comparison is considered the cleanest because it is the leg the move actually touched and cannot be confounded by model changes § {#receipts}. The article notes that the move put the database, the search index, and the reranker on a server-class VPS, but generation stayed on the original GPU box § {#what-moved}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.7920178539934568,
   "ttft_ms": 296,
   "tokens_in": 5978,
   "tokens_out": 98,
   "cached_tokens": 5728,
   "cites": [
    "receipts",
    "what-moved"
   ],
   "gates": {
    "cites": [
     "receipts",
     "what-moved"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "242",
    "77.5"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-32",
   "kind": "on-page"
  },
  {
   "slug": "the-move",
   "q": "can you write a python script that reproduces this for me?",
   "ok": true,
   "answer": "The article does not answer this question § {#the-machine}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-machine}. The August arrivals — the fresh class sits the house exams. covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-machine",
   "seconds": 0.30344212500494905,
   "ttft_ms": 206,
   "tokens_in": 5976,
   "tokens_out": 14,
   "cached_tokens": 5952,
   "cites": [
    "the-machine"
   ],
   "gates": {
    "cites": [
     "the-machine"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-08",
   "kind": "off-page"
  },
  {
   "slug": "chair-trials",
   "q": "how many models and arms sat the five exams, and did the box stop serving?",
   "ok": true,
   "answer": "Nine models and eleven arms sat the five chairs, though only five of the eleven arms sat all five chairs § {#roster}. The serving box did not stop serving, as the article states that serving stayed up all day § {#what-the-day-actually-settled}. The article notes that the roster registers no on-disk blob size because no leg of this arc measured one § {#roster}.",
   "served": "Nine models and eleven arms sat the five chairs, though only five of the eleven arms sat all five chairs § {#roster}. The serving box did not stop serving, as the article states that serving stayed up all day § {#what-the-day-actually-settled}. The article notes that the roster registers no on-disk blob size because no leg of this arc measured one § {#roster}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.0624880190007389,
   "ttft_ms": 430,
   "tokens_in": 29973,
   "tokens_out": 84,
   "cached_tokens": 29728,
   "cites": [
    "roster",
    "what-the-day-actually-settled"
   ],
   "gates": {
    "cites": [
     "roster",
     "what-the-day-actually-settled"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-33",
   "kind": "on-page"
  },
  {
   "slug": "cove-voice-head-to-head",
   "q": "what happened to the August class's best voice at the cove's gate?",
   "ok": true,
   "answer": "The article does not state what happened to the August class's best voice at the cove's gate § {#floor-1-canon}. It only notes that the panel's warmest voice from the August arrivals was a model that invented names for the drowned diver twice § {#floor-1-canon}.",
   "served": "This page doesn't say. The nearest it gets is § {#floor-1-canon}. The August arrivals — the fresh class sits the house exams. covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "floor-1-canon",
   "seconds": 0.8024995900050271,
   "ttft_ms": 346,
   "tokens_in": 16345,
   "tokens_out": 63,
   "cached_tokens": 16096,
   "cites": [
    "floor-1-canon"
   ],
   "gates": {
    "cites": [
     "floor-1-canon"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-34",
   "kind": "on-page"
  },
  {
   "slug": "outside-judges",
   "q": "how many outside labs re-judged, and what was the widest cross-vendor gap?",
   "ok": true,
   "answer": "Four rival labs re-judged the sealed rounds § {#vendors}. The widest cross-vendor gap on any arm was 0.178 § {#vendors}. This gap occurred on muse-glimmer:30b-q8_0-dflash, where mistral read it at 0.450 against nvidia at 0.628 § {#vendors}.",
   "served": "Four rival labs re-judged the sealed rounds § {#vendors}. The widest cross-vendor gap on any arm was 0.178 § {#vendors}. This gap occurred on muse-glimmer:30b-q8_0-dflash, where mistral read it at 0.450 against nvidia at 0.628 § {#vendors}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.8089479490008671,
   "ttft_ms": 321,
   "tokens_in": 10005,
   "tokens_out": 81,
   "cached_tokens": 9760,
   "cites": [
    "vendors"
   ],
   "gates": {
    "cites": [
     "vendors"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "0.178",
    "30",
    "80",
    "0.450",
    "0.628"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-35",
   "kind": "on-page"
  },
  {
   "slug": "outside-judges",
   "q": "what is this workshop's hourly rate for consulting work?",
   "ok": true,
   "answer": "The article does not state the workshop's hourly rate for consulting work § {#the-rest-of-the-seminar}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-rest-of-the-seminar}. qwen3.8:27b across four house benches: one seat filled, one floor missed covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-rest-of-the-seminar",
   "seconds": 0.47013424400938675,
   "ttft_ms": 286,
   "tokens_in": 9998,
   "tokens_out": 28,
   "cached_tokens": 9952,
   "cites": [
    "the-rest-of-the-seminar"
   ],
   "gates": {
    "cites": [
     "the-rest-of-the-seminar"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-11",
   "kind": "off-page"
  },
  {
   "slug": "august-arrivals",
   "q": "how many models landed that week, and what share broke the response contract?",
   "ok": true,
   "answer": "Four fresh models arrived that week, and all four broke the response contract on more than 10% of calls § {#exam-one-the-judge-chair}. Specifically, the failure rates for the models were 16.3%, 51.2%, 55.8%, and 62.8% § {#exam-one-the-judge-chair}. The article notes that the fourth model, olmo-3.1:32b-think-q4_K_M, was never on the voice leg's roster § {#exam-two-the-narrators-chair}.",
   "served": "Four fresh models arrived that week, and all four broke the response contract on more than 10% of calls § {#exam-one-the-judge-chair}. Specifically, the failure rates for the models were 16.3%, 51.2%, 55.8%, and 62.8% § {#exam-one-the-judge-chair}. The article notes that the fourth model, olmo-3.1:32b-think-q4_K_M, was never on the voice leg's roster § {#exam-two-the-narrators-chair}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.0851111949887127,
   "ttft_ms": 310,
   "tokens_in": 10936,
   "tokens_out": 132,
   "cached_tokens": 10688,
   "cites": [
    "exam-one-the-judge-chair",
    "exam-two-the-narrators-chair"
   ],
   "gates": {
    "cites": [
     "exam-one-the-judge-chair",
     "exam-two-the-narrators-chair"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "10",
    "16.3",
    "51.2",
    "55.8",
    "62.8",
    "3.1",
    "32",
    "4"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-36",
   "kind": "on-page"
  },
  {
   "slug": "seat-trials",
   "q": "what are the kill-recall and preservation floors a model must clear?",
   "ok": true,
   "answer": "A model must achieve a kill-recall of at least 23/27 and a preservation of at least 13/16 to clear the floors § {#the-instrument}. These thresholds were written down before any model ran and each floor binds on its own § {#the-instrument}. The article notes that one candidate was blocked at the billing probe and never ran § {#the-instrument}.",
   "served": "A model must achieve a kill-recall of at least 23/27 and a preservation of at least 13/16 to clear the floors § {#the-instrument}. These thresholds were written down before any model ran and each floor binds on its own § {#the-instrument}. The article notes that one candidate was blocked at the billing probe and never ran § {#the-instrument}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.7133326710027177,
   "ttft_ms": 251,
   "tokens_in": 6987,
   "tokens_out": 83,
   "cached_tokens": 6752,
   "cites": [
    "the-instrument"
   ],
   "gates": {
    "cites": [
     "the-instrument"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "23/27",
    "13/16"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-37",
   "kind": "on-page"
  },
  {
   "slug": "seat-trials",
   "q": "how many models will sit this exam next year?",
   "ok": true,
   "answer": "The article does not state how many models will sit this exam next year § {#the-instrument}.",
   "served": "This page doesn't say. The nearest it gets is § {#the-instrument}. The Instrument Travels covers it more directly.",
   "abstained": true,
   "abstain_source": "model",
   "abstain_section": "the-instrument",
   "seconds": 0.3391742020030506,
   "ttft_ms": 207,
   "tokens_in": 6983,
   "tokens_out": 21,
   "cached_tokens": 6944,
   "cites": [
    "the-instrument"
   ],
   "gates": {
    "cites": [
     "the-instrument"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "off-21",
   "kind": "off-page"
  },
  {
   "slug": "voice-trials",
   "q": "how many model tags were tried, and what does the anchor model read?",
   "ok": true,
   "answer": "The article states that 24 distinct model tags were tried across 40 scored runs § {#provenance-4cae43}. The anchor model reads 9.0, 7.0, and 6.2 on three different record sets, remaining unchanged across them § {#the-method-limits-first}. As a note, the article mentions that no delta across record sets is ever published and that the 2026-08 field never sampled a per-arm resident size § {#the-2026-08-field-six-arms-on-the-96g-workstation-blind}.",
   "served": "The article states that 24 distinct model tags were tried across 40 scored runs § {#provenance-4cae43}. The anchor model reads 9.0, 7.0, and 6.2 on three different record sets, remaining unchanged across them § {#the-method-limits-first}. As a note, the article mentions that no delta across record sets is ever published and that the 2026-08 field never sampled a per-arm resident size § {#the-2026-08-field-six-arms-on-the-96g-workstation-blind}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.0909319079946727,
   "ttft_ms": 266,
   "tokens_in": 10109,
   "tokens_out": 136,
   "cached_tokens": 9856,
   "cites": [
    "provenance-4cae43",
    "the-method-limits-first",
    "the-2026-08-field-six-arms-on-the-96g-workstation-blind"
   ],
   "gates": {
    "cites": [
     "provenance-4cae43",
     "the-method-limits-first",
     "the-2026-08-field-six-arms-on-the-96g-workstation-blind"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "24",
    "40",
    "9.0",
    "7.0",
    "6.2",
    "2026-08"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-38",
   "kind": "on-page"
  },
  {
   "slug": "licences",
   "q": "what does the page say it does with each model's licence?",
   "ok": true,
   "answer": "The page states that each licence is read in full and the date of the last full read is recorded on each row § {#how-to-read-this-page}. The method for each read is named once per table so that the reader can repeat the read themselves § {#how-to-read-this-page}. The article notes that unread models are withheld and that the diffusion bench's 1,261 considered renders split 661 published, 338 held, and 262 withheld § {#provenance-063a00}.",
   "served": "The page states that each licence is read in full and the date of the last full read is recorded on each row § {#how-to-read-this-page}. The method for each read is named once per table so that the reader can repeat the read themselves § {#how-to-read-this-page}. The article notes that unread models are withheld and that the diffusion bench's 1,261 considered renders split 661 published, 338 held, and 262 withheld § {#provenance-063a00}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.9723546169989277,
   "ttft_ms": 262,
   "tokens_in": 8383,
   "tokens_out": 122,
   "cached_tokens": 8128,
   "cites": [
    "how-to-read-this-page",
    "provenance-063a00"
   ],
   "gates": {
    "cites": [
     "how-to-read-this-page",
     "provenance-063a00"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "1,261",
    "661",
    "338",
    "262"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-39",
   "kind": "on-page"
  },
  {
   "slug": "hear-it-for-yourself",
   "q": "how many clips does the page carry, and how many presentations made up the sitting?",
   "ok": true,
   "answer": "The page carries eighty-four clips § {#three-sets-of-bytes-and-which-one-the-player-fetches}, and the sitting was composed of thirty-six presentations § {#the-blind-sheet-unsealed}. These presentations consisted of twenty-four scoring pairs and twelve sentinels § {#the-blind-sheet-unsealed}. The article notes that the twenty-four half-strength renders were never put in front of anyone § {#the-blind-sheet-unsealed}.",
   "served": "The page carries eighty-four clips § {#three-sets-of-bytes-and-which-one-the-player-fetches}, and the sitting was composed of thirty-six presentations § {#the-blind-sheet-unsealed}. These presentations consisted of twenty-four scoring pairs and twelve sentinels § {#the-blind-sheet-unsealed}. The article notes that the twenty-four half-strength renders were never put in front of anyone § {#the-blind-sheet-unsealed}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 1.0322037300211377,
   "ttft_ms": 332,
   "tokens_in": 18661,
   "tokens_out": 107,
   "cached_tokens": 18400,
   "cites": [
    "three-sets-of-bytes-and-which-one-the-player-fetches",
    "the-blind-sheet-unsealed"
   ],
   "gates": {
    "cites": [
     "three-sets-of-bytes-and-which-one-the-player-fetches",
     "the-blind-sheet-unsealed"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-40",
   "kind": "on-page"
  },
  {
   "slug": "the-ceiling-is-not-the-corpus",
   "q": "how did the adapter score against the base model in the listening, and on how many house prompts did it win?",
   "ok": true,
   "answer": "The adapter did not clear the listening threshold, as the listener expressed a preference for the untouched model in 11 of the 15 answered pairs § {#the-blind-sitting}. The adapter won zero of the house prompts, as the listener chose the untouched model for all five house-related requests where a preference was expressed § {#what-the-adapter-actually-did}. The article notes that the listener abstained on 9 of the 24 pairs and that the sentinel block did not accept under the registered design § {#the-blind-sitting}.",
   "served": "The adapter did not clear the listening threshold, as the listener expressed a preference for the untouched model in 11 of the 15 answered pairs § {#the-blind-sitting}. The adapter won zero of the house prompts, as the listener chose the untouched model for all five house-related requests where a preference was expressed § {#what-the-adapter-actually-did}. The article notes that the listener abstained on 9 of the 24 pairs and that the sentinel block did not accept under the registered design § {#the-blind-sitting}.",
   "abstained": false,
   "abstain_source": null,
   "abstain_section": null,
   "seconds": 0.9916669219965115,
   "ttft_ms": 319,
   "tokens_in": 8749,
   "tokens_out": 116,
   "cached_tokens": 8480,
   "cites": [
    "the-blind-sitting",
    "what-the-adapter-actually-did"
   ],
   "gates": {
    "cites": [
     "the-blind-sitting",
     "what-the-adapter-actually-did"
    ],
    "cites_dropped": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "redacted": false,
    "figure_unverified": false,
    "uncited": false,
    "new_noun": false,
    "blocked": false,
    "directive_hit": false
   },
   "gate_detail": {
    "reasons": [],
    "new_noun_tokens": [],
    "new_noun_tokens_withheld": 0,
    "cites_dropped": []
   },
   "figures": [
    "11",
    "15",
    "9",
    "24"
   ],
   "figures_verbatim": true,
   "cites_valid": true,
   "redaction_hits": 0,
   "id": "on-41",
   "kind": "on-page"
  }
 ],
 "verdicts": {
  "on_page_answered": true,
  "non_verbatim_figures": true,
  "off_page_abstained": true,
  "redaction_hits": true,
  "p50_warm_seconds": true,
  "p95_warm_seconds": true
 },
 "passed": true
}
