{
  "schema": "s2s-bench-v1",
  "kind": "exhibit-data-discovery",
  "site": "strata→signal research",
  "site_url": "https://research.strata2signal.com/",
  "generated_utc": "2026-09-04",
  "licence": "CC BY 4.0 on every data kit listed here",
  "attribution": "strata→signal research, research.strata2signal.com",
  "contact": "hello@strata2signal.com",
  "posture": "Every published table on this site is meant to ship a JSON companion emitted by the same code that renders the page's rows, so page and data reconcile by construction. Exhibits published before that law carry `data: null` until their kit is backfilled; their raw rows are on request in the meantime, and saying so is the point of listing them here as null rather than omitting them.",
  "order_note": "listed in the site's curation order; the exhibit number is publication-order identity and never renumbers",
  "not_a_standard": "Our own trials, our own hardware, for our own chairs. Not first independent numbers, and not a benchmark.",
  "exhibits": [
    {
      "exhibit": 38,
      "slug": "the-beat-lab-asked-and-answered",
      "title": "The Beat Lab, asked and answered",
      "url": "https://research.strata2signal.com/the-beat-lab-asked-and-answered/",
      "published": "2026-09-04",
      "data": null,
      "data_note": "No kit. The page is DERIVED, not written: the Beat Lab's own repository holds the question bank (beat/bank.json, twenty-five entries, fingerprinted bank 92f6c73c2afc — the first twelve hex of the file's sha256, and the same twelve hex the lab stamps into its page at every build) and the script that reads it into this page (tools/bank_faq_md.py); the lab's own test re-derives the page from the bank and byte-compares. Each answer is set beside the sentence of exhibit twenty-eight it was written from, resolved by the lab's citation checker, which refuses to build when an answer and its sentence part company. No model is asked for any answer and none ran to make the page.",
      "licence": "CC BY 4.0",
      "one_liner": "The Beat Lab, asked and answered: the twenty-five questions the lab's genie can answer about the lab itself, answered here as a page — the same twenty-five answers, out of the same file the genie reads (the lab's question bank, fingerprinted bank 92f6c73c2afc), each set beside the sentence of the field guide it was written from, so an edit to one word moves the genie's answer and this page in the same commit. An answer in the lab is instant and spends no wish; no model is asked for one, and none ran to derive the page."
    },
    {
      "exhibit": 36,
      "slug": "ten-minutes-with-living-artists",
      "title": "Ten Minutes with Living Artists",
      "url": "https://research.strata2signal.com/ten-minutes-with-living-artists/",
      "published": "2026-09-03",
      "data": "https://research.strata2signal.com/ten-minutes-with-living-artists/data/index.json",
      "data_note": "Forty-five files. index.json lists forty-two shipped artefacts with sizes, sha256s and the rules applied to each: the ten clips' full digests and the player-to-file map (every digest re-read off the staged bytes), both training runs' logs start to finish, the two datasets' splits with the holdout rule as executable code and the caption rule's own PLACEHOLDER status, the 208-row intake manifest with its checksums beside the 41 releases' licence records, the bandwidth audit with its twenty flagged rows, the eight-track exclusion and the ruling behind it, all four rounds' render records with the loudness the harness wrote and the raw sample-peak and flat-factor diagnostic for all sixty-five renders, the coherence bench's sealed pre-registration with its five amendments and its complete output down to every embedding, and Round 4's blind-sheet layout without its key. Two disagreements are published rather than resolved: the trainer's banner against its own per-pass clock (s-per-step.md gives both and the counting rule for each), and two loudness key spellings across the rounds. provenance.json's 'omitted' section is the list of what is not here and why.",
      "licence": "CC BY 4.0",
      "one_liner": "A genre corpus instead of a composer — 41 Creative-Commons releases, 159 tracks, ten minutes of training on a workstation card — produced the first adapter this shelf's only listener preferred to switch off; a one-artist control trained in ninety-three seconds to test why, and a pre-registered embedding bench measured the corpus coherence the explanation rests on."
    },
    {
      "exhibit": 37,
      "slug": "listen-for-yourself",
      "title": "Listen for Yourself",
      "url": "https://research.strata2signal.com/listen-for-yourself/",
      "published": "2026-09-03",
      "data": [
        {
          "url": "https://research.strata2signal.com/ten-minutes-with-living-artists/data/index.json",
          "role": "exhibit thirty-six's kit, which owns every measured record behind this page: all four rounds' render records with the loudness the harness wrote, the raw sample-peak and flat-factor diagnostic for all sixty-five renders, the ten article clips' digests and the player-to-file map, and Round 4's blind-sheet layout without its key",
          "media_type": "application/json"
        }
      ],
      "data_note": "POINTERS, not a kit of this page's own — deliberately, and the split is a licence boundary rather than a filing convenience. Every number printed here is read off exhibit thirty-six's records, which are CC BY 4.0 and live in that page's kit; the AUDIO is share-alike, because the corpus that trained the adapter was, so the WAVs sit beside this page at /listen-for-yourself/audio/ (63 raw renders) and /listen-for-yourself/audio/master/ (63 all-linear listening masters, one constant gain per file by gain = min(−16 − I, −1 − TP), nothing limited), each directory carrying its own MANIFEST.sha256 and the masters carrying master-records.json — the rule, the ffmpeg version, and each clip's loudness and true peak before and after. Sixty-five clip slots, sixty-three files: round three re-rendered round two's two untouched clips and got byte-identical bytes, so they are registered once and round three points back rather than serving the same bytes twice, and the page does that arithmetic on its own face.",
      "licence": "CC BY 4.0",
      "one_liner": "Every render of the fourth adapter's arc — four rounds, sixty-five clip slots, sixty-three files — each one raw and again at a common loudness with both fingerprints and the gain printed beside its player, so a reader can hear what one listener heard and disagree with the verdict on the evidence."
    },
    {
      "exhibit": 35,
      "slug": "half-an-hour-with-dead-composers",
      "title": "Half an Hour with Dead Composers",
      "url": "https://research.strata2signal.com/half-an-hour-with-dead-composers/",
      "published": "2026-09-02",
      "data": "https://research.strata2signal.com/half-an-hour-with-dead-composers/data/index.json",
      "data_note": "Thirty files. index.json lists twenty-seven shipped artefacts with sizes, sha256s and the rules applied to each: the eleven clips' full digests and the player-to-file map (every digest re-read off the served bytes), five render requests verbatim, the pass and step lines of both trainers' logs quoted with the receipts' digests (the log files did not survive their run directories — named absent, not reconstructed), six adapter PEFT configs and the merged adapters' own records including the one merge with no build log, the merge instrument and its finale log, five distance tables including the pair that killed the retracted sentence and the four-seed noise floor, and both rights chains carrying each corpus manifest's own digest (the manifests themselves are named absent). provenance.json's 'omitted' section is the list of what is not here and why.",
      "licence": "CC BY 4.0",
      "one_liner": "Two more dead composers taught the same way — a hundred and six marches, then Bach — then blending composers with linear algebra: the obvious merge is wrong, the right one costs a measurable one percent, and a sentence already staged for publication is taken back with its receipt."
    },
    {
      "exhibit": 34,
      "slug": "chopin-in-five-minutes",
      "title": "Teaching a Music Model Chopin in Five Minutes",
      "url": "https://research.strata2signal.com/chopin-in-five-minutes/",
      "published": "2026-09-01",
      "data": "https://research.strata2signal.com/chopin-in-five-minutes/data/index.json",
      "data_note": "Ten files. index.json lists seven shipped artefacts with sizes, sha256s and the rules applied to each (it, provenance.json and the browsable index.html are not listed among the files they describe): the five clips' full digests and the player-to-file map, the render request verbatim, the training log with its ten epoch lines, the adapter's PEFT config, the 24-row corpus manifest with per-file checksums and source URLs, and the rights-chain receipts. Two files carry a projection and no other change — 17 absolute paths in the log, 1 in the config, both to role-relative form; every measured value is as the instrument wrote it. The five audio clips themselves ride beside the article at /chopin-in-five-minutes/audio/ with their own MANIFEST.sha256, served as the exact rendered bytes so the fingerprints printed on the page check against the files it plays.",
      "licence": "CC BY 4.0",
      "one_liner": "What a LoRA adapter does to an open-weights music model after 4 minutes 49 seconds of training on nineteen public-domain Chopin recordings, on one five-year-old 24 GB consumer card. Five 60-second clips at one request and one seed, with the adapter's strength the only thing that moves — 0%, 25%, 50%, 75%, full — playable on the page. The measured claims are the training cost (fifty weight updates, 5.7 GB peak, per-pass error 0.94 to 0.61 with a bump at pass nine) and that each step of the dial moves the audio measurably further from the baseline without backtracking. What is NOT claimed is that any of it sounds like Chopin: no blinded listening test has been run, and the page says so twice."
    },
    {
      "exhibit": 33,
      "slug": "the-same-sixteen",
      "title": "The Same Sixteen",
      "url": "https://research.strata2signal.com/the-same-sixteen/",
      "published": "2026-08-31",
      "data": "https://research.strata2signal.com/the-same-sixteen/data/index.json",
      "data_note": "nineteen files, pointed at through the kit's own manifest rather than enumerated here; index.json lists seventeen shipped artefacts with sizes, sha256s and the rules applied to each (it, provenance.json and the browsable index.html are the other two), and provenance.json carries a second sha256 per file — the pre-sanitisation one — for everything not copied byte for byte. What is in it: the full results narrative for the arm, ending in TWO dated corrections (2026-08-30) appended rather than folded in, so a reader who quotes §3, §5 or §7 without reading to the end will quote a corrected sentence; RESULTS-TABLES.md and C3-RESULTS.md, the latter putting all nineteen toolbench tasks against each of the four rows with the checker's own note on every one; the gate receipt, which is the evidence behind every leg reported NOT-RUN; the field exam per item, all sixty including the twenty that were excluded; every scored attempt and every individual tool call the arm made; the run's own manifest; the 2026-08-12 local rows the count is set beside; the reference-arm registration as it was written before the run; and the counting rules. THE TWO FROZEN ARTEFACTS SHIP SO THE PAGE'S FINGERPRINTS ARE TAKEABLE: tasks.json is byte-identical to the pinned file, and tools-manifest.json is the tools array in the canonical form the run hashes, so `sha256sum` on either returns the sha printed on the page. TWO things are deliberately left as captured and one is named as unresolved: the frozen fixture's own fictional harbour clock (92 occurrences, quoted inside graded answers a published checker matches, so converting it would destroy the re-derivability of those grades), tasks.json's bytes, and the narrative's own second correction, which records that its evidence names TWO literal-string checker tasks where the text had said three — the recount is owed and the claim is not cited at three anywhere. Four things are omitted and named rather than implied by absence: the bench's runner and harness, the 2026-08-26 battery's scorer file and lane log (already public in exhibit thirty-one's kit, and neither was modified by this run), the rest of the bench's plan, and any currency figure, because none was taken.",
      "licence": "CC BY 4.0",
      "one_liner": "What `glm-5.3-flash:cloud` — a hosted model its vendor lists at 320 billion total parameters and 18 billion active per token — does on two of this workshop's own frozen instruments, read once on 2026-08-28 between 16:37:59Z and 16:49:43Z. It is a REFERENCE ARM · cloud · dated, a row class registered before the run: never a candidate for any seat here (ollama publishes the tag cloud-only, and this house does not put cloud AI inside live products), ranked against nothing, and no threshold is ever derived from it. NINE instruments sit on its row and TWO could honestly be read. Four were refused by the path: the judge and assistant exams bind their verdicts on a JSON-schema grammar the cloud path accepts and silently ignores (ollama issue #12362, re-proved at 16:49:43Z after the last scored call), and decode speed and long-context recall mean nothing measured over the public internet against someone else's fleet under someone else's batching. Three never applied to a model that cannot hold a seat. Every one carries its reason rather than a blank, and the gate receipt is in the kit. THE HEADLINE IS A TIE AND THE FINDING IS ABOUT THE RULER: the arm scored 16/19 on the frozen toolbench (84.2%, Wilson 95% [62.4%, 94.5%]), which is the exact count `gemma4:12b` and `qwen3.6:27b` — local, dense, 12 and 27 billion parameters — set on the same nineteen tasks on 2026-08-12, and the production seat `gemma4:26b` reads 14/19 and 15/19 on the two postures it runs. Against a 16/19 row this bench separates a comparator only at a gap of seven tasks or more (Fisher exact, α = 0.05: 16-vs-9 separates, 16-vs-10 does not), so most of its own table falls inside its own resolution: models that far apart in scale can land on the same count and this instrument has no way to tell whether their true rates differ at all. The field exam says the same thing from the other side — the arm 39/40 on the forty items it could sit, against local rows of 39 to 40 out of 40 — an exam that has stopped discriminating at this level. The texture under the count is the part a headline loses: right final answer on 5 of 5 call chains, the best this instrument has produced, while following the pre-registered call sequence on 1 of 5; unnecessary calls at 44.3% (39 of 88), inside the local 23.8–47.3% range and dominated by round-cap loops, so no interval is printed on it; and argument fidelity last of every row this instrument has produced, 49 of 66 slots (74.2%) against a local spread of 54 to 59. TWO METHOD FINDINGS TRAVEL FURTHER THAN ANY SCORE. `think:false` does not switch this model's reasoning off on this path — it only decides whether ollama parses the reasoning out of the reply, so the raw deliberation lands in the field every leg scores: four probes, four leaks, three closed by a literal `</think>` and one not delimited at all. That is a measurement failure and not a low score, so it was never run and every figure here is `think:true`. And two of the nineteen tasks are graded by checkers that match string literals and therefore fail correct answers — one accepts only the word \"depth\" and rejected \"a measure of how deep she sits in the water\"; the other wants an ISO date and rejected \"09:00 on Wednesday, 12 August 2026\" — and the 26B seat and the 320B arm fail both identically, which makes every row those two tasks touch a floor rather than a reading. The frozen numbers stay as scored; widening a checker is a pre-registered revision applied to every row equally. One costing figure for anyone sizing a hosted tool-calling workload: across the 104 toolbench requests prompt tokens ran 24× completion tokens (195,532 against 8,115), 17× pooled across all 164 requests the arm made, because a toolbench resends the tool schemas and the whole transcript every round — and the completion count includes reasoning no dial removes on this model. No currency figure is quoted, because none was taken."
    },
    {
      "exhibit": 32,
      "slug": "two-hours-on-battery",
      "title": "Two Hours at 12 tok/s — On Battery",
      "url": "https://research.strata2signal.com/two-hours-on-battery/",
      "published": "2026-08-31",
      "data": "https://research.strata2signal.com/two-hours-on-battery/data/index.json",
      "data_note": "seventy-four files, so this points at the kit's own manifest rather than enumerating them here; the manifest lists all seventy-one shipped artefacts with sizes, sha256s and the rules applied to each (the manifest, provenance.json and the browsable index.html are the other three), and provenance.json carries a second sha256 per file — the pre-sanitisation one — for everything that was not copied byte for byte. What is in it: runs/, the main bench's 31 scored arms plus cpu-server-m1.CONTAMINATED.json, the arm the bench discarded and kept under that name; supplement-runs/, the same evening's tuned-thread arms, the same-session t=4 control, both thread sweeps with their stability repeats and the dedicated MoE sweep; receipts/, the STREAM-triad bandwidth run on each machine and the CPUID core-type walk on both hybrid boxes; analysis/, the per-cell dispersion table, the bandwidth table and the thread-count cause with its four receipts; quality/, the sealed field exam and the 374-item moderation set with its pre-registered gates; and at the root the counting rules, the pre-registration with all eight deviations, and NOTES.md, which re-derives every headline figure with the command that produces it. TWO sittings, and they are dated separately: the main bench 2026-08-25 14:09:19Z–17:17:57Z and the supplement 17:57–19:46Z the same day. TWO omissions are named in the kit rather than implied by absence: the bench's own working ledgers and results tables, which name machines and paths throughout and whose measurements are re-derived here instead of quoted; and the frozen prompt's TEXT — sha-pinned on the page, in prereg.md and in every arm's prompt_sha256, but held pending a licensing call on the source, so a reader can verify a copy they already hold and cannot reconstruct it from the kit. One sanitisation is worth knowing about before you read a timestamp: Ollama's /api/ps returns a loaded model's expires_at on the serving machine's own local clock, so 42 of those fields across 22 files were converted to the same instant Z-stamped, under a named and counted rule",
      "licence": "CC BY 4.0",
      "one_liner": "What six language models do on three machines with no graphics card in the path, measured on 2026-08-25 on a 32-core server, a hybrid laptop and a hybrid mini PC, all six tags public and all Q4_K_M under Ollama 0.32.15. The headline inverts the intuition that bigger is slower: the 30B mixture-of-experts (~3B active per token) out-decodes the 9B dense model on every machine, and the two MoEs beat the nearest dense model tested — a 24B — by 3.5–5.7× at the default thread count. The mechanism is checked rather than asserted: CPU decode is a memory-bandwidth game, so tok/s should track each box's measured bandwidth, and it does — gemma4:26b decodes at 24.7 / 13.3 / 12.7 tok/s against 100.7 / 59.3 / 51.1 GB/s, giving 0.245 / 0.224 / 0.248, the same constant to within a tenth. The second finding is the one free lever: llama.cpp's thread-count heuristic walks the performance cores and skips every second one as a hyperthread sibling, which on a hybrid laptop with no hyperthreading at all returns 4 where the machine has 8 — so every dense model there was running at up to 1.9× below its own tuned speed, while the same tuned setting forced onto an MoE costs it half its throughput and, at 16 threads on the mini, 11× (14.4 → 1.27 tok/s). Dense models want every core, MoEs want a handful, and the buggy default happens to sit near the MoE optimum. Give every model its fair thread count and the MoE-over-dense advantage narrows from 3.5–5.7× to 2.6–4.4× and the headline still holds, 1.2× to 2.3×. Prefill, not decode, is what more cores buy: the server leads the mini by 3.1–4.4× on prefill and only 1.5–2.2× on decode. Two counting traps are published with their corrections rather than folded away: a repeated prompt is prefix-cached, so a re-sent first token comes back 76× faster than a fresh one on the same cell (310× on the worst, 96× if the model load is counted in), and the concurrency-1 rows must be read at the median rep (24.5 → 17.4 tok/s per stream for 1.38× total throughput), not across reps. On judgment: all six cleared every bar of the sealed 60-item field exam at 57–59 of 60 — a floor-check that saturated — and on the 374-item moderation set the four candidates differ from the production baseline by 2, 3, 4 and 5 items, every disagreement on clean text, a band too tight to rank and reported as such. GPU-absence is a receipt, not a claim: 19 scored decode-and-prefill arms carry size_vram 0, no bench process in nvidia-smi's compute-apps list, and a byte-identical device snapshot before and after; the twentieth file carrying all three is the discarded arm, and its snapshot is the only one in the pack that changed, which is how the bench knew to drop it"
    },
    {
      "exhibit": 31,
      "slug": "the-instrument-travels",
      "title": "The Instrument Travels",
      "url": "https://research.strata2signal.com/the-instrument-travels/",
      "published": "2026-08-29",
      "data": "https://research.strata2signal.com/the-instrument-travels/data/index.json",
      "data_note": "177 files, so this points at the kit's own manifest rather than enumerating them here; index.json lists 175 of them with sizes, sha256s and the rules applied to each, provenance.json lists 174 with a SECOND sha256 apiece — the pre-sanitisation one — for everything not copied byte for byte, and index.html is the browsable listing (the three describe each other and so cannot list themselves). What is in it: the five fit receipts at a 32,768-token context, which are the only instrument all five candidates completed; every scorer file the page's five tables read (summary-C1-<tag>.json, seat43-summary.json, c2-scores.json, c3-scores.json, c5-recovered.json, c7-<tag>.json, field/<tag>.json); the schema probes behind the SEAT-BLOCKED and inverted-trap findings; raw/, the per-call records for every leg, INCLUDING raw/seat43/, whose 390 records carry that exam's own num_ctx 16384 and think:omit exactly as sent, because the disclosure that one exam ran outside the battery's house laws is only checkable if those bytes are published unaltered; digests.md, the build identity for every tag with the 2026-08-12 comparison and the size-vs-digest anomaly recorded as an open flag; battery.log, the lane log with the disk halt and resume as data rather than an outage note; TEARDOWN-RECEIPT.md, the four-guard teardown with a read-only probe of every live seat; prereg/, the judge trial's registration (sha256 423ed960…) and the 21-item judge fixture's hash (fede4e32…); and counting-rules.md, which states the count-travels rule, the 10% ceiling, the two-item tie band and the judge-seat deviation in one place. ONE sitting, dated at both ends: 2026-08-26 07:00:15Z to 15:13:00Z, halted by the harness's own disk floor at 09:58:39Z, a gap-fill lane to 10:14:59Z, scoring resumed 13:55:19Z. TWO omissions are named in the kit rather than implied by absence: the bench's runner, and a fully runnable fixture set — the kit ships receipts, not a harness. One sanitisation is worth knowing before reading a timestamp: every stamp of the run itself is UTC and Z-stamped, and the one exception is fixture content — the tool-use trial's 19 frozen tasks are set in a fictional harbour with its own scenario clock dated 2026-08-01/-02/-12, published as captured, 185 of them, because the published checker matches the clock reading itself and converting them would rewrite graded answers",
      "licence": "CC BY 4.0",
      "one_liner": "Whether a frozen exam says the same thing in a smaller room. Five models sat the SAME registered battery that ranked their predecessors on a 96 GB workstation card — same items, same seeds, same pass marks — on a 24 GB consumer card on 2026-08-26, one at a time, that card emptied for the run, under Ollama 0.32.13 at a 32,768-token context with think:false explicit on every scored call but one exam's. THE DOOR FIRST: three of five board whole at 100% on GPU (muse-glimmer:30b 15,831 MiB, qwen3.6:27b 16,469, laguna-xs-2.1:latest 19,440); gemma4:31b runs 92.5% on GPU and spills 1,541 MiB; nemotron-3.5-lightning:30b spills 4,136 MiB at 83.1% on GPU, its 25.43 GB of shipped weights leaving nowhere to sit once a 32k context loads beside them. Speed and long-context rows are SKIPPED rather than measured through a spill, by rule. THE REPRO ROW is the point of the piece: qwen3.6:27b earned RANKED on 2026-08-12 at kill-recall 11/12 and preservation 9/9; re-pulled at the same tag it came back a DIFFERENT build (manifest digest 9d5803d4… against the earlier a50eda8e…), on different silicon, and the frozen exam read 11/12 and 8/9 — both floors cleared, RANKED again, with the one-item preservation move inside the pre-registered two-item tie band and all three of its truncated calls landing on a single item, which therefore carries no verdict. The detail that makes it measurement rather than memory: the kill-recall count is identical but the MISSED ITEM MOVED, from c1-k-bez-c1 to c1-k-bez-h1, a different case in the same family at the same score. Two other frozen instruments sat both runs and counts travel across cards, so both are fair to compare: the assistant trial reproduced EXACTLY at 15/20 (11 of 13 exact-checked items, 4 of 7 by the registered proxy rule, zero response failures), and the toolbench SLIPPED, 16/19 task success to 14/19 and grounded 14/19 to 12/19, a two-item move the tie band does not cover and a build-to-build move rather than a card-to-card one. One reproduced exactly, one at the floor, one slipped — which is what an instrument travelling at n=1 honestly looks like. THE OTHER FOUR CANDIDATES returned four different kinds of nothing. muse-glimmer:30b is SEAT-BLOCKED and the diagnosis inverts the first write-up: under an enforced schema it emits schema-perfect JSON and the runtime then leaks a trailing end-of-text sentinel into the bytes, which breaks a strict parser — the verdict stands because strict parse binds, but it is a one-line fix at the caller, not a model to discard, and the same model was flawless on the long-context filing exam at 18/18 recall, 18/18 correct abstentions and zero fabrications across 72 calls. gemma4:31b caught ALL 27 planted kills on the judge seat's 43-case exam and FAILED anyway, on preservation 9/16 against a floor of 13 — which is exactly what that trial exists to price — and 117 of its 129 replies arrived wrapped in a code fence and were unwrapped before scoring under that exam's own frozen rule, where the other two candidates arrived at zero fenced replies each. laguna-xs-2.1:latest showed the battery's only INVERTED trap — its schema output parses clean under think:false, the posture these seats contract, and drops the constraint under think:true, the opposite of the documented majority — and then lost its battery twice, first to the disk stop mid-leg and then to the time-box, leaving 19 valid calls across 7 of the judge trial's 21 items that ship in the kit and are not a result. nemotron-3.5-lightning:30b was time-boxed to its fit gate; its manifest digest is byte-identical to 2026-08-12's, so that row was a harness-and-hardware check rather than a model comparison, and its earlier judge row had failed both floors anyway. TWO candidates came back UNMEASURABLE on the judge seat — glimmer at 18 of 129 calls failing to score (14%) and qwen at 93 of 129 (72%), truncations against that exam's own 1,024-token reply budget in its own 16,384-token context — and the counting rule differs between the two exams by design: this battery's judge trial puts a truncation in its own run-quality bucket outside the 10% ceiling, the older seat exam counts it as a response failure inside it, and the seat exam runs at its OWN registered settings because comparability against its published floors of 23/27 and 13/16 is its entire value. Decode is near-flat across a 32× prompt spread (qwen 62.0 / 60.8 / 59.1 tok/s at 1k / 8k / 32k prompts, glimmer 41.1 and 39.5 with its 8k tier NOT-CARRIED on a build-specific empty-counter anomaly), and rates are never compared to the 96 GB run, because a rate belongs to the model AND the silicon. NOTHING EARNED A SEAT: qwen returned a ranked or clean result on seven of the eight exams it sat and the eighth was refused by its own budget; a re-run would be a new exam version applied to every candidate equally, published fenced from the floors. And the fifth result is the bench's own: it HALTED ITSELF at a conservative disk floor rather than gamble on the margin, resumed only after every digest re-verified, and tore its model store down under four guards — weights are re-pullable, measurements are not."
    },
    {
      "exhibit": 30,
      "slug": "nine-worlds-one-dog",
      "title": "Nine Worlds, One Dog",
      "url": "/nine-worlds-one-dog/",
      "published": "2026-08-28",
      "data": "https://research.strata2signal.com/nine-worlds-one-dog/data/index.json",
      "data_note": "four files (prompts.json, README, index.json/html); prompts.json carries all nine verbatim prompts, seeds, sizes, steps and per-render seconds",
      "licence": "CC BY 4.0",
      "one_liner": "The founder's dog takes the same nap under the same tree in all nine painted worlds — same dog, same curl; only the world changes."
    },
    {
      "exhibit": 29,
      "slug": "what-150-watts-buys",
      "title": "What 150 Watts Buys",
      "url": "https://research.strata2signal.com/what-150-watts-buys/",
      "published": "2026-08-28",
      "data": "https://research.strata2signal.com/what-150-watts-buys/data/index.json",
      "data_note": "eighty-three files — the largest kit on this shelf, so this points at the kit's own manifest rather than enumerating it here; the manifest lists all eighty-three with sizes, sha256s and the rules applied to each, and the directory also carries a browsable index.html. What is in it: the twelve rungs of the 2026-08-25 power-cap arm (per-rung summary.json, raw.jsonl, power.csv, throttle.csv and throttle-delta.json), the noise-floor table the verdicts are read against, the morning serving bench's two reference legs, NOTES.md and README.md — and under ladder/, a SECOND sitting with its own registration and window: the 2026-08-27 cap ladder on the image lane at 600/500/450 W, its four dated amendments, the sealed 139-cell registry, 605 render rows, the serving lane probed at every rung, three per-rung power CSVs whose every sample carries the limit it was taken under, the three one-shot nvidia-smi flag reads (perf-600W.txt / perf-500W.txt / perf-450W.txt) the page's flag caveat is read from, and both arms' completion notes. The two sittings are different days, lanes and cell sets and their rows must not be pooled. TWO omissions are named in the kit rather than implied by absence: each rung's run.log, and the frozen prompt's TEXT — sha-pinned on the page, in every summary.json and in the ladder's registration, but held pending a licensing call on the source, so a reader can verify a copy they already hold and cannot reconstruct it from the kit",
      "licence": "CC BY 4.0",
      "one_liner": "What a lower board power limit costs the two seats this workshop actually runs, measured on one 96 GB workstation card on 2026-08-25 at 600 W and 450 W with nothing else changed. On the dense arm — a 24B model at Q4, the one actually pinned to the ceiling — the cap costs 2.0% of throughput at concurrency 16 (94.6 → 92.7 tok/s per stream) and buys back 138 W (586.5 → 448.4 W busy-window), 11 °C (83 → 72) and a 28% jump in efficiency (0.161 → 0.206 tok/s per watt); time-to-first-token p50 moves 20.6 → 22.2 ms. On the MoE arm the cap clips only the top of the distribution — 16 of 386 busy-window samples reached 450 W across the uncapped legs, peaking at 461 W — and costs no measurable throughput (sustained decode 170.24 → 170.09 tok/s, −0.09%, inside the measured noise floor). The mechanism is checked rather than inferred: a cap is a clock cap, and the dense arm gave up 18.7% of its SM clock (2764 → 2246 MHz) for those two percent, because Q4 decoding is memory-bandwidth-bound. The page's second finding is about the instrument: across twelve rungs the card's power-cap FLAG discriminated correctly in both directions (119 of 275 in-window samples Active on the clamped rung at 2137–2400 MHz; 1 of 24 on the rung that only brushed the ceiling; zero on the other ten) while the cumulative counter that should total that time did not advance one microsecond through ~60 s of flagged clamping, resuming twenty minutes after the bench ended — so on this driver generation a zero flag is evidence when the flag is sampled continuously, as it was there at ~2 Hz — read once it is luck of the draw — and a zero counter is not evidence either way. An earlier internal write-up called both gauges blind; that was a local-time windowing error, found in this page's own audit and corrected with its date. A second sitting on 2026-08-27, 16:19–18:14Z, walks the same card's image lane down 600 → 500 → 450 W over 139 sealed cells per rung and finds the same law rather than a new one: the model that fills the card pays and the ones that idle under the cap do not — flux1-dev, pegged at 599.3 W uncapped, pays +10.2% at 500 W and +19.0% at 450; flux1-schnell +5.9/+12.1; the limner +3.7/+8.4; the two SDXL rows read as noise around zero — while the serving lane, probed at every rung with single warm calls against the production Ollama seats rather than the batched vLLM cell above, barely moves: gemma flat to a tenth of a percent (207.3 / 207.1 / 207.2 tok/s at 600 / 500 / 450 W) and mistral paying about 1.5% at 450 W (95.0 / 94.6 / 93.5), the same dense-lane bill the serving arm priced at 2.0%. The dial was ruled at 500 W on the day of that measurement. One environment finding is published rather than hidden: 88 cells with sealed 600 W times from 2026-08-23 re-ran ~6% slower today on the two lighter lanes (the pegged painter reproduced exactly), and four things changed between the sittings — including the card moving onto a sine-wave UPS — so it is an observation, not a controlled result. 4,044 scored requests, 0 failed, on the serving arm; 0 failures on the ladder; every rung's window, busy-window rule (utilisation ≥ 50%) and noise floor published with the rows"
    },
    {
      "exhibit": 28,
      "slug": "how-the-beat-lab-works",
      "title": "How the Beat Lab works",
      "url": "https://research.strata2signal.com/how-the-beat-lab-works/",
      "published": "2026-08-27",
      "data": null,
      "data_note": "no kit, by design — a field guide that opens up a shipped tool, not a bench with rows. Its one table of figures is a live measurement of the public site rather than an archived run: re-taken 2026-08-28 (09:05–09:10Z) from our own web server through https://studio.strata2signal.com/, the same way as the 2026-08-27 reading the page first published, and on the cards the genie moved onto that morning (ten drum voices, twenty pad sounds), each figure the median of three warm runs with the models already loaded. The rule is the whole method, so anybody can take the reading again against the same page and get their own numbers, plus their own connection's round trip",
      "licence": "CC BY 4.0",
      "one_liner": "How the Beat Lab works, from the browser outward: ten drum voices and six pads — each pad set to any of twenty sounds — synthesised on your own device from raw waveforms, with no drum recordings to download, one file of about 330 KB and zero third-party requests; a genie on hardware we own, a Mistral Small 3.2 doorman in front of a Gemma 4 writer, both open-weight, that builds a whole machine from a typed vibe in about 7 s or from a photograph in about 8 s, reviews the beat you built yourself in about 1.5 s, and speaks its first words 1.5 s in (under 0.1 s on a repeat); a share link that carries the beat in the part of a web address a browser never sends to a server; and a public board that holds no name, no account and no address."
    },
    {
      "exhibit": 27,
      "slug": "the-typist-and-the-developer",
      "title": "The Typist and the Developer",
      "url": "https://research.strata2signal.com/the-typist-and-the-developer/",
      "published": "2026-08-27",
      "data": [
        {
          "url": "https://research.strata2signal.com/the-typist-and-the-developer/data/prereg.md",
          "role": "The pre-registration, written 2026-08-26 ~13:4xZ before any measured call, with its one dated addendum: the declared window and the beside-traffic disclosure, the two stacks in their production shapes, the five rows and their sample counts, the four named self-refutation checks, and the publication plan. Every rule and UTC date is as registered — including the declared window that the addendum then corrects.",
          "media_type": "text/markdown",
          "bytes": 4064,
          "sha256": "ac2257656fa5bedeeb324766c02de672fc2d93c25e49e32de411246f1603feed"
        },
        {
          "url": "https://research.strata2signal.com/the-typist-and-the-developer/data/rows.jsonl",
          "role": "Every sample this bench paid, one JSON object per line — the 47 recorded rows plus the harness's own open and close lines: five solo token rows, five solo renders, five contended token rows with the six background renders that kept the card busy, five contended renders, the re-grab readings (the chat model's load time on the first call after a render burst) and the free-memory reads before and after. The first line is the provenance header added at publication; everything after it is as the harness wrote it. The excluded-failures list is empty because nothing failed.",
          "media_type": "application/x-ndjson",
          "bytes": 8203,
          "sha256": "d935c8ff1723cccf3fc242d4213352703e6c1f515e5b5f3bace7c5a701f3b33f"
        },
        {
          "url": "https://research.strata2signal.com/the-typist-and-the-developer/data/residency-klein-0825.json",
          "role": "The two-painter residency probe from the klein bake-off of 2026-08-25 (the same 96 GB card, one day before the contention rows): each entry is one render — its label, the painter's own execution seconds, and the card's free memory before and after. The alternation the piece re-quotes (32.7 ↔ 16.2 GiB free as the two paint sets evict each other; +2.9 s per swap) is read straight off these rows. Published byte-for-byte; it names no box.",
          "media_type": "application/json",
          "bytes": 628,
          "sha256": "d0196f98bbcf541d90968913a6e14f2e29698e3f99cf9938ac7d60d4472198bf"
        },
        {
          "url": "https://research.strata2signal.com/the-typist-and-the-developer/data/README.md",
          "role": "The kit's own front page: what each file holds, the licence, the full text of every sanitisation rule applied, and a plain statement of what these rows will and will not tell you.",
          "media_type": "text/markdown",
          "bytes": 4861,
          "sha256": "efc6b1ca41f28e809d774180ffde41ca8754ab171048e253393dea4a6f9c0463"
        },
        {
          "url": "https://research.strata2signal.com/the-typist-and-the-developer/data/provenance.json",
          "role": "Both published files with both sha256s — as sourced and as published — the named rules applied to each with the count of times each fired, and the run's window.",
          "media_type": "application/json",
          "bytes": 5155,
          "sha256": "eaf9cae4ba36a3e0803043320e3f027a98cfb3d32d963488005f90f6c4c52fe3"
        },
        {
          "url": "https://research.strata2signal.com/the-typist-and-the-developer/data/index.json",
          "role": "This directory, listed for machines: file names, sizes, sha256s, and what each one answers.",
          "media_type": "application/json",
          "bytes": 3598,
          "sha256": "deacf58dee47904a3065afa702f2fd3b5964835e5e32d7b46d966d6ae375530e"
        }
      ],
      "data_note": "the kit is browsable: `data/` serves its own index page listing every file below with the same bytes and sha256, and `data/index.json` is that listing in machine-readable form. Published from one run — `contention-0826`, 2026-08-26T18:18:02Z to 18:18:37Z, 35 seconds end to end, 47 recorded rows, 0 failures and an empty exclusion list. The rows are the raw file as the harness wrote it, with ONE line added first at publication naming the card class, the model, the graph, the steps and the canvas — the raw file named none of them. Every number the article states that did not come from this run is labelled a re-quote where it appears and carries its own provenance: part two's 208.0 tok/s, the voice trials' 89-to-41 render cut (re-quoted as a shape, never as a delta), the rig piece's under-a-second sketch, part one's reading/writing pair and July's fifty-four-second painter wake-up are all earlier published benches and are not rows in this kit. The two-painter eviction pair is the one exception — its reading ships here, as `residency-klein-0825.json`.",
      "licence": "CC BY 4.0",
      "one_liner": "What a language model and a picture model cost each other when they share one graphics card, measured on the live serving card the game actually runs on: the chat seat wrote at 207.7 tokens per second alone and 51.6 while a sketch developed, and a 16-step 512-pixel sketch took 1.54 seconds alone and 1.82 seconds beside a full token stream. The lopsidedness is the finding — the developer keeps 85 per cent of its solo speed, the typist 25 per cent, and the two together deliver about 110 per cent of one card's serial output. Free memory sat at 16.17 GiB before and after every burst, so residency contention — the fifty-four-second painter wake-up of the 24 GB era — never appeared for this pair; a different pair measured one day earlier still evicts each other at 2.9 seconds a swap."
    },
    {
      "exhibit": 26,
      "slug": "a-rig-your-friend-already-owns",
      "title": "A Rig Your Friend Already Owns",
      "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/",
      "published": "2026-08-26",
      "data": [
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/rematch-3090-0826-manifest.json",
          "role": "The consumer desktop's run (RTX 3090 24G, single GPU): 171 rows, one per render — the verbatim prompt, the seed, steps, sampler, scheduler, cfg, the bare checkpoint filename, width and height, the engine's own render_seconds, the paired baseline_seconds for the identical cell on the 96G-class card, the baseline_id it pairs to, the model's licence citation, the UTC instant, the arm, the axis, and the rig in its own `card` field.",
          "media_type": "application/json",
          "bytes": 363653,
          "sha256": "aa95171f9a14764d6a83f79f3611677b1f9e483f7f83fdf2425437f1d48316ca"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/rematch-rigB-0826-manifest.json",
          "role": "The gaming laptop's run (RTX 5090 Laptop GPU 24G): the same 171 cells in the same order, the same schema, its own `card` field, and the same 96G-class baselines in baseline_seconds — so the two rigs are comparable row for row.",
          "media_type": "application/json",
          "bytes": 362970,
          "sha256": "dfb67b0e7a7bd45d764f3a062549b2768c20ef32976c3f09d7147e3746dfbb2d"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/rematch-3090-0826.log",
          "role": "The desktop run's driver log as it printed, live: the seeded subject sample, the worklist, every warm-up announced, and every one of the 165 timed renders numbered [n/165] with its seconds and its ratio, closing on REMATCH COMPLETE. The artefact behind “zero failures” — a failure prints `!!` or `XX`, and neither string appears.",
          "media_type": "text/plain",
          "bytes": 28370,
          "sha256": "ef1013cd94d6bf3a65a7077d77d3ba92f3798c460dcdf5e19c00dd006581c2fc"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/rematch-rigB-0826.log",
          "role": "The laptop run's driver log, same shape, same 165 numbered lines, same closing REMATCH COMPLETE.",
          "media_type": "text/plain",
          "bytes": 28352,
          "sha256": "86fad12c9038448de4077f553cd0e1a5c6e0406e61f588d0e300299a4f2f984a"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/graph-flux2-klein-4b.json",
          "role": "The ComfyUI API-format workflow graph for FLUX.2 klein-4B, as this bench submitted it: the harness's own template with its `_`-prefixed note keys dropped and its {prompt} / {width} / {height} / {seed} tokens substituted by the same code path the run used, at the settings of the standard 768×768 subject cell. Its prompt and seed are a fixture — the real values of the published row named in _fixture.fixture_row_id — which any other row's prompt and seed substitute into.",
          "media_type": "application/json",
          "bytes": 3531,
          "sha256": "a3509ba5d34f4d57a3ae117c9eae089e14fc4ea4c57b2cc2861d454ba8256a0a"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/graph-ctrl-z-image-turbo.json",
          "role": "The ComfyUI API-format workflow graph for z-image-turbo, as this bench submitted it: the harness's own template with its `_`-prefixed note keys dropped and its {prompt} / {width} / {height} / {seed} tokens substituted by the same code path the run used, at the settings of the standard 768×768 subject cell. Its prompt and seed are a fixture — the real values of the published row named in _fixture.fixture_row_id — which any other row's prompt and seed substitute into.",
          "media_type": "application/json",
          "bytes": 3237,
          "sha256": "a5173248654f0489e6e67b6fdf3ed2e0b48b37bdcbb97a2c4b810bfaa47602c0"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/graph-krea2-turbo.json",
          "role": "The ComfyUI API-format workflow graph for krea2-turbo, as this bench submitted it: the harness's own template with its `_`-prefixed note keys dropped and its {prompt} / {width} / {height} / {seed} tokens substituted by the same code path the run used, at the settings of the standard 768×768 subject cell. Its prompt and seed are a fixture — the real values of the published row named in _fixture.fixture_row_id — which any other row's prompt and seed substitute into.",
          "media_type": "application/json",
          "bytes": 3099,
          "sha256": "98502b344f87b804ee40265cbec6600a55357bbaa0b6f4cf37341190a51d3fdc"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/graph-hidream-o1-dev.json",
          "role": "The ComfyUI API-format workflow graph for hidream-o1-dev, as this bench submitted it: the harness's own template with its `_`-prefixed note keys dropped and its {prompt} / {width} / {height} / {seed} tokens substituted by the same code path the run used, at the settings of the standard 768×768 subject cell. Its prompt and seed are a fixture — the real values of the published row named in _fixture.fixture_row_id — which any other row's prompt and seed substitute into.",
          "media_type": "application/json",
          "bytes": 3391,
          "sha256": "f7a7c2e85464a4bd3fae991bd5656a55a9fa8bb084f67d99c4416c6bbe1c90c9"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/graph-kandinsky5-lite.json",
          "role": "The ComfyUI API-format workflow graph for kandinsky5-lite, as this bench submitted it: the harness's own template with its `_`-prefixed note keys dropped and its {prompt} / {width} / {height} / {seed} tokens substituted by the same code path the run used, at the settings of the standard 768×768 subject cell. Its prompt and seed are a fixture — the real values of the published row named in _fixture.fixture_row_id — which any other row's prompt and seed substitute into.",
          "media_type": "application/json",
          "bytes": 3313,
          "sha256": "0b7183f539f9eb7d078c9de15e50636e486928f41d1397ec242b7838d54351d2"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/graph-z-image-base.json",
          "role": "The ComfyUI API-format workflow graph for z-image-base, as this bench submitted it: the harness's own template with its `_`-prefixed note keys dropped and its {prompt} / {width} / {height} / {seed} tokens substituted by the same code path the run used, at the settings of the standard 768×768 subject cell. Its prompt and seed are a fixture — the real values of the published row named in _fixture.fixture_row_id — which any other row's prompt and seed substitute into.",
          "media_type": "application/json",
          "bytes": 3252,
          "sha256": "4ae2dfb2238f00261c2b65c90026557356eab33a7fa182e3ff6616c48e0512a4"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/provenance.json",
          "role": "every published file with both sha256s, the named rule applied to it, the two runs' windows and counts, and the graph templates' hashes at pack time",
          "media_type": "application/json",
          "bytes": 14660,
          "sha256": "3861794d4cab40ca9fb98647bc46e4d8f8c4216de89fe8be2d1339d38f280f16"
        },
        {
          "url": "https://research.strata2signal.com/a-rig-your-friend-already-owns/data/README.md",
          "role": "the kit's own README: licence, what each file holds, the counting rules a reader needs before the numbers, the id-prefix wart, the licence table, the graph-reproduction steps, and the receipts",
          "media_type": "text/markdown",
          "bytes": 18163,
          "sha256": "3d7b56609328509aad7b5810f957e2bed8b668febdca1b9a1d922c60b5205097"
        }
      ],
      "data_note": "the kit is browsable: `data/` serves its own index page listing every file below with the same bytes and sha256, and `data/index.json` is that listing in machine-readable form. Published from two runs — `rematch-3090-0826` (the consumer desktop) and `rematch-rigB-0826` (the gaming laptop) — 171 rows each, 165 timed renders plus six untimed warm-ups, 0 failures on both. The workstation-card baseline is NOT a run in this kit: every row carries its own `baseline_seconds` for the identical cell, archived 2026-08-23, so each pairing is checkable inside a single row. The 342 rendered images are deliberately NOT published — a heavy download for a page about how long they took, and the manifests and logs are the receipts; `images_note` in `data/index.json` says so, and each row's `file` field is a name rather than a link. Sanitisation was ONE named rule, `server-card-model-to-vram-class`: the baseline card's model number left the manifests as a field name (`speed_ratio_vs_baseline`) and the logs as a per-render annotation (`vs 96G-class`), and every timing value is unchanged from the originals. The counting rules the README and this index both print — medians half-up, multipliers as the ratio of the two PRINTED medians — are what make the page's table reconcile by hand.",
      "licence": "CC BY 4.0",
      "one_liner": "The game's own art re-rendered on hardware a normal person owns — same prompts, same seeds, the same six checkpoint files — on a 2021-vintage consumer RTX 3090 and an RTX 5090 Laptop, each of 330 timed renders paired cell-for-cell against the 96 GB-class card's own archived output, with two of the three runs measured in the same physical computer with only the graphics card changed between the dates. The sketch the game actually waits on lands in 0.88 s at its live 512 size on the 3090 (0.26 s on the big card) and under three seconds at 1024; across the six models the big card is 3.4–4.9× the desktop and 2.9–4.0× the laptop. Speed class survives the hardware drop on cards of 24 GB and up: five of six models keep their exact speed rank on all three rigs. The sixth is the finding the page distrusts and publishes anyway — hidream-o1-dev's headline 1.4× on the laptop is an artifact of the resolution mix sitting on a baseline that is nearly flat across a 4× pixel increase, so its two multiplier cells are withheld from the table and the open question is printed instead of a conclusion. 342 renders, 0 failures, one post-hoc reporting rule disclosed as exactly that."
    },
    {
      "exhibit": 25,
      "slug": "how-a-vision-model-sees",
      "title": "How a Vision Model Sees: Eyes for a Machine That Reads",
      "url": "https://research.strata2signal.com/how-a-vision-model-sees/",
      "published": "2026-08-26",
      "data": [
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/prereg.md",
          "role": "The pre-registration, written 2026-08-24 17:10Z before any image was scored, with its seven addenda dated to the hour: the two questions, the photo set and its licence rule, the seats, the ten metrics, every gate threshold with its reasoning, the six self-refutation triggers, the nonce law, and the publication plan. Every threshold, rule and UTC date is as registered.",
          "media_type": "text/markdown",
          "bytes": 61610,
          "sha256": "8a753b385e0b3bdba61eac5c37e5a0e51ded7bbc8958835a6817ac9b3efadbd9"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/manifest.json",
          "role": "The merged run's receipt: run id and its rule, the prompt shas and the drift check, the request options exactly as the product sends them, the per-seat order nonces, the card's memory at every open / first-load / close snapshot with its settle wait, the card floor, the eight seat blocks, and the crash seam (`resumed_from_crash`) that names the failing row, the fix, and the two run ids.",
          "media_type": "application/json",
          "bytes": 36585,
          "sha256": "6e4db002b4bcb9207b592b891134165b6ca79fa1ed446174a2b9289dd9b76ffc"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/manifest-resumed.json",
          "role": "The resumed run's OWN manifest (`full-0825b`, the six seats that re-ran after the crash), so the seam recorded in manifest.json has a receipt on both sides of it.",
          "media_type": "application/json",
          "bytes": 27464,
          "sha256": "95580a91c0351b239a3d93dcf4e2224ca9c57d599558d3cb2a9ba771a30a9803"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/rows.jsonl",
          "role": "Every call this bench paid, one JSON object per line — 3,344 of them across eight seats: the seat, the photograph's prepared sha256, the raw reply verbatim, the parsed verdict or the nine described fields, the runtime's own timings, and the resident census before and after. The raw replies ARE the receipts; nothing is summarised away.",
          "media_type": "application/x-ndjson",
          "bytes": 3361807,
          "sha256": "3763e30c7f8a72c67314800874cb9fff5b298e2c22e2be4ece2e091dd82639bf"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/photo-ledger.md",
          "role": "Every photograph, with its Commons title, its page URL, the machine licence token and short name as read from the host's own field on the stated date, the uploader credit verbatim, source and fed dimensions, and both sha256s (as fetched, and as prepared for the seats). 94 CC0 and 7 public-domain; the receipt is the pointer, not the bytes.",
          "media_type": "text/markdown",
          "bytes": 48618,
          "sha256": "66c986a7a4bb369d4a4b962b00254afb0a337ec4d85d6d76229db5caf2852700"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/human-key.json",
          "role": "The answer sheet (key v3, 36 rows), frozen and hashed into the run id before any seat was called. Its `author` field is published AS RECORDED: the key was written by an agent lane from the prepared photographs, not by a person — the page says so, and rewriting the field would falsify the receipt.",
          "media_type": "application/json",
          "bytes": 37500,
          "sha256": "b104ebfaa07cf683d246225ddb1d4240eba63faad00a5c1c96dec4b61834a356"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/scored-full.json",
          "role": "The full scoring over all eight seats, as a structure: the manifest, every gate result with its count and its verbatim rule, every metric, and the six self-refutation checks computed rather than asserted.",
          "media_type": "application/json",
          "bytes": 387018,
          "sha256": "7f426ffb8072a5da1b08cbe1266e484f3f4a29cadcc44b2015a74cbcf44628ac"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/scored-full.txt",
          "role": "The same full scoring as the console prints it — the table a reader can read straight down. Captured verbatim from the files published here.",
          "media_type": "text/plain",
          "bytes": 50623,
          "sha256": "1731184f2c1c07f4cc8a3b75413bd00276a82e6ab179c3dc89f978e129620ca9"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/scored-intro.json",
          "role": "The `--subset intro` view: the pre-registered twelve photographs on the two intro seats, the demonstration slice tonight's piece draws from. No threshold table — a 12-photo subset cannot pass or fail a 60-photo gate, and the scorer refuses to pretend otherwise.",
          "media_type": "application/json",
          "bytes": 227944,
          "sha256": "01b974f51e480594cd6f1b14cd97934082ab4741f0b61097f72fbab0dee2b4d4"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/scored-intro.txt",
          "role": "The intro subset as the console prints it. Captured verbatim from the files published here.",
          "media_type": "text/plain",
          "bytes": 12553,
          "sha256": "de9698f52b49ae622dfa72aba91adc1c3f5bbd2f4d46c0443db2c0e7a6fb82df"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/m10-text.json",
          "role": "M10-TEXT — the pre-registered likeness panel with its render leg NOT RUN (there was no painter on the box that ran this bench). Three judge families over the composed phrase, the 0–4 anchors verbatim, the blindness audit, and the per-cell scores.",
          "media_type": "application/json",
          "bytes": 184760,
          "sha256": "4081afa48602a06d9fb2c4eafc1678581dbd6721f144afed858c205c167cd5e7"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/provenance.json",
          "role": "every published file with both sha256s, the named sanitisation rules applied to it, the run and key ids, and the harness hashes at pack time",
          "media_type": "application/json",
          "bytes": 11706,
          "sha256": "59434fcf4cfbaba12bb055a693e9b035b59da85b7600ceb992d7213d6110eae2"
        },
        {
          "url": "https://research.strata2signal.com/how-a-vision-model-sees/data/README.md",
          "role": "the kit's own README: licence, what each file holds, the sanitisation rules, the three caveats a careful reader needs first, and the receipts",
          "media_type": "text/markdown",
          "bytes": 12445,
          "sha256": "b33515ba3fb5495ec13b477fea4015b5404d274036bca0295bd5cd9dd2a2dfdb"
        }
      ],
      "data_note": "the kit is browsable: `data/` serves its own index page listing every file below with the same bytes and sha256, and `data/index.json` is that listing in machine-readable form. Published from run full-0825 (resumed as full-0825b after a recorded mid-run crash), answer-sheet key v3. Every file is the sanitised copy — the rules applied to each are named in its `stripped` list in `data/index.json`, and the answer sheet `human-key.json` is published byte for byte, so its sha256 is the one frozen inside the run id.",
      "licence": "CC BY 4.0",
      "one_liner": "What a photograph becomes on the way into a language model — the token cost, measured, and what two small seats actually saw."
    },
    {
      "exhibit": 24,
      "slug": "fifty-seven-milliseconds",
      "title": "Fifty-Seven Milliseconds: What a Diffusion Step Actually Buys",
      "url": "https://research.strata2signal.com/fifty-seven-milliseconds/",
      "published": "2026-08-25",
      "data": [
        {
          "url": "https://research.strata2signal.com/diffusion/step-ladder-512/receipts.json",
          "role": "the first ladder's own receipts — all 42 renders at 512² with their verbatim prompts, painter, sampler, size, seed, step count and per-image seconds; the manifest findings one to three are computed from",
          "media_type": "application/json"
        },
        {
          "url": "https://research.strata2signal.com/diffusion/steps-limner-flux2-klein-4b/receipts.json",
          "role": "the second ladder's receipts — six of the game's real sketch prompts at 768² across six step counts on the season's new painters, one of them distilled to finish in four; the manifest finding four's tier move is computed from",
          "media_type": "application/json"
        }
      ],
      "data_note": "no kit of its OWN, by design — the receipts this piece argues from are the two step-ladder galleries', published in the diffusion wing rather than under this slug, because the galleries host the archives and the piece reads them; `owns_kit()` therefore reads False, the exhibit-fourteen shape. That the step count is the only field moving is a verified property of each manifest (exactly one distinct value for every other field) rather than a promise, and the pixel-difference rule is printed in full on the page in three lines of Python, so the second half of the arithmetic reproduces from the archives' original JPEGs without a kit from us",
      "licence": "CC BY 4.0",
      "one_liner": "What a diffusion step actually buys, measured twice a month apart on two painters that could not be more different. A step is a fixed-price good: the shipped painter's per-rung medians fit seconds = 0.238 + 0.0569 × steps at 512² (56.9 ms a step, and the slope does not budge whether the fit takes the medians, all forty-two renders individually or the post-exclusion set — only the intercept moves, 0.216 to 0.257), while a painter distilled to finish in four costs 0.25 + 0.073 × steps at 768². The dial then has two regimes: below a painter's finishing point the picture converges rung by rung — six of six subjects, monotonically, on the distilled ladder — and above it the rungs are siblings rather than stages, with zero of six subjects approaching their own 40-step image monotonically on the first ladder. The one misbehaving rung (30 steps, slower than 32) is shown rather than smoothed, explained by the session it ran in, and the fit published both ways; and the piece ends by moving one of our own settings rather than only arguing about it — the in-play sketch tier goes to 768² at 12 steps, 1.51 s against 3.48 s at the old 512²-at-32-steps setting on the same card."
    },
    {
      "exhibit": 23,
      "slug": "the-dog-the-dice-and-the-painter",
      "title": "The Dog, the Dice, and the Painter: How a Small World Draws Itself",
      "url": "https://research.strata2signal.com/the-dog-the-dice-and-the-painter/",
      "published": "2026-08-25",
      "data": null,
      "data_note": "no kit, by design — a walk through a shipped engine, not a bench with rows. Every figure it prints carries the rule that re-derives it from an artifact that already ships: the census JSON grouped by checkpoint, file counts taken at a named engine version, and the FNV-1a/xorshift32 goldens both implementations pin in their own test suites, which anyone can reproduce in about ten lines without anything from us",
      "licence": "CC BY 4.0",
      "one_liner": "How a small world draws itself, walked one station at a time: a 16,705-line campaign file that says what to paint and is forbidden by an architecture test from saying how; an append-only event log whose 128-bit world seed is minted once at first boot and never re-minted, so every draw replays to the same number forever; a sketching service handed an enumerated list of public facts, which illustrates what the log already decided rather than imagining a scene; and a photograph-to-creature pipeline that moderates first and fails closed — a four-value verdict whose refusal has no wire token and so cannot be forged, a nine-field description schema with nowhere to ask where the animal was, and image bytes that are never written down — arriving at the same pet shape the world's hand-authored dog already had. 2,828 in-play renders counted by checkpoint (81.2% on one painter); 419 of 429 architecture-test files; 86 parity test files."
    },
    {
      "exhibit": 22,
      "slug": "diffusion/new-arrivals",
      "title": "The new-arrivals diffusion bench",
      "url": "https://research.strata2signal.com/diffusion/new-arrivals/",
      "published": "2026-08-24",
      "data": null,
      "data_note": "no kit, by design — a gallery bench whose receipts ARE the wing: seventy-five public galleries, the licence ledger with its first-hand dated readings, and the census JSON grouped by checkpoint all ship beside the 3,588 renders at the exhibit's own path; there are no bench rows apart from the pictures themselves, and the drop counts are stated on the page rather than kept"
    },
    {
      "exhibit": 21,
      "slug": "where-new-knowledge-comes-from",
      "title": "It Was Already There: Where New Knowledge Actually Comes From",
      "url": "https://research.strata2signal.com/where-new-knowledge-comes-from/",
      "published": "2026-08-24",
      "data": null,
      "data_note": "no kit, by design — a field guide reading the published record, not a bench with rows; every figure on the page is quoted from a named public source and cited at the point of use, and the page says so about itself in its “How to check our work” section",
      "licence": "CC BY 4.0",
      "one_liner": "Where new knowledge actually comes from: two engines that were there all along — the archive nobody had read (co-evolution fingerprints; a rule about which molecular shapes kill bacteria, latent in 2,335 labelled dish results) and the rooms the rules imply (a planet's position in 1631, an anti-electron, a shoulder hit on the fifth line) — and the check outside the model, never the proposal, is what makes either one knowledge: the A-Lab proposed in 17 days, and the published record took two years to correct 41 of 58 to 36 of 57."
    },
    {
      "exhibit": 20,
      "slug": "compression-is-the-objective",
      "title": "It's Not a Metaphor: A Language Model Is a Compression of Its Corpus",
      "url": "https://research.strata2signal.com/compression-is-the-objective/",
      "published": "2026-08-23",
      "data": null,
      "data_note": "no kit, by design — a field-guide explainer, not a bench with rows; its two house receipts name the measurement rather than the vendor by the piece's own rule, and the third is an external published result cited at point of use",
      "licence": "CC BY 4.0",
      "one_liner": "A language model is a compression of its corpus — the training objective, not a figure of speech: Kepler and Kolmogorov, then bias, hallucination, scale and the frozen moment as consequences of one fact, with a replay probe whose hidden variable was what ran before, a 387-of-387 code-fence habit against a sibling's zero, and the Hutter accounting that charges a model for its own weights."
    },
    {
      "exhibit": 19,
      "slug": "the-free-speed-wasnt-free",
      "title": "The free speed wasn't free — the runtime dividend, traced to one commit",
      "url": "https://research.strata2signal.com/the-free-speed-wasnt-free/",
      "published": "2026-08-22",
      "data": "https://research.strata2signal.com/the-free-speed-wasnt-free/data/index.json",
      "data_note": "sixteen files: the sealed-request toggle's forensic and raw probe rows; the version pair's pre-registration with pinned sha, runner, raw rows with reply hashes, and run note carrying its own retraction; the toll probes with every sample; the commit message fetched raw plus the raw API response; the first-party vocabulary receipt; the qwen penalty probe; and three recon reports with their UNVERIFIED lists and redaction disclosures intact"
    },
    {
      "exhibit": 18,
      "slug": "the-compressed-photograph",
      "title": "The compressed photograph — what those Q4_K_M tags actually mean",
      "url": "https://research.strata2signal.com/the-compressed-photograph/",
      "published": "2026-08-21",
      "data": "https://research.strata2signal.com/the-compressed-photograph/data/index.json",
      "data_note": "nine files: all four builds' tensor inventories (quant-recipes.json), the 2026-08-19 pair bench and the 2026-08-20 four-rung ladder with verbatim replies and per-load card receipts (robber-runs.json, ladder-runs.json), the refused first attempt banked with its posture ruling (ladder-runs-attempt1.json), the availability record with reconciliation notes (LADDER-AVAILABILITY.md), and all three pre-registrations with their pin index (ROBBER-PREREG.md, ROBBER-LADDER-PREREG.md, ROBBER-LADDER2-PREREG.md, PREREG-INDEX.txt)",
      "licence": "CC BY 4.0",
      "one_liner": "Quantization in plain english, measured on our own files: the tag is a recipe's center of gravity, not its contents — and down a four-build ladder of one 11.9B, the phrasing of the question moved answers while the compression never did."
    },
    {
      "exhibit": 17,
      "slug": "three-at-the-table",
      "title": "Everyone on the payroll, three at the table — dense models, mixtures of experts, and active weights",
      "url": "https://research.strata2signal.com/three-at-the-table/",
      "published": "2026-08-20",
      "data": "https://research.strata2signal.com/three-at-the-table/data/index.json",
      "data_note": "two files: active-weights-derivation.md (every weight bucket and byte for both mixtures, computed from the model files' own tensor shapes, with the bandwidth arithmetic) and robber-asks.json (the pre-registered n=5 robber mini-bench — both arms' verbatim replies, byte-identical across repeats at temperature 0, with the runtime's own counters)",
      "licence": "CC BY 4.0",
      "one_liner": "Dense vs mixture-of-experts in plain English: the payroll and the meeting. A 25.8B MoE out-writes its own 11.9B dense sibling (124.975 vs 95.195 tok/s at the 32k tier) while holding more memory; active weights computed from tensor shapes, not transcribed from labels; and the robber asked both — one muddles, deterministically."
    },
    {
      "exhibit": 16,
      "slug": "reading-is-fast",
      "title": "Reading is fast, writing is slow — why AI answers arrive word by word",
      "url": "https://research.strata2signal.com/reading-is-fast/",
      "published": "2026-08-19",
      "data": "https://research.strata2signal.com/reading-is-fast/data/index.json",
      "data_note": "two files: runs-prefill-decode.json (the n=10-per-leg prefill/decode timing rows behind every speed figure on the page, from the runtime's own nanosecond counters) and drafthead-acceptance-receipt.md (the draft-head acceptance receipt — rates and per-slot acceptance read from the serving logs — that resolved the measurement that argued back)",
      "licence": "CC BY 4.0",
      "one_liner": "Why AI swallows your essay at once and answers word by word: prefill vs decode in plain english, measured at ~5,400 tokens/s reading against ~135 writing on the same card in the same second — and the arithmetic that caught our own card lawfully cheating."
    },
    {
      "exhibit": 15,
      "slug": "three-librarians",
      "title": "Three librarians and a careful reader — how RuleSage finds the right page",
      "url": "https://research.strata2signal.com/three-librarians/",
      "published": "2026-08-18",
      "data": null,
      "data_note": "no kit, by design — a field-guide explainer, not a bench with rows; its receipts are live: every figure comes from the shipped configuration, and the closing specimen is ruling 4615 on rulesage-live, whose public wrench shows the fused score, arm count, ceiling with its formula, and cited pages",
      "licence": "CC BY 4.0",
      "one_liner": "The journey from a typed rules question to a cited page, in plain english: exact-words search, meaning search, a 2-or-3-arm RRF election at k=60, a 22 MB cross-encoder, lost-in-the-middle order — and a live ruling you can open."
    },
    {
      "exhibit": 14,
      "slug": "the-new-kid",
      "title": "qwen3.8:27b across four house benches: one seat filled, one floor missed",
      "url": "https://research.strata2signal.com/the-new-kid/",
      "published": "2026-08-17",
      "data": [
        {
          "url": "https://research.strata2signal.com/the-open-call/data/index.json",
          "role": "exhibit eleven's kit, which carries this model's narrator round as its twenty-first arm: the scored cells, each judge's note verbatim, the leave-one-family-out table and the judging bill re-derived two ways",
          "media_type": "application/json"
        },
        {
          "url": "https://research.strata2signal.com/llms-txt/data/index.json",
          "role": "exhibit thirteen's kit, which carries this model's site-reading row: its per-condition cells, the sealed golden set and the counting rules",
          "media_type": "application/json"
        }
      ],
      "data_note": "POINTERS, not a kit of this exhibit's own. Two of its four instruments are siblings' benches and publish this model's rows there, so those rows are linked rather than copied; each row names the sibling's manifest because neither data directory has an autoindex. No size or sha256 rides with these two, deliberately: the sibling exhibit owns those files and stamps its own receipts on them, and a hash copied here would go stale behind an edit nobody would think to check this row against. The other two instruments — the classifier bench and the judge-seat addendum — publish no files at all: their fixtures hold real slur specimens and lane data respectively, so the exhibit cites them by run id, pre-registration sha and fixture sha, all printed on the page, and says on its own face that they are not independently checkable. Whether either publishes is an open operator ruling as of this date",
      "licence": "CC BY 4.0",
      "one_liner": "qwen3.8:27b landed the day after it shipped and sat four exams we already had: a five-gate PASS that filled a live screening seat, a mid-table TIED narrator score, a FAIL against the judge seat's preservation floor, and a site-reading row the bench registers no verdict word for."
    },
    {
      "exhibit": 13,
      "slug": "llms-txt",
      "title": "The map nobody picks up — and whether it helps when it arrives",
      "url": "https://research.strata2signal.com/llms-txt/",
      "published": "2026-08-17",
      "data": "https://research.strata2signal.com/llms-txt/data/index.json",
      "data_note": "the kit's own manifest, listing all 25 files with their sizes, sha256s and the redactions applied to each. The directory has no autoindex, so this points at the manifest rather than at the folder",
      "licence": "CC BY 4.0",
      "one_liner": "No AI crawler asked us for ours in thirty days of logs. We measured whether it helps when it arrives anyway: eight arms, three token-matched conditions, a sealed question set."
    },
    {
      "exhibit": 11,
      "slug": "the-open-call",
      "title": "The open call — a kid, an elder, and a tired parent walk into the cove",
      "url": "https://research.strata2signal.com/the-open-call/",
      "published": "2026-08-15",
      "data": "https://research.strata2signal.com/the-open-call/data/index.json",
      "data_note": "the kit's own manifest, which lists all nine files with their sizes and sha256s; the eight it attests are also stamped on the exhibit page itself, which is the root of trust. The directory has no autoindex, so this points at the manifest rather than at the folder",
      "licence": "CC BY 4.0",
      "one_liner": "Twenty language models were handed the same frozen moment of a small fishing town and asked to speak as its people, then read blind by seven judges from six rival families."
    },
    {
      "exhibit": 12,
      "slug": "where-your-question-goes",
      "title": "Where your question goes — a plain-english walk",
      "url": "https://research.strata2signal.com/where-your-question-goes/",
      "published": "2026-08-15",
      "data": null,
      "one_liner": "The privacy explainer — the estate's two published promises unpacked in plain words for a non-technical reader."
    },
    {
      "exhibit": 10,
      "slug": "how-we-work",
      "title": "How we work — who writes this, and how",
      "url": "https://research.strata2signal.com/how-we-work/",
      "published": "2026-08-14",
      "data": null,
      "data_note": "no kit, and none is pending — this is the notes page describing the workflow itself, not a bench with rows; its receipts are the linked exhibits",
      "one_liner": "The honest answer to the fair question: the workflow that makes every exhibit — humans and agents, named plainly, with the receipts."
    },
    {
      "exhibit": 9,
      "slug": "the-move",
      "title": "The move — the same yardstick, before and after",
      "url": "https://research.strata2signal.com/the-move/",
      "published": "2026-08-14",
      "data": [
        {
          "url": "https://research.strata2signal.com/the-move/data/PREREG-MOVE.md",
          "role": "the pre-registration: what would be measured, on which eras, and the one window amendment, disclosed and dated",
          "media_type": "text/markdown",
          "bytes": 4657,
          "sha256": "8d198b31bb8a24b9cd7bf44a5d2e29928f843348313e28a1f8b80851b1f4f177"
        },
        {
          "url": "https://research.strata2signal.com/the-move/data/PREREG-INDEX.md",
          "role": "the sha index the pre-registration was pinned in before any scored call",
          "media_type": "text/markdown",
          "bytes": 989,
          "sha256": "33d7388ae92600fb807e11c8e1f00cacbe5185b3272d19538d7666209f71fb2d"
        },
        {
          "url": "https://research.strata2signal.com/the-move/data/c6-vps-rerun.json",
          "role": "the serving re-run's full per-ask records on the new host",
          "media_type": "application/json",
          "bytes": 458446,
          "sha256": "9df02d685ebd2b76cd9fbc23ab674f0af92d26adac8e861f5ea8f7aff91ee13f"
        },
        {
          "url": "https://research.strata2signal.com/the-move/data/era-comparison.json",
          "role": "the era comparison — the same frozen bench measured before and after the move",
          "media_type": "application/json",
          "bytes": 1974,
          "sha256": "93690ffa5a527020c568c06b625b11e602d790eacd222718ba649a0566eaa1e2"
        },
        {
          "url": "https://research.strata2signal.com/the-move/data/receipts-era-stats.json",
          "role": "the app's own timing receipts, per era, as statistics",
          "media_type": "application/json",
          "bytes": 1069,
          "sha256": "a52f98029771a9a5cce41509d2a63acc46abf9a0d7b9c3efe97d58b87f4d048b"
        },
        {
          "url": "https://research.strata2signal.com/the-move/data/ruling-debug-timings-asof-20260813.csv",
          "role": "the registered extract, sha-pinned in the registration index before the run",
          "media_type": "text/csv",
          "bytes": 198987,
          "sha256": "fbaf744b92de119abce2b07e003b9b818cae22231256b9b451a2c692ed8c7c9d"
        },
        {
          "url": "https://research.strata2signal.com/the-move/data/ruling-debug-timings-authoring.csv",
          "role": "the authoring-time extract the published statistics read",
          "media_type": "text/csv",
          "bytes": 192343,
          "sha256": "74dcaf92300d9cc0fbc3fe91a3444937a679e9c5d840c5be684ef9ea44b00ceb"
        }
      ],
      "data_note": "listed file-by-file: this kit has no manifest of its own, and the directory itself is not browsable (no autoindex), so a URL ending in `data/` 404s. Each file below is addressed directly.",
      "licence": "CC BY 4.0",
      "one_liner": "Our apps left the gaming laptop for a real server — the same frozen bench measured both eras, honestly."
    },
    {
      "exhibit": 6,
      "slug": "chair-trials",
      "title": "The chair trials — five fresh exams for the thirty-billion class",
      "url": "https://research.strata2signal.com/chair-trials/",
      "published": "2026-08-13",
      "data": [
        {
          "url": "https://research.strata2signal.com/chair-trials/data/roster.json",
          "role": "the eleven arms and what each costs to keep resident",
          "media_type": "application/json",
          "bytes": 22142,
          "sha256": "a37ad080f3a7186f5ec0c758acc4af1dcc6a65775f9223dd65f08f381997fdab"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/c1-judge.json",
          "role": "chair one, the judge — ten rows against two floors, with intervals, the fence probes and the markdown-fence caveat",
          "media_type": "application/json",
          "bytes": 69197,
          "sha256": "7bbba26d0581ca241699692759fcc32e641eee952e8fc1932d24de8cd1ebdcc0"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/c2-assistant.json",
          "role": "chair two, the assistant — twenty code-checked items, per-item matrix across all eight posture rows",
          "media_type": "application/json",
          "bytes": 358812,
          "sha256": "0ffee8f453cfbc838cebdc51e05f2ef38a22b24b14e003ae3a41ca96db729e4d"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/c3-toolbench.json",
          "role": "chair three, the toolbench — nineteen tasks with every call the models made, and the pre-probe that gated the leg",
          "media_type": "application/json",
          "bytes": 570948,
          "sha256": "75098737f370310b288f9bdcfba46e5120fd8ce0fcfb31caa306f2f2ef1ae847"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/c4-narrator.json",
          "role": "chair four, the narrator — 1 080 blind pairwise comparisons, the distinctness round and the canon adjudication",
          "media_type": "application/json",
          "bytes": 131602,
          "sha256": "8de910fb384032990e446f008b69045d641d7f120da664b4204fe8e76cd9460e"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/c5-stopwatch.json",
          "role": "chair five, the stopwatch — three prompt tiers with every raw repeat, both anomalies counted, the drafter byte-compare",
          "media_type": "application/json",
          "bytes": 231155,
          "sha256": "3a9d8e5a4585f1abc5d5a276471aac1b9d3b719c4e525e02fef12ea419fa3c79"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/c6-serving.json",
          "role": "the serving probes — twenty-six real asks against the live app's production path, with the quiet baselines beside them",
          "media_type": "application/json",
          "bytes": 231039,
          "sha256": "41183bd4fea2b362a67e30862c957c83403496da1b8f3cad12056226050805d8"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/index.json",
          "role": "the kit's own listing: files, sizes, sha256s",
          "media_type": "application/json",
          "bytes": 9480,
          "sha256": "bbf75bab3618ec843f9393f4cc8fc5307122bdb554c4301f91ed6a89ae51ef1b"
        },
        {
          "url": "https://research.strata2signal.com/chair-trials/data/README.md",
          "role": "licence, receipts, the contamination caveat, and how to read a number here",
          "media_type": "text/markdown",
          "bytes": 10902,
          "sha256": "c9490ec1f641ebff33b3c0598f082ddc4cd1e6f6849c34251803a0c807e7edc1"
        }
      ],
      "data_note": "the first exhibit published whole under the machine-readable law: every table's rows, every runnable set, the counting rules and the pre-registration hashes, all under data/ — nothing on the page is rows-on-request except the full tool transcripts, and the kit says why",
      "one_liner": "Nine models, eleven arms, five new exams in one day on a box that never stopped serving: a judge trial with floors, twenty code-checked assistant tasks, nineteen real tool tasks, a blind pairwise narrator round, a stopwatch — and twenty-six real asks measuring what the visitors paid for all of it."
    },
    {
      "exhibit": 7,
      "slug": "cove-voice-head-to-head",
      "title": "The narrator's chair, refused",
      "url": "https://research.strata2signal.com/cove-voice-head-to-head/",
      "published": "2026-08-13",
      "data": [
        {
          "url": "https://research.strata2signal.com/cove-voice-head-to-head/data/rows.json",
          "role": "one row per judged sample and arm — 13 prompts across 3 models, 39 rows and 117 scored cells: the replies as generated, per-sample timings with the contention flag and its re-run, all three blind judges' sheets, the un-blinded canon findings, the prompts with their canon anchors, and the counting rules",
          "media_type": "application/json",
          "bytes": 181616,
          "sha256": "1b818f416dc702e42ebfea2920142dc3590a0fdea7b438926b9e16407992a539"
        },
        {
          "url": "https://research.strata2signal.com/cove-voice-head-to-head/data/gate.json",
          "role": "the pre-registered gate — five floors with their registered text verbatim, the verdict each returned and the evidence behind it, and the no-threshold-table rule the REJECTED verdict triggers",
          "media_type": "application/json",
          "bytes": 5837,
          "sha256": "735d0c64cad4fba11246798101f8fde4d706eb43a7d757c0168f028736ed2557"
        },
        {
          "url": "https://research.strata2signal.com/cove-voice-head-to-head/data/index.json",
          "role": "the kit's own listing: files, sizes, sha256s",
          "media_type": "application/json",
          "bytes": 2822,
          "sha256": "3f9ce3a954be2835ed16e8a3dc321648e1c1b48637d1d4181aea1911eaec06a0"
        },
        {
          "url": "https://research.strata2signal.com/cove-voice-head-to-head/data/README.md",
          "role": "licence, receipts, the contamination caveat, and how to reuse these rows",
          "media_type": "text/markdown",
          "bytes": 3964,
          "sha256": "552a22f1de797341c5f9f4e5d35e3a8a9a0c2d86795480c1dbdf000a4bfff696"
        }
      ],
      "data_note": "REJECTED on the canon and latency floors — a refused candidate gets no threshold table, so no figure in this kit is published as an adoption target or carries forward as a bar to clear; its 0-10 scores share no ruler with the voice-trials exhibit, and no delta between the two is computed anywhere",
      "one_liner": "The fresh class's best voice, brought home to the cove's actual narrator chair: 13 real prompts from one night of live play, 3 arms, 3 blind judges, and a five-floor gate registered before the first call — refused on canon and on latency."
    },
    {
      "exhibit": 8,
      "slug": "outside-judges",
      "title": "The outside judges — four rival labs re-check our work",
      "url": "https://research.strata2signal.com/outside-judges/",
      "published": "2026-08-13",
      "data": [
        {
          "url": "https://research.strata2signal.com/outside-judges/data/audition.json",
          "role": "the audition: the whole registered judging queue in queue order — the four candidates that auditioned with their raw first responses verbatim, and the six that never did with the rule that stopped each one — plus the seat map and every round's carriage recomputed from disk",
          "media_type": "application/json",
          "bytes": 21993,
          "sha256": "ef3f61c571e90e993b630819dc597d2cb25466ef7ec57ba514883b11afb19033"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/panel-vendors.json",
          "role": "the five-vendor splits: every sealed round of the chair trials and the voice addendum scored once per vendor, plus the combined view and the original four-seat panel verbatim and unchanged",
          "media_type": "application/json",
          "bytes": 883870,
          "sha256": "15b9c82673649e472dfc36fdc2b373d031ca5bc7c76ef6c022a698da88e5c683"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/agreement.json",
          "role": "the agreement matrix: all ten vendor pairs, per round, with each pair's mean and maximum per-arm difference and its rank correlation, and every arm's spread across the five vendors — disagreements kept, not averaged away",
          "media_type": "application/json",
          "bytes": 41879,
          "sha256": "10d84b43ba0dff8ce2bf1363f05bd0b6ad46ee1140e8602522d7e883ecb677c8"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/gauntlet.json",
          "role": "three candidates on the frozen seat instrument: every case, every repeat, the kill-class breakdown, both protocol classes, and the instrument's whole prior history recounted so the word FIRST is a count rather than a memory",
          "media_type": "application/json",
          "bytes": 60769,
          "sha256": "c7504dbaea549a95eb014934175501c094b4e53471a3980d4e9f4bf5b0636953"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/bill.json",
          "role": "the bill: every duty on both sides of the table with the endpoint's own token counters, the price list, the pre-call estimate it was measured against, the budget rule it ran under, and the one under-count named",
          "media_type": "application/json",
          "bytes": 15663,
          "sha256": "8c1f1dfbc7309b03c5344e5a4ddb5705cf43a1fc1c5bef275227ec8bdd814793"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/verdict-sets.json",
          "role": "the answer sheets: every cloud judge's verdict objects as answered, paired with the original seat that read the same sealed pages, so the pairs compare slot for slot. The letter map is withheld — the sealed batches do not expire and the next re-audition uses them",
          "media_type": "application/json",
          "bytes": 1347776,
          "sha256": "b242c9277cc7323168d0b7606ef77821acda40f4b4c32783eee5483615c5d803"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/counting-rules.json",
          "role": "every table's counting rules in one file, copied out of the companions that use them",
          "media_type": "application/json",
          "bytes": 14436,
          "sha256": "7df3ac311ec430c8ddcb837ed758cd9a2e770ea22e572dd9b526634180bddc77"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/provenance.json",
          "role": "what was pinned before it could be scored: two pre-registrations with two shas each, the seat instrument's frozen three recomputed from disk, and the two provenance findings published rather than smoothed",
          "media_type": "application/json",
          "bytes": 9404,
          "sha256": "593edcb1fa3828b126071eb28598ae42a1246e2e40b3fd3c206f60c2beff21a6"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/index.json",
          "role": "the kit's own listing: files, sizes, sha256s",
          "media_type": "application/json",
          "bytes": 4899,
          "sha256": "fc7255441b5ac3f06d27303f8315a72af1dbf34e74cdd41e15a10fff7d59ad16"
        },
        {
          "url": "https://research.strata2signal.com/outside-judges/data/README.md",
          "role": "licence, receipts, the data-boundary statement, the contamination caveat, and how to read a number here",
          "media_type": "text/markdown",
          "bytes": 7222,
          "sha256": "ca14f13a7742485236b3b6a39d1951533da1051a7225d6f493a0bbaeda3742f0"
        }
      ],
      "data_note": "the multi-vendor judging dataset: five vendors' judges scoring identical blinded comparisons, with the disagreements kept. The rounds it extends stay DESCRIPTIVE — more vendors widen the error bar a reader is entitled to see, they do not turn a descriptive leg into a ranked one — and the seat instrument it reports is floors-and-counts, so nothing in this kit orders two candidates that both cleared",
      "one_liner": "Four rival frontier labs re-judged every sealed round this hub had published — 2 592 new verdict objects on byte-identical, still-blinded inputs — and our own two judges sat the house's frozen seat exam as candidates beside the challenger. Consensus on the headline, the disagreements printed, two perfect sheets the instrument declined to rank, and the bill to the cent."
    },
    {
      "exhibit": 5,
      "slug": "august-arrivals",
      "title": "The August arrivals — the fresh class sits the house exams",
      "url": "https://research.strata2signal.com/august-arrivals/",
      "published": "2026-08-12",
      "data": [
        {
          "url": "https://research.strata2signal.com/august-arrivals/data/seat-rows.json",
          "role": "the judge-seat rows for the fresh class — the same four scored rows as the seat-trials addendum kit, byte for byte, plus exhibit two's twelve July rows carried after them for side-by-side reading, which the seat-trials copy does not hold",
          "media_type": "application/json",
          "bytes": 30717,
          "sha256": "aed07dd3174150ae0bd97a4128f878661c9924acadfd20918eb4c8b65ccfbb26"
        },
        {
          "url": "https://research.strata2signal.com/august-arrivals/data/voice-rows.json",
          "role": "the narrator's-chair rows for the fresh class — byte-identical to the voice-trials addendum kit, verified 2026-08-15",
          "media_type": "application/json",
          "bytes": 61379,
          "sha256": "6fa1aeae38bc8f1b9fc116bc31e7feda632b70861c8ee8cfb7f01266af7b8835"
        },
        {
          "url": "https://research.strata2signal.com/august-arrivals/data/residency.json",
          "role": "the VRAM residency census — all seventeen registered (model, context length) rows with loaded size_vram in bytes and GiB, load seconds, outcomes, the daemon configuration, counting rules, the open finding on the three drafter rows, and provenance hashes",
          "media_type": "application/json",
          "bytes": 20038,
          "sha256": "1a6c7acf5790f59bfe21796ad2c5ba6dee3e3f9358682cadb664d8bb5e819617"
        },
        {
          "url": "https://research.strata2signal.com/august-arrivals/data/index.json",
          "role": "the kit's own listing: files, sizes, sha256s",
          "media_type": "application/json",
          "bytes": 3981,
          "sha256": "2766532088ffb1393ae574817a13c0a946fc79afa9e1ade40e12ad36a0bf49d4"
        },
        {
          "url": "https://research.strata2signal.com/august-arrivals/data/README.md",
          "role": "licence, receipts, the contamination caveat, and how to reuse these rows",
          "media_type": "text/markdown",
          "bytes": 8612,
          "sha256": "82b0682054ac95b30b0aacb331b364c2d4d0f135bf33cf13b777c983be0e4b35"
        }
      ],
      "one_liner": "Four models that arrived in one week of August 2026, sat both frozen house exams, and broke the same floor on each — the length floor, not the format floor."
    },
    {
      "exhibit": 1,
      "slug": "diffusion",
      "title": "The diffusion research bench",
      "url": "https://research.strata2signal.com/diffusion/",
      "published": "2026-08-11",
      "data": null,
      "data_note": "no kit yet — this exhibit predates the machine-readable law; every render on the page carries its full recipe inline, and raw rows are on request",
      "one_liner": "661 published renders across 21 experiments — which local image pipeline paints a living-world RPG, on real campaign content."
    },
    {
      "exhibit": 2,
      "slug": "seat-trials",
      "title": "The seat trials — how a local model earns a chair",
      "url": "https://research.strata2signal.com/seat-trials/",
      "published": "2026-08-11",
      "data": [
        {
          "url": "https://research.strata2signal.com/seat-trials/data/addendum-2026-08-12.json",
          "role": "the four fresh-class rows added 2026-08-12 — counts, verdicts, floors, failure ledgers by kind, counting rules, deviations, provenance hashes",
          "media_type": "application/json",
          "bytes": 25429,
          "sha256": "4819109f292c3343068bd266de9b9d2a42e8ad9eb0eb82ab8dd8af6c1c86e7c2"
        }
      ],
      "data_note": "the 2026-08-12 addendum ships a kit; the 21 rows published before it stay as-published, with raw rows on request",
      "one_liner": "23 models across 25 scored runs on one judge fixture: 43 claim-cases, a human-verified key, twin pre-registered floors, and five local seats that cleared both."
    },
    {
      "exhibit": 3,
      "slug": "voice-trials",
      "title": "The voice trials — how the narrator earned its voice",
      "url": "https://research.strata2signal.com/voice-trials/",
      "published": "2026-08-11",
      "data": [
        {
          "url": "https://research.strata2signal.com/voice-trials/data/addendum-2026-08-12.json",
          "role": "the six arms of the 2026-08 field — per-panel and combined dimension scores, envelope ledger, abstention adjudication with per-judge votes, letter map, provenance hashes",
          "media_type": "application/json",
          "bytes": 61379,
          "sha256": "6fa1aeae38bc8f1b9fc116bc31e7feda632b70861c8ee8cfb7f01266af7b8835"
        }
      ],
      "data_note": "the 2026-08 field ships a kit; the three earlier fields stay as-published, with the source record on request",
      "one_liner": "40 evaluations of 24 model tags across four fields, hunting one thing: a local model that can inhabit a character rather than assist."
    },
    {
      "exhibit": 4,
      "slug": "licences",
      "title": "The licence ledger — every model, read first-hand",
      "url": "https://research.strata2signal.com/licences/",
      "published": "2026-08-11",
      "data": null,
      "data_note": "no kit yet — a JSON companion for the rows added 2026-08-12 is the next lane's work; until it exists the page itself is the record",
      "one_liner": "The licensing authority for everything on this site: every model our products run or benched, its licence read from the actual text and stamped with the date of the last read."
    }
  ]
}
