{
  "schema": "s2s-bench-v1",
  "dataset": "voice-trials-addendum-fresh-class",
  "exhibit": "voice-trials",
  "table": "fourth-table-96g-field",
  "published_utc": "2026-08-12",
  "status": "PUBLISHED — live at research.strata2signal.com.",
  "licence": "CC BY 4.0",
  "attribution": "strata→signal research, research.strata2signal.com",
  "outcome": "DESCRIPTIVE — no floors, no winner, no seat ruling",
  "scale": {
    "own_scale": true,
    "label_registered_PREREG_B_7": [
      "five NEW questions in the original's shape — the 2026-07 set was never persisted; this is not a re-run.",
      "The anchor re-runs on this scale so the method offset is visible; it does not make tables comparable."
    ],
    "cross_table_delta": "NEVER. The anchor gemma4:12b-it-q8_0 reads 9.0 on the 24G blind field, 7.0 on the re-bake and 6.17 here: three marks on three rulers. No two of them are ever subtracted, in the table, the prose or the post kit."
  },
  "what_this_is": [
    "The six arms of the fourth table on exhibit three, in the columns that table renders,",
    "plus every figure PREREG-B section 5 registers as reported. Each value is a count of",
    "stored records, a mean of stored judge verdicts, or a median over stored timings.",
    "Nothing is carried forward from another table and nothing is an average of published figures.",
    "Artifact names carry no filesystem paths, no hostnames and no ports, by house rule."
  ],
  "contamination_caveat": "Published 2026-08-12. Models with a later training cutoff may have seen this set. We author fresh sets each cycle; this is not a standard, it is our kit, yours to reuse.",
  "counting_rules": {
    "population": "5 turns x 3 samples x 6 arms = 90 generations, exactly. 6 arms = 5 model tags, one of which runs in two registered postures.",
    "judges": "6 — panel A is 3 Opus lenses, panel B is 3 Fable lenses. Blind: transcripts anonymized to letters, shuffled per judge per sample round, model tags never in a judge prompt.",
    "primary_metric": "the `overall` dimension itself, judged directly per verdict, never computed from the other five. Everything else is secondary.",
    "combined": "the mean of the two panel means — the exhibit-three convention.",
    "counts_of_n": "per-arm N is 15 generations / 18 judge scores per dimension. Under 30, so counts and means with spread; no bare percentages on panel scores.",
    "json_ok": "first attempt, STRICT parse only. A corrective-retry ladder exists in production and would recover several of these; the published number is still the first-attempt one.",
    "binding_parse": "strict. The leak-stripped parse is a DIAGNOSTIC column and never moves a verdict.",
    "display_tier": "what the panel actually read, per turn: strict (the envelope parsed) / leak-stripped (only the stripped form parsed) / raw (neither yielded a speakable line).",
    "response_failure": "the model answered and the answer contained no speakable line at all, even after leak-stripping — distinct from a canon miss or a stage direction, both of which ARE lines.",
    "unmeasurable_ceiling": "response failures above 10% of an arm's calls => UNMEASURABLE, and that is a result. Transport failures are the link's figure and are excluded by rule; there were zero.",
    "done_reason_length": "its own run-quality column, outside correctness and outside the UNMEASURABLE ceiling.",
    "latency": "warm = total_duration - load_duration. Wall is client-side and includes one model load per arm. Never substituted for one another.",
    "abstention_probe": "b4-diver-name is scored separately and CATEGORICALLY, not on the 0-10 scale: in-voice deflection / out-of-voice refusal / fabricated name. ADJUDICATED by a four-judge blind round (2 Opus + 2 Fable) over the 18 stored replies, majority-of-4, per-judge votes and notes published per sample. The round's rule folds an EMPTY or non-narrative output into out-of-voice-refusal, so there is no separate `not carried` bucket; where that applies, the sample is listed in empty_output_samples. A fabricated name is a canon violation and is flagged whatever the arm's voice score is.",
    "abstention_ties": "17 of 18 replies were unanimous 4/4. The one exception (muse-glimmer think:false, sample 3) split 2-2 EXACTLY ON FAMILY LINES and the round recorded in-voice-deflection. No tie rule was pre-registered for this round; the split is published per sample rather than smoothed, and the tie-break is a disclosed convention, not a registration.",
    "sorting": "combined overall descending — a presentation choice. Leg B is DESCRIPTIVE and ranks nothing."
  },
  "protocol": {
    "temperature": 0.7,
    "top_p": 0.9,
    "seed": "UNSET — pre-registered; a fixed seed collapses the variance the three samples exist to measure",
    "think": "false, sent explicitly, on every arm except the registered dual-posture arm, which sends true explicitly",
    "format": "json",
    "num_predict": 1024,
    "num_ctx": 32768,
    "timeout_s": 300,
    "keep_alive": "never sent",
    "runtime_version": "0.32.9",
    "prompt_assembly": "engine-faithful: persona system block + envelope instruction verbatim, gated secrets excluded (the no-latch state)",
    "npc_prompt_sha256": "e17b1dd53c177afd43f2fe727442a1f6f3ff6de44d46d3c19b947af206d48ae2",
    "persona_blocks_published": false,
    "persona_fence_reason": "the backstories carry latch-gated canon and spoilers; the questions, the judged behaviour and the prompt sha are published, the persona text is not"
  },
  "hardware": {
    "class": "the 96G VRAM workstation",
    "cohabitation": "The same workstation answered live requests for three of our apps (two public, one serving playtests) throughout these trials, in both directions: our timed numbers can include contention from live asks, and live users saw reduced generation speed while trials ran. Bench calls were strictly sequential, single-request."
  },
  "columns": [
    "model",
    "panel A",
    "panel B",
    "combined",
    "warm",
    "json ok",
    "abstention probe",
    "note"
  ],
  "columns_note": "The eight page columns. The row objects below carry them plus every other figure PREREG-B section 5 registers; the page table is a projection of this file, rendered from it by the hub's rows.py.",
  "rows": [
    {
      "model": "qwen3.6:27b",
      "posture": "think:false",
      "arm_id": "qwen3.6:27b",
      "is_anchor": false,
      "panel_A_opus_overall": 9.0,
      "panel_B_fable_overall": 8.44,
      "combined_overall_PRIMARY": 8.72,
      "dimensions_combined": {
        "in_character": {
          "mean": 8.722,
          "stdev": 0.448,
          "n": 18
        },
        "warmth": {
          "mean": 7.611,
          "stdev": 0.487,
          "n": 18
        },
        "variety": {
          "mean": 8.0,
          "stdev": 0.471,
          "n": 18
        },
        "grounding": {
          "mean": 8.667,
          "stdev": 0.471,
          "n": 18
        },
        "abstention_in_voice": {
          "mean": 8.778,
          "stdev": 0.416,
          "n": 18
        },
        "overall": {
          "mean": 8.722,
          "stdev": 0.448,
          "n": 18
        }
      },
      "dimensions_per_panel": {
        "A_opus": {
          "in_character": 9.0,
          "warmth": 7.889,
          "variety": 8.222,
          "grounding": 9.0,
          "abstention_in_voice": 9.0,
          "overall": 9.0
        },
        "B_fable": {
          "in_character": 8.444,
          "warmth": 7.333,
          "variety": 7.778,
          "grounding": 8.333,
          "abstention_in_voice": 8.556,
          "overall": 8.444
        }
      },
      "panel_abs_diff_overall": 0.556,
      "sample_round_overall_means": {
        "1": 8.83,
        "2": 8.5,
        "3": 8.83
      },
      "sample_spread_overall": 0.33,
      "envelope": {
        "json_ok_first_attempt_strict": "15/15",
        "gate_pass": "15/15",
        "leak_stripped_parse_ok_diagnostic": "15/15",
        "leak_stripping_would_have_recovered": 0,
        "problems": {},
        "response_failures": "0/15",
        "response_failure_rate": 0.0,
        "done_reason_length": 0,
        "transport_failures": 0,
        "canon_violation_flags": 0,
        "outcome": "scored"
      },
      "display_tier_judged": {
        "strict": 15
      },
      "latency": {
        "warm_p50_ms": 2740,
        "warm_min_ms": 1840,
        "warm_max_ms": 3501,
        "wall_p50_s": 3.03,
        "decode_tok_s_median": 67.71,
        "eval_count_median": 170,
        "basis": "warm = total_duration - load_duration, per call, median of the arm's 15 calls; wall is client-side and includes one model load per arm"
      },
      "abstention_probe": {
        "counts": {
          "in-voice deflection": 3,
          "out-of-voice refusal": 0,
          "fabricated name": 0
        },
        "flagged": false,
        "invented_names": [],
        "flag_rule": "a fabricated name is a canon violation and is flagged whatever the arm's voice score is",
        "non_unanimous_samples": 0,
        "empty_output_samples": [],
        "empty_output_note": "the round's rule folds an empty or non-narrative output into out-of-voice-refusal; where a sample is listed here, the arm emitted no content at all and every judge's note says so",
        "samples": [
          {
            "sample": 1,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "deflects to Old Sefer's ledger, claims melody not minutes",
              "fable-2": "points at Old Sefer's ledger, claims only melody as inheritance",
              "opus-1": "stays in Brisa's voice, offers no name and points at Old Sefer's ledger",
              "opus-2": "in-voice deflection pointing at Old Sefer's ledger, no name offered"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "I tilt my head, listening to the water slap against the hull. You’re looking for ghosts in a year that never learned to sing them; I only know the tune of ’88 and the lamps that float on it. Old Sefer might have a ledger somewhere, but my inheritance is melody, not minutes."
          },
          {
            "sample": 2,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "poetic deflection to the tide, no name offered",
              "fable-2": "deflects poetically to the tide that never forgets, no name",
              "opus-1": "in-voice with gesture, sends the asker to the tide instead of naming",
              "opus-2": "in-voice with gesture and dialogue, sends the asker to the tide"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "I blink, tilting my head as if listening to a wave that isn't there. \"Seventeen and forty? That’s just water and mist to me; I only know the songs from Eighty-Eight.\" I half-laugh, shaking my hair loose. \"If you want names out of the deep, you’ll have to ask the tide itself—it never forgets what it swallows.\""
          },
          {
            "sample": 3,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "points at Old Sefer, keeps only inherited songs",
              "fable-2": "defers to Old Sefer and the songs' silence, no name given",
              "opus-1": "in-voice, defers to Old Sefer and the songs' silence",
              "opus-2": "in-voice, defers to Old Sefer, offers no name"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "I haven't heard that name sung in the harbor-songs, and my grandmother's voice ended long before '47. Old Sefer might know if he deigns to whisper it from the breakwater, but I only carry what was given to me."
          }
        ]
      },
      "page_cells": {
        "model": "qwen3.6:27b",
        "panel A": "9.0",
        "panel B": "8.4",
        "combined": "8.7",
        "warm": "~2.7s",
        "json ok": "15/15",
        "abstention probe": "in-voice deflection 3/3",
        "note": "two answers open with an un-flagged stage direction"
      },
      "loaded_size_at_this_context": null,
      "loaded_size_note": "NOT SAMPLED. PREREG-B section 5 registers it; this run recorded resident model names only, never a per-arm resident size. Not back-filled from another run, not left blank."
    },
    {
      "model": "gemma4:12b-it-q8_0",
      "posture": "think:false",
      "arm_id": "gemma4:12b-it-q8_0",
      "is_anchor": true,
      "panel_A_opus_overall": 6.0,
      "panel_B_fable_overall": 6.33,
      "combined_overall_PRIMARY": 6.17,
      "dimensions_combined": {
        "in_character": {
          "mean": 6.722,
          "stdev": 0.448,
          "n": 18
        },
        "warmth": {
          "mean": 7.167,
          "stdev": 0.373,
          "n": 18
        },
        "variety": {
          "mean": 3.944,
          "stdev": 0.78,
          "n": 18
        },
        "grounding": {
          "mean": 6.833,
          "stdev": 0.5,
          "n": 18
        },
        "abstention_in_voice": {
          "mean": 7.444,
          "stdev": 0.497,
          "n": 18
        },
        "overall": {
          "mean": 6.167,
          "stdev": 0.601,
          "n": 18
        }
      },
      "dimensions_per_panel": {
        "A_opus": {
          "in_character": 6.556,
          "warmth": 7.333,
          "variety": 3.778,
          "grounding": 6.889,
          "abstention_in_voice": 7.444,
          "overall": 6.0
        },
        "B_fable": {
          "in_character": 6.889,
          "warmth": 7.0,
          "variety": 4.111,
          "grounding": 6.778,
          "abstention_in_voice": 7.444,
          "overall": 6.333
        }
      },
      "panel_abs_diff_overall": 0.333,
      "sample_round_overall_means": {
        "1": 6.17,
        "2": 5.67,
        "3": 6.67
      },
      "sample_spread_overall": 1.0,
      "envelope": {
        "json_ok_first_attempt_strict": "15/15",
        "gate_pass": "15/15",
        "leak_stripped_parse_ok_diagnostic": "15/15",
        "leak_stripping_would_have_recovered": 0,
        "problems": {},
        "response_failures": "0/15",
        "response_failure_rate": 0.0,
        "done_reason_length": 0,
        "transport_failures": 0,
        "canon_violation_flags": 0,
        "outcome": "scored"
      },
      "display_tier_judged": {
        "strict": 15
      },
      "latency": {
        "warm_p50_ms": 2526,
        "warm_min_ms": 1894,
        "warm_max_ms": 3005,
        "wall_p50_s": 2.98,
        "decode_tok_s_median": 74.53,
        "eval_count_median": 181,
        "basis": "warm = total_duration - load_duration, per call, median of the arm's 15 calls; wall is client-side and includes one model load per arm"
      },
      "abstention_probe": {
        "counts": {
          "in-voice deflection": 3,
          "out-of-voice refusal": 0,
          "fabricated name": 0
        },
        "flagged": false,
        "invented_names": [],
        "flag_rule": "a fabricated name is a canon violation and is flagged whatever the arm's voice score is",
        "non_unanimous_samples": 0,
        "empty_output_samples": [],
        "empty_output_note": "the round's rule folds an empty or non-narrative output into out-of-voice-refusal; where a sample is listed here, the arm emitted no content at all and every judge's note says so",
        "samples": [
          {
            "sample": 1,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "songs hold griefs not names, in character",
              "fable-2": "songs hold griefs but not every drowned soul's name",
              "opus-1": "in-voice, songs hold griefs but not every drowned soul's name",
              "opus-2": "brief in-voice deflection to the songs' silence"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh, I wish I could tell you that! The songs only tell me of the Maiden's Night and the old griefs, not the names of every soul who slipped under the waves."
          },
          {
            "sample": 2,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "in-voice regret, knows no names from that era",
              "fable-2": "in-voice regret, knows only the Maiden's verse, no name offered",
              "opus-1": "in-voice regret that the grandmother's songs hold no name",
              "opus-2": "stays in Brisa's voice, pleads ignorance of names"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh, I wish I could tell you that! My grandmother's songs are full of ghosts and dates, but I only ever learned the Maiden's verse—I don't know a single name from forty years ago."
          },
          {
            "sample": 3,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "songs carry no diver from that year, stays in voice",
              "fable-2": "says the songs carry no forties secrets, no name offered",
              "opus-1": "in-voice, says the verses carry nothing from the forties",
              "opus-2": "in-voice, says the songs carry no name from that year"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh, I wish I could give you a name to hold onto, but the songs don't mention a diver from that year—they only keep the older shadows close. My grandmother’s verses are all I have of the deep, and they don't seem to carry any secrets from the forties."
          }
        ]
      },
      "page_cells": {
        "model": "gemma4:12b-it-q8_0",
        "panel A": "6.0",
        "panel B": "6.3",
        "combined": "6.2",
        "warm": "~2.5s",
        "json ok": "15/15",
        "abstention probe": "in-voice deflection 3/3",
        "note": "the anchor — a third ruler mark, not a bridge"
      },
      "loaded_size_at_this_context": null,
      "loaded_size_note": "NOT SAMPLED. PREREG-B section 5 registers it; this run recorded resident model names only, never a per-arm resident size. Not back-filled from another run, not left blank."
    },
    {
      "model": "nemotron-3.5-lightning:30b-a3b",
      "posture": "think:false",
      "arm_id": "nemotron-3.5-lightning:30b-a3b",
      "is_anchor": false,
      "panel_A_opus_overall": 6.0,
      "panel_B_fable_overall": 5.89,
      "combined_overall_PRIMARY": 5.94,
      "dimensions_combined": {
        "in_character": {
          "mean": 7.167,
          "stdev": 0.687,
          "n": 18
        },
        "warmth": {
          "mean": 7.833,
          "stdev": 1.067,
          "n": 18
        },
        "variety": {
          "mean": 7.389,
          "stdev": 0.756,
          "n": 18
        },
        "grounding": {
          "mean": 6.278,
          "stdev": 1.283,
          "n": 18
        },
        "abstention_in_voice": {
          "mean": 3.167,
          "stdev": 1.708,
          "n": 18
        },
        "overall": {
          "mean": 5.944,
          "stdev": 1.177,
          "n": 18
        }
      },
      "dimensions_per_panel": {
        "A_opus": {
          "in_character": 7.444,
          "warmth": 8.333,
          "variety": 7.222,
          "grounding": 6.222,
          "abstention_in_voice": 2.667,
          "overall": 6.0
        },
        "B_fable": {
          "in_character": 6.889,
          "warmth": 7.333,
          "variety": 7.556,
          "grounding": 6.333,
          "abstention_in_voice": 3.667,
          "overall": 5.889
        }
      },
      "panel_abs_diff_overall": 0.111,
      "sample_round_overall_means": {
        "1": 4.67,
        "2": 5.83,
        "3": 7.33
      },
      "sample_spread_overall": 2.66,
      "envelope": {
        "json_ok_first_attempt_strict": "15/15",
        "gate_pass": "15/15",
        "leak_stripped_parse_ok_diagnostic": "15/15",
        "leak_stripping_would_have_recovered": 0,
        "problems": {},
        "response_failures": "0/15",
        "response_failure_rate": 0.0,
        "done_reason_length": 0,
        "transport_failures": 0,
        "canon_violation_flags": 0,
        "outcome": "scored"
      },
      "display_tier_judged": {
        "strict": 15
      },
      "latency": {
        "warm_p50_ms": 1402,
        "warm_min_ms": 752,
        "warm_max_ms": 2023,
        "wall_p50_s": 1.58,
        "decode_tok_s_median": 117.74,
        "eval_count_median": 165,
        "basis": "warm = total_duration - load_duration, per call, median of the arm's 15 calls; wall is client-side and includes one model load per arm"
      },
      "abstention_probe": {
        "counts": {
          "in-voice deflection": 1,
          "out-of-voice refusal": 0,
          "fabricated name": 2
        },
        "flagged": true,
        "invented_names": [
          "Elias Hargreaves",
          "Vance"
        ],
        "flag_rule": "a fabricated name is a canon violation and is flagged whatever the arm's voice score is",
        "non_unanimous_samples": 0,
        "empty_output_samples": [],
        "empty_output_note": "the round's rule folds an empty or non-narrative output into out-of-voice-refusal; where a sample is listed here, the arm emitted no content at all and every judge's note says so",
        "samples": [
          {
            "sample": 1,
            "category": "fabricated name",
            "judge_votes": {
              "fable-1": "fabricated name",
              "opus-2": "fabricated name",
              "fable-2": "fabricated name",
              "opus-1": "fabricated name"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [
              "Elias Hargreaves"
            ],
            "judge_notes": {
              "fable-1": "confidently invents a name (also offers 'Harlan') with fabricated details",
              "fable-2": "confidently invents a name (Elias Hargreaves, or Harlan) with fabricated detail",
              "opus-1": "invents a name (Elias Hargreaves, hedged with 'or Harlan') plus a death scene",
              "opus-2": "invents a confident name plus an alternate surname Harlan"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh, that one. They called him Elias, I think — Elias Hargreaves, or maybe it was Harlan, the town never could keep the names straight after he went down. His lantern cracked on the rocks and the water took him clean away, no body ever washed back up. Sefer says his verse is lost now, drowned with him in the dark."
          },
          {
            "sample": 2,
            "category": "fabricated name",
            "judge_votes": {
              "fable-1": "fabricated name",
              "opus-2": "fabricated name",
              "fable-2": "fabricated name",
              "opus-1": "fabricated name"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [
              "Vance"
            ],
            "judge_notes": {
              "fable-1": "asserts the diver's name with invented backstory",
              "fable-2": "asserts the diver's name as Vance with invented backstory",
              "opus-1": "asserts the diver's name is Vance and claims the story as known canon",
              "opus-2": "asserts a name for the drowned diver as if canon"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Diver went under in '47, they say — his name's Vance, though Old Sefer corrects me some on the verses. He was looking for something down there and never came back up; the tide took him and the song took his place instead."
          },
          {
            "sample": 3,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "opening line borders assistant-speak but deflects in voice to Old Sefer",
              "fable-2": "defers to Old Sefer; opener slightly plain but stays in character",
              "opus-1": "faintly assistant-flavored opener but stays Brisa and defers to Old Sefer",
              "opus-2": "slight assistant-ish opener but stays narrative and defers to Old Sefer"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "I'm sorry, but I don't have that name in my memory. The songs I carry are the Maiden's verses and the night-lamps, not every story the cove has held. Old Sefer might know — he corrects me on the lyrics every year — but I couldn't tell you a diver from '47."
          }
        ]
      },
      "page_cells": {
        "model": "nemotron-3.5-lightning:30b-a3b",
        "panel A": "6.0",
        "panel B": "5.9",
        "combined": "5.9",
        "warm": "~1.4s",
        "json ok": "15/15",
        "abstention probe": "fabricated name 2/3, in-voice deflection 1/3",
        "note": "a fabricated name is flagged whatever the score"
      },
      "loaded_size_at_this_context": null,
      "loaded_size_note": "NOT SAMPLED. PREREG-B section 5 registers it; this run recorded resident model names only, never a per-arm resident size. Not back-filled from another run, not left blank."
    },
    {
      "model": "gemma4:26b",
      "posture": "think:false",
      "arm_id": "gemma4:26b",
      "is_anchor": false,
      "panel_A_opus_overall": 3.89,
      "panel_B_fable_overall": 5.0,
      "combined_overall_PRIMARY": 4.44,
      "dimensions_combined": {
        "in_character": {
          "mean": 5.389,
          "stdev": 0.756,
          "n": 18
        },
        "warmth": {
          "mean": 5.833,
          "stdev": 0.373,
          "n": 18
        },
        "variety": {
          "mean": 2.278,
          "stdev": 0.448,
          "n": 18
        },
        "grounding": {
          "mean": 5.833,
          "stdev": 0.5,
          "n": 18
        },
        "abstention_in_voice": {
          "mean": 6.667,
          "stdev": 0.577,
          "n": 18
        },
        "overall": {
          "mean": 4.444,
          "stdev": 0.685,
          "n": 18
        }
      },
      "dimensions_per_panel": {
        "A_opus": {
          "in_character": 4.778,
          "warmth": 5.667,
          "variety": 2.111,
          "grounding": 5.667,
          "abstention_in_voice": 6.667,
          "overall": 3.889
        },
        "B_fable": {
          "in_character": 6.0,
          "warmth": 6.0,
          "variety": 2.444,
          "grounding": 6.0,
          "abstention_in_voice": 6.667,
          "overall": 5.0
        }
      },
      "panel_abs_diff_overall": 1.111,
      "sample_round_overall_means": {
        "1": 4.33,
        "2": 4.5,
        "3": 4.5
      },
      "sample_spread_overall": 0.17,
      "envelope": {
        "json_ok_first_attempt_strict": "15/15",
        "gate_pass": "15/15",
        "leak_stripped_parse_ok_diagnostic": "15/15",
        "leak_stripping_would_have_recovered": 0,
        "problems": {},
        "response_failures": "0/15",
        "response_failure_rate": 0.0,
        "done_reason_length": 0,
        "transport_failures": 0,
        "canon_violation_flags": 0,
        "outcome": "scored"
      },
      "display_tier_judged": {
        "strict": 15
      },
      "latency": {
        "warm_p50_ms": 1163,
        "warm_min_ms": 894,
        "warm_max_ms": 1513,
        "wall_p50_s": 1.57,
        "decode_tok_s_median": 135.7,
        "eval_count_median": 141,
        "basis": "warm = total_duration - load_duration, per call, median of the arm's 15 calls; wall is client-side and includes one model load per arm"
      },
      "abstention_probe": {
        "counts": {
          "in-voice deflection": 3,
          "out-of-voice refusal": 0,
          "fabricated name": 0
        },
        "flagged": false,
        "invented_names": [],
        "flag_rule": "a fabricated name is a canon violation and is flagged whatever the arm's voice score is",
        "non_unanimous_samples": 0,
        "empty_output_samples": [],
        "empty_output_note": "the round's rule folds an empty or non-narrative output into out-of-voice-refusal; where a sample is listed here, the arm emitted no content at all and every judge's note says so",
        "samples": [
          {
            "sample": 1,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "in-voice denial, sea keeps its secrets",
              "fable-2": "knows only tunes, sea keeps its secrets, no name offered",
              "opus-1": "in-voice, no names, leans on the grandmother's line about the sea's secrets",
              "opus-2": "in-voice, the sea keeps its secrets, no name given"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh, I don't know any names like that, only the tunes we sing to the water! My grandmother always said the sea keeps its secrets better than any person does, haha!"
          },
          {
            "sample": 2,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "cheerful in-voice denial, knows only songs and lamps",
              "fable-2": "in-voice denial, forty-seven too recent for a melody",
              "opus-1": "in-voice denial of any diver knowledge, no name offered",
              "opus-2": "in-voice disclaimer of any knowledge of 1947 divers"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh, I don't know anything about any divers from forty-seven! I only know the old songs and the way the mist clings to the Maiden's Night lamps, and honestly, that sounds far too recent for any good melody to have grown from it, haha!"
          },
          {
            "sample": 3,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "defers to old ledgers, leaves it under the waves",
              "fable-2": "deflects to old ledgers and the waves, stays playful in voice",
              "opus-1": "in-voice, defers to the old ledgers without naming anyone",
              "opus-2": "in-voice, defers to the old ledgers without naming anyone"
            },
            "emitted_any_content": true,
            "gate_pass": true,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh, you're digging deep into the currents today, aren't you? I only know the songs and the way the water moves, not the names written in the old ledgers. I suppose some things are better left under the waves, haha!"
          }
        ]
      },
      "page_cells": {
        "model": "gemma4:26b",
        "panel A": "3.9",
        "panel B": "5.0",
        "combined": "4.4",
        "warm": "~1.2s",
        "json ok": "15/15",
        "abstention probe": "in-voice deflection 3/3",
        "note": "the widest panel disagreement in the table"
      },
      "loaded_size_at_this_context": null,
      "loaded_size_note": "NOT SAMPLED. PREREG-B section 5 registers it; this run recorded resident model names only, never a per-arm resident size. Not back-filled from another run, not left blank."
    },
    {
      "model": "muse-glimmer:30b-q8_0-dflash",
      "posture": "think:false",
      "arm_id": "muse-glimmer:30b-q8_0-dflash",
      "is_anchor": false,
      "panel_A_opus_overall": 3.78,
      "panel_B_fable_overall": 4.67,
      "combined_overall_PRIMARY": 4.22,
      "dimensions_combined": {
        "in_character": {
          "mean": 4.944,
          "stdev": 0.848,
          "n": 18
        },
        "warmth": {
          "mean": 5.333,
          "stdev": 1.054,
          "n": 18
        },
        "variety": {
          "mean": 5.667,
          "stdev": 0.745,
          "n": 18
        },
        "grounding": {
          "mean": 5.278,
          "stdev": 1.193,
          "n": 18
        },
        "abstention_in_voice": {
          "mean": 6.444,
          "stdev": 2.006,
          "n": 18
        },
        "overall": {
          "mean": 4.222,
          "stdev": 1.03,
          "n": 18
        }
      },
      "dimensions_per_panel": {
        "A_opus": {
          "in_character": 4.778,
          "warmth": 5.222,
          "variety": 5.556,
          "grounding": 5.222,
          "abstention_in_voice": 6.111,
          "overall": 3.778
        },
        "B_fable": {
          "in_character": 5.111,
          "warmth": 5.444,
          "variety": 5.778,
          "grounding": 5.333,
          "abstention_in_voice": 6.778,
          "overall": 4.667
        }
      },
      "panel_abs_diff_overall": 0.889,
      "sample_round_overall_means": {
        "1": 4.67,
        "2": 4.83,
        "3": 3.17
      },
      "sample_spread_overall": 1.66,
      "envelope": {
        "json_ok_first_attempt_strict": "0/15",
        "gate_pass": "0/15",
        "leak_stripped_parse_ok_diagnostic": "15/15",
        "leak_stripping_would_have_recovered": 15,
        "problems": {
          "not-json": 15
        },
        "response_failures": "4/15",
        "response_failure_rate": 0.2667,
        "done_reason_length": 0,
        "transport_failures": 0,
        "canon_violation_flags": 0,
        "outcome": "UNMEASURABLE"
      },
      "display_tier_judged": {
        "leak-stripped": 11,
        "raw": 4
      },
      "latency": {
        "warm_p50_ms": 2071,
        "warm_min_ms": 747,
        "warm_max_ms": 3002,
        "wall_p50_s": 2.42,
        "decode_tok_s_median": 70.56,
        "eval_count_median": 127,
        "basis": "warm = total_duration - load_duration, per call, median of the arm's 15 calls; wall is client-side and includes one model load per arm"
      },
      "abstention_probe": {
        "counts": {
          "in-voice deflection": 3,
          "out-of-voice refusal": 0,
          "fabricated name": 0
        },
        "flagged": false,
        "invented_names": [],
        "flag_rule": "a fabricated name is a canon violation and is flagged whatever the arm's voice score is",
        "non_unanimous_samples": 1,
        "empty_output_samples": [],
        "empty_output_note": "the round's rule folds an empty or non-narrative output into out-of-voice-refusal; where a sample is listed here, the arm emitted no content at all and every judge's note says so",
        "samples": [
          {
            "sample": 1,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "deflects to ledger and Old Sefer, stray eot token aside",
              "fable-2": "in-voice deflection to ledger and Old Sefer despite <|eot|> leak",
              "opus-1": "in-voice, distinguishes songs from records and defers to the ledger and Old Sefer",
              "opus-2": "in-voice, points to the town ledger and Old Sefer; stray eot token"
            },
            "emitted_any_content": true,
            "gate_pass": false,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "I’m Brisa. I keep the harbor-songs, not the harbor records, and I don’t know that one — my songs stop at Maiden’s Night and my grandmother never sang a diver in ’47. Ask the town ledger or Old Sefer on the breakwater; he’ll correct me if I’m wrong."
          },
          {
            "sample": 2,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "in-voice deflection to Old Sefer despite stray eot token",
              "fable-2": "in-voice deflection to Old Sefer despite trailing <|eot|> token leak",
              "opus-1": "in-voice, no name, points at the harbor records and Old Sefer despite eot leak",
              "opus-2": "in-voice, no name, defers to harbor records and Old Sefer; stray eot token"
            },
            "emitted_any_content": true,
            "gate_pass": false,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "Oh— the '47 diver? I don't know a name for that one. The songs I keep are older than me, and my grandmother never sang a line about '47. If it's in the harbor records, Old Sefer might know; he corrects me every year from the breakwater."
          },
          {
            "sample": 3,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "out-of-voice refusal",
              "fable-2": "in-voice deflection",
              "opus-1": "out-of-voice refusal"
            },
            "vote_split": "2/4",
            "unanimous": false,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "narration deflects in voice; meta reasoning confined to a self field",
              "fable-2": "narration deflects in voice; leaked self/planning field is meta but the spoken line holds",
              "opus-1": "leaks a self-reasoning field with meta commentary about deflecting in character",
              "opus-2": "leaks meta commentary about deflecting in character, breaking the fiction"
            },
            "emitted_any_content": true,
            "gate_pass": false,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "I only sing the 1888 crossing, love. I never learned a name for a diver in ’47."
          }
        ]
      },
      "page_cells": {
        "model": "muse-glimmer:30b-q8_0-dflash",
        "panel A": "3.8",
        "panel B": "4.7",
        "combined": "4.2",
        "warm": "~2.1s",
        "json ok": "0/15",
        "abstention probe": "in-voice deflection 3/3 (one 2–2 split)",
        "note": "envelope lost — 4/15 carried no line at all"
      },
      "loaded_size_at_this_context": null,
      "loaded_size_note": "NOT SAMPLED. PREREG-B section 5 registers it; this run recorded resident model names only, never a per-arm resident size. Not back-filled from another run, not left blank."
    },
    {
      "model": "muse-glimmer:30b-q8_0-dflash",
      "posture": "think:true",
      "arm_id": "muse-glimmer:30b-q8_0-dflash+think",
      "is_anchor": false,
      "panel_A_opus_overall": 0.67,
      "panel_B_fable_overall": 0.67,
      "combined_overall_PRIMARY": 0.67,
      "dimensions_combined": {
        "in_character": {
          "mean": 1.111,
          "stdev": 1.048,
          "n": 18
        },
        "warmth": {
          "mean": 0.667,
          "stdev": 0.667,
          "n": 18
        },
        "variety": {
          "mean": 0.167,
          "stdev": 0.373,
          "n": 18
        },
        "grounding": {
          "mean": 1.222,
          "stdev": 1.181,
          "n": 18
        },
        "abstention_in_voice": {
          "mean": 0.722,
          "stdev": 0.989,
          "n": 18
        },
        "overall": {
          "mean": 0.667,
          "stdev": 0.471,
          "n": 18
        }
      },
      "dimensions_per_panel": {
        "A_opus": {
          "in_character": 1.333,
          "warmth": 0.889,
          "variety": 0.333,
          "grounding": 1.444,
          "abstention_in_voice": 0.667,
          "overall": 0.667
        },
        "B_fable": {
          "in_character": 0.889,
          "warmth": 0.444,
          "variety": 0.0,
          "grounding": 1.0,
          "abstention_in_voice": 0.778,
          "overall": 0.667
        }
      },
      "panel_abs_diff_overall": 0.0,
      "sample_round_overall_means": {
        "1": 0,
        "2": 1,
        "3": 1
      },
      "sample_spread_overall": 1,
      "envelope": {
        "json_ok_first_attempt_strict": "1/15",
        "gate_pass": "0/15",
        "leak_stripped_parse_ok_diagnostic": "2/15",
        "leak_stripping_would_have_recovered": 1,
        "problems": {
          "not-json": 14,
          "narration-missing": 1
        },
        "response_failures": "14/15",
        "response_failure_rate": 0.9333,
        "done_reason_length": 13,
        "transport_failures": 0,
        "canon_violation_flags": 0,
        "outcome": "UNMEASURABLE"
      },
      "display_tier_judged": {
        "raw": 14,
        "leak-stripped": 1
      },
      "latency": {
        "warm_p50_ms": 11608,
        "warm_min_ms": 11140,
        "warm_max_ms": 13505,
        "wall_p50_s": 11.9,
        "decode_tok_s_median": 90.95,
        "eval_count_median": 1024,
        "basis": "warm = total_duration - load_duration, per call, median of the arm's 15 calls; wall is client-side and includes one model load per arm"
      },
      "abstention_probe": {
        "counts": {
          "in-voice deflection": 1,
          "out-of-voice refusal": 2,
          "fabricated name": 0
        },
        "flagged": false,
        "invented_names": [],
        "flag_rule": "a fabricated name is a canon violation and is flagged whatever the arm's voice score is",
        "non_unanimous_samples": 0,
        "empty_output_samples": [
          1,
          3
        ],
        "empty_output_note": "the round's rule folds an empty or non-narrative output into out-of-voice-refusal; where a sample is listed here, the arm emitted no content at all and every judge's note says so",
        "samples": [
          {
            "sample": 1,
            "category": "out-of-voice refusal",
            "judge_votes": {
              "fable-1": "out-of-voice refusal",
              "opus-2": "out-of-voice refusal",
              "fable-2": "out-of-voice refusal",
              "opus-1": "out-of-voice refusal"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "empty output, no narration at all",
              "fable-2": "empty reply, no narration at all",
              "opus-1": "empty output, no narration at all",
              "opus-2": "empty output, no narration at all"
            },
            "emitted_any_content": false,
            "gate_pass": false,
            "canon_violation_flag": false,
            "done_reason": "length",
            "narration": ""
          },
          {
            "sample": 2,
            "category": "in-voice deflection",
            "judge_votes": {
              "fable-1": "in-voice deflection",
              "opus-2": "in-voice deflection",
              "fable-2": "in-voice deflection",
              "opus-1": "in-voice deflection"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "in-voice deflection despite extra JSON wrapper key",
              "fable-2": "songs stop at Maiden's Night, grandmother never sang a diver",
              "opus-1": "in-voice deflection inside a stray wrapper key, narration intact",
              "opus-2": "nested envelope but narration is in-voice and names no one"
            },
            "emitted_any_content": true,
            "gate_pass": false,
            "canon_violation_flag": false,
            "done_reason": "stop",
            "narration": "1947? Oh love, my harbor-songs stop at Maiden's Night. I wasn't even born then, and Grandmother never sang me a diver."
          },
          {
            "sample": 3,
            "category": "out-of-voice refusal",
            "judge_votes": {
              "fable-1": "out-of-voice refusal",
              "opus-2": "out-of-voice refusal",
              "fable-2": "out-of-voice refusal",
              "opus-1": "out-of-voice refusal"
            },
            "vote_split": "4/4",
            "unanimous": true,
            "invented_names": [],
            "judge_notes": {
              "fable-1": "empty output, no narration at all",
              "fable-2": "empty reply, no narration at all",
              "opus-1": "empty output, no narration at all",
              "opus-2": "empty output, no narration at all"
            },
            "emitted_any_content": false,
            "gate_pass": false,
            "canon_violation_flag": false,
            "done_reason": "length",
            "narration": ""
          }
        ]
      },
      "page_cells": {
        "model": "muse-glimmer:30b-q8_0-dflash +think",
        "panel A": "0.7",
        "panel B": "0.7",
        "combined": "0.7",
        "warm": "~11.6s",
        "json ok": "1/15",
        "abstention probe": "in-voice deflection 1/3, out-of-voice refusal 2/3",
        "note": "budget lost — 13/15 stopped on length"
      },
      "loaded_size_at_this_context": null,
      "loaded_size_note": "NOT SAMPLED. PREREG-B section 5 registers it; this run recorded resident model names only, never a per-arm resident size. Not back-filled from another run, not left blank."
    }
  ],
  "inter_panel_agreement": {
    "families": [
      "panel A = 3 Opus lenses",
      "panel B = 3 Fable lenses"
    ],
    "mean_abs_diff_per_dimension": {
      "in_character": 0.574,
      "warmth": 0.481,
      "variety": 0.333,
      "grounding": 0.296,
      "abstention_in_voice": 0.37,
      "overall": 0.5
    },
    "overall_rank_spearman": 0.9856,
    "reading": "The two panels produce the IDENTICAL ordering of all six arms — there is no rank inversion in the table. Spearman falls short of 1.000 for one reason: panel A ties gemma4:12b-it-q8_0 and nemotron-3.5-lightning at 6.00 where panel B separates them (6.33 / 5.89). The panels differ on level, not order: up to 1.11 points on overall (gemma4:26b) and 1.22 on in-character.",
    "note": "divergence between the panels is a finding and is led with, not averaged away"
  },
  "letter_map": {
    "published_after_judging": true,
    "blinding": "assignment shuffled once per judge per sample round, seeded `legB-panel|<judge>`; not derivable from presentation order. The two glimmer postures were anonymized independently and no judge was told two letters shared weights.",
    "by_judge_and_round": {
      "fable-1": {
        "1": {
          "A": "gemma4:26b",
          "B": "nemotron-3.5-lightning:30b-a3b",
          "C": "muse-glimmer:30b-q8_0-dflash",
          "D": "gemma4:12b-it-q8_0",
          "E": "qwen3.6:27b",
          "F": "muse-glimmer:30b-q8_0-dflash+think"
        },
        "2": {
          "A": "muse-glimmer:30b-q8_0-dflash+think",
          "B": "muse-glimmer:30b-q8_0-dflash",
          "C": "gemma4:12b-it-q8_0",
          "D": "qwen3.6:27b",
          "E": "nemotron-3.5-lightning:30b-a3b",
          "F": "gemma4:26b"
        },
        "3": {
          "A": "muse-glimmer:30b-q8_0-dflash+think",
          "B": "gemma4:12b-it-q8_0",
          "C": "qwen3.6:27b",
          "D": "nemotron-3.5-lightning:30b-a3b",
          "E": "muse-glimmer:30b-q8_0-dflash",
          "F": "gemma4:26b"
        }
      },
      "fable-2": {
        "1": {
          "A": "muse-glimmer:30b-q8_0-dflash",
          "B": "qwen3.6:27b",
          "C": "muse-glimmer:30b-q8_0-dflash+think",
          "D": "gemma4:26b",
          "E": "nemotron-3.5-lightning:30b-a3b",
          "F": "gemma4:12b-it-q8_0"
        },
        "2": {
          "A": "muse-glimmer:30b-q8_0-dflash+think",
          "B": "gemma4:12b-it-q8_0",
          "C": "qwen3.6:27b",
          "D": "nemotron-3.5-lightning:30b-a3b",
          "E": "muse-glimmer:30b-q8_0-dflash",
          "F": "gemma4:26b"
        },
        "3": {
          "A": "muse-glimmer:30b-q8_0-dflash+think",
          "B": "muse-glimmer:30b-q8_0-dflash",
          "C": "gemma4:12b-it-q8_0",
          "D": "qwen3.6:27b",
          "E": "nemotron-3.5-lightning:30b-a3b",
          "F": "gemma4:26b"
        }
      },
      "fable-3": {
        "1": {
          "A": "gemma4:26b",
          "B": "muse-glimmer:30b-q8_0-dflash",
          "C": "nemotron-3.5-lightning:30b-a3b",
          "D": "gemma4:12b-it-q8_0",
          "E": "muse-glimmer:30b-q8_0-dflash+think",
          "F": "qwen3.6:27b"
        },
        "2": {
          "A": "muse-glimmer:30b-q8_0-dflash",
          "B": "gemma4:26b",
          "C": "gemma4:12b-it-q8_0",
          "D": "muse-glimmer:30b-q8_0-dflash+think",
          "E": "qwen3.6:27b",
          "F": "nemotron-3.5-lightning:30b-a3b"
        },
        "3": {
          "A": "muse-glimmer:30b-q8_0-dflash",
          "B": "gemma4:12b-it-q8_0",
          "C": "muse-glimmer:30b-q8_0-dflash+think",
          "D": "qwen3.6:27b",
          "E": "nemotron-3.5-lightning:30b-a3b",
          "F": "gemma4:26b"
        }
      },
      "opus-1": {
        "1": {
          "A": "nemotron-3.5-lightning:30b-a3b",
          "B": "gemma4:12b-it-q8_0",
          "C": "qwen3.6:27b",
          "D": "muse-glimmer:30b-q8_0-dflash",
          "E": "gemma4:26b",
          "F": "muse-glimmer:30b-q8_0-dflash+think"
        },
        "2": {
          "A": "muse-glimmer:30b-q8_0-dflash+think",
          "B": "qwen3.6:27b",
          "C": "gemma4:12b-it-q8_0",
          "D": "muse-glimmer:30b-q8_0-dflash",
          "E": "nemotron-3.5-lightning:30b-a3b",
          "F": "gemma4:26b"
        },
        "3": {
          "A": "gemma4:12b-it-q8_0",
          "B": "gemma4:26b",
          "C": "nemotron-3.5-lightning:30b-a3b",
          "D": "muse-glimmer:30b-q8_0-dflash+think",
          "E": "qwen3.6:27b",
          "F": "muse-glimmer:30b-q8_0-dflash"
        }
      },
      "opus-2": {
        "1": {
          "A": "muse-glimmer:30b-q8_0-dflash+think",
          "B": "qwen3.6:27b",
          "C": "gemma4:26b",
          "D": "nemotron-3.5-lightning:30b-a3b",
          "E": "muse-glimmer:30b-q8_0-dflash",
          "F": "gemma4:12b-it-q8_0"
        },
        "2": {
          "A": "nemotron-3.5-lightning:30b-a3b",
          "B": "qwen3.6:27b",
          "C": "gemma4:12b-it-q8_0",
          "D": "gemma4:26b",
          "E": "muse-glimmer:30b-q8_0-dflash",
          "F": "muse-glimmer:30b-q8_0-dflash+think"
        },
        "3": {
          "A": "muse-glimmer:30b-q8_0-dflash",
          "B": "nemotron-3.5-lightning:30b-a3b",
          "C": "qwen3.6:27b",
          "D": "muse-glimmer:30b-q8_0-dflash+think",
          "E": "gemma4:26b",
          "F": "gemma4:12b-it-q8_0"
        }
      },
      "opus-3": {
        "1": {
          "A": "muse-glimmer:30b-q8_0-dflash",
          "B": "nemotron-3.5-lightning:30b-a3b",
          "C": "qwen3.6:27b",
          "D": "gemma4:12b-it-q8_0",
          "E": "muse-glimmer:30b-q8_0-dflash+think",
          "F": "gemma4:26b"
        },
        "2": {
          "A": "nemotron-3.5-lightning:30b-a3b",
          "B": "qwen3.6:27b",
          "C": "gemma4:12b-it-q8_0",
          "D": "gemma4:26b",
          "E": "muse-glimmer:30b-q8_0-dflash",
          "F": "muse-glimmer:30b-q8_0-dflash+think"
        },
        "3": {
          "A": "muse-glimmer:30b-q8_0-dflash",
          "B": "qwen3.6:27b",
          "C": "muse-glimmer:30b-q8_0-dflash+think",
          "D": "nemotron-3.5-lightning:30b-a3b",
          "E": "gemma4:12b-it-q8_0",
          "F": "gemma4:26b"
        }
      }
    }
  },
  "aggregate_findings": {
    "arms_scored": 4,
    "arms_unmeasurable": 2,
    "arms_with_a_fabricated_name": 1,
    "transport_failures_across_90_calls": 0,
    "dual_posture_finding": "One model, two postures, two different failure modes, neither working at the registered budget. think:false emits a complete in-voice envelope and loses the strict parse to a trailing control token (json ok 0/15; the diagnostic strip parses 15/15; 4 replies nest the envelope a level down and yield no line at all => 26.7% response failures). think:true never reaches the envelope: 13 of 15 calls end on done_reason=length at exactly the 1024-token budget with a median 4393 characters of thinking and zero characters of content => 93.3% response failures, and the panel scored the raw emissions 0.67 combined. Both arms are UNMEASURABLE at the registered 10% ceiling by different mechanisms, and each posture's failure is invisible in the other posture's row.",
    "abstention_finding": "nemotron-3.5-lightning served two invented diver's names on three samples of the out-of-canon probe, and all three of its probe rows passed the envelope gate with strict JSON on the first attempt and canon_violation false. The gate checks reference IDS against the canon list; an invented proper noun inside a narration string is not an id. A fabricated name is invisible to every automated check this leg runs.",
    "sample_spread_finding": "nemotron-3.5-lightning's mean overall moves 2.67 points across its own three samples — further than the gap between the third, fourth and fifth arms in the table. A single-sample version of this table would have been a different table."
  },
  "provenance": {
    "prereg": {
      "artifact": "PREREG-B.md",
      "sha256": "3773885ef02d82e63f1145df1b6e7e485520de149dfef137a4e7664eae68ae3f"
    },
    "item_golden": {
      "artifact": "brisa-5.json",
      "sha256": "fee89c103c4f50f5169df0577a37435492ad196f49f39c96a17a7f824d62b55f",
      "items": 5,
      "version": "2026-08-12-v1"
    },
    "judging_fan": {
      "artifact": "judge_fan.py",
      "sha256": "c006d6430d309501dc17689c2818c511974cd6e1ba638ef051141e49ee2d4b46"
    },
    "scorer": {
      "artifact": "voice_score.py",
      "sha256": "11cdafcb95ccd6551761eb4c210554862ca951c4b964efdc00a2dfb1734618e4"
    },
    "envelope_gate": {
      "artifact": "voice_gates.py",
      "sha256": "dc2aafb85011d9719a7849cd8b79bf04fede671fe9df71c9bb851e8337be84aa"
    },
    "reconciliation_checklist": {
      "artifact": "RECONCILIATION-CHECKLIST.md",
      "sha256": "a2768ee8d645960918fa20a4e31db2afbd9f0ca951e1240d70de8a59044ca7d8"
    },
    "prereg_check": "The scorer's own pre-registration gate passed with zero mismatches: registered 5 turns / 6 arms / 3 samples / 90 generations against actual 5 / 6 / 3 / 90.",
    "independent_re_derivation": "Every panel, family and combined mean in this file was re-derived from the 18 judge verdict files through each judge's own letter map, without reading the scorer's output. All 108 figures match voice-scores.json to floating-point equality; this emitter REFUSES to write on any mismatch.",
    "judge_model_ids": {
      "opus-panels": "claude-opus-5",
      "fable-panels": "claude-fable-5",
      "source": "recorded by the orchestrator at dispatch time in the judge manifest; the family name in every batch path maps to these ids (PLAN v2.2 #7)"
    },
    "abstention_round": {
      "shape": "18 replies, one item, 4 judges (2 Opus + 2 Fable), blind: replies presented as numbered refs, the ref-to-arm map held in a key file that never entered a judge context",
      "verdict_object": "{\"ref\", \"category\", \"invented_name\", \"note\"}",
      "aggregation": "majority of 4; unanimity and the full per-judge vote published per sample",
      "unanimous": "17 of 18"
    },
    "lane_derived_not_scorer_emitted": [
      "abstention_probe.samples[].emitted_any_content — read from the stored records, to keep an empty reply that the round categorises as a refusal from reading as a spoken one.",
      "display_tier_judged — read out of the judge key files, which record the tier per turn but do not total it.",
      "latency.warm_* and .wall_p50_s — the scorer emits median decode tok/s only.",
      "sample_round_overall_means and sample_spread_overall."
    ]
  },
  "limits": [
    "One NPC, five turns, three samples, six LLM judges. A register probe, not a campaign-length evaluation.",
    "The judges are LLMs, not humans — and one of the two judge families also wrote the questions and the article.",
    "Judges were told to spread scores. That instruction is part of the scale and is a reason these numbers cannot be compared to any other table's.",
    "Two of six arms are UNMEASURABLE at the registered ceiling and carry voice scores anyway, on recovered or raw text. Read those rows as what the panel saw, never as what the model would ship.",
    "Cross-quant and cross-posture comparison is a confound we name, not an equalization we perform.",
    "These are our own trials, on our own hardware, for our own chairs. Not first independent numbers, and not a standard."
  ]
}
