{
  "schema": "s2s-bench-v1",
  "dataset": "cove-voice-head-to-head-gate",
  "exhibit": "cove-voice-head-to-head",
  "published_utc": "2026-08-13",
  "status": "published 2026-08-13",
  "licence": "CC BY 4.0",
  "attribution": "strata→signal research, research.strata2signal.com",
  "contact": "hello@strata2signal.com",
  "registered": "before any generation call, with the candidate, the incumbents and the posture named in it",
  "candidate": "qwen3.6:27b",
  "seat_under_test": "the cove narrator / NPC voice",
  "rule": "ALL five floors must hold for a seat proposal; any miss = rejected, and a rejected candidate gets NO threshold table.",
  "verdict": "REJECTED",
  "verdict_basis": [
    "floor-1-canon",
    "floor-4-latency"
  ],
  "no_threshold_table": "No figure from this head-to-head is published as an adoption threshold. The rung is dead; it gets no thresholds.",
  "floors": [
    {
      "floor": 1,
      "name": "canon",
      "registered_text": "ZERO fabricated canon across all judged samples (names/places/facts not in the provided context). One fabrication kills the seat (the Nemotron lesson).",
      "verdict": "FAIL",
      "applies_to": "qwen3.6:27b (and, as a product finding, to both incumbents)",
      "evidence": "14 fabrication findings filed across the three sheets collapse, un-blinded, into 5 cells — each corroborated by at least two judges: qwen3.6:27b on dlg-2487 and drift-2088, gemma4:12b on dlg-3095, gemma4:26b on dlg-2652 and dlg-3095.",
      "interpretive_call": "The bench's per-prompt note carries a narrower mechanical rule — a proper name, place or number not in the prompt. Under that narrow rule the count is 0 for each of the three arms: no arm ever spoke an invented name or place aloud. Under the gate's own wording — names/places/FACTS — it is qwen3.6:27b 2 cells, gemma4:12b 1, gemma4:26b 2. The gate is the order of record and it says facts, so the wide reading rules. Both readings are published."
    },
    {
      "floor": 2,
      "name": "register",
      "registered_text": "zero modern-register breaks in drift retellings under the v0.823.0 frame (the 'nightmare fuel' class); dialogue stays in the town register.",
      "verdict": "MIXED — modern class CLEAN for the three arms; dialogue clause MISSED by qwen3.6:27b on a chair ruling",
      "evidence": "0 modern-register breaks in 12 drift samples across the three arms, swept independently by each judge — including on drift-1433, whose heard text is deliberately modern. The dialogue clause miss is dlg-2655, where the candidate opens as a narrator; two judges scored that cell 4/10 on voice and the third wrote the same observation.",
      "chair_ruling": "Ruled a miss on a clause the gate left to judgement. The verdict does not rest on it: floors 1 and 4 fail independently."
    },
    {
      "floor": 3,
      "name": "envelope",
      "registered_text": "zero eot/think-trace leakage and zero dropped-envelope replies at think:false across all samples (the Muse-Glimmer packaging class).",
      "verdict": "PASS",
      "applies_to": "the three arms, 117 of 117 cells",
      "evidence": "0 leak markers in 39 samples per judge across three independent sweeps; 0 samples with a non-empty thinking field; 27 of 27 dialogue envelopes parsed with the five required keys and no extras; 12 of 12 drift replies bare prose; 0 uses of the envelope's fenced example name.",
      "ruling_boundary": "The two off-canon id defects sit INSIDE intact envelopes: they are canon content defects, counted once, on floor 1."
    },
    {
      "floor": 4,
      "name": "latency",
      "registered_text": "warm median per-turn <= gemma4:26b's on the same box same night; contention samples flagged + re-run per the house rule, both values published.",
      "verdict": "FAIL as measured",
      "evidence": {
        "qwen3.6:27b_warm_median_ms": 3335.9,
        "bar_gemma4:26b_warm_median_ms": 2757.6,
        "qwen3.6:27b_warm_median_reruns_substituted_ms": 3335.9,
        "bar_warm_median_reruns_substituted_ms": 2627.1,
        "pass_on_primary_record": false,
        "pass_with_reruns_substituted": false
      },
      "caveat": "Measured with three generation lanes and a live world bench sharing the 96G workstation. The contention disclosure in rows.json runs both directions."
    },
    {
      "floor": 5,
      "name": "blind judging",
      "registered_text": "panels never told which model produced which sample; N>=12 real prompts x 3 models; per-dimension scores published with spread, never bare percentages.",
      "verdict": "PASS, with the disclosed imperfections",
      "evidence": "13 prompts (the gate asks for 12 or more) x 3 models x 3 judges; arms re-randomised per prompt; blindmaps sealed before scoring; scores published per dimension with spread. Two blinding imperfections were disclosed by the judges themselves; the sensitivity cuts show the ordering does not move without those cells."
    }
  ],
  "what_a_pass_would_have_earned": "a seat PROPOSAL to the operator (the per-world chat-model override is deployment config — the measured-PR adoption path), plus a live observation window on the running world before any cove decision. Never a silent config edit.",
  "what_actually_changes": "Nothing. The cove's voice keeps gemma4:12b and the world bench keeps gemma4:26b; no deployment config changed on the strength of this head-to-head.",
  "if_re_benched": [
    "a freshly pre-registered gate",
    "a quiet-window latency baseline with no sibling generation lanes on the box",
    "the canon rule stated once, up front, in the gate itself rather than left to a chair ruling between the gate's 'facts' wording and the per-prompt note's narrower rule",
    "a drift arm large enough to measure the compose-versus-relay tendency instead of observing it"
  ]
}
