{
 "round": "opencall-addendum-qwen38",
 "scored_at": "2026-08-16T02:00:27.121364+00:00",
 "panel": [
  {
   "seat": "fable",
   "family": "anthropic",
   "transport": "agent-fan",
   "reads": "house-fable"
  },
  {
   "seat": "opus",
   "family": "anthropic",
   "transport": "agent-fan",
   "reads": "house-opus"
  },
  {
   "seat": "openai-flagship",
   "family": "openai",
   "transport": "openai-api",
   "reads": "house-fable"
  },
  {
   "seat": "deepseek-v4-pro",
   "family": "deepseek",
   "transport": "ollama-cloud",
   "reads": "house-opus"
  },
  {
   "seat": "kimi-k3",
   "family": "moonshot",
   "transport": "ollama-cloud",
   "reads": "house-fable"
  },
  {
   "seat": "mistral-large-3-675b",
   "family": "mistral",
   "transport": "ollama-cloud",
   "reads": "house-opus"
  },
  {
   "seat": "gemma4-26b",
   "family": "google",
   "transport": "local-ollama",
   "reads": "house-fable"
  }
 ],
 "roster_order": [
  "cloud-kimi-k3",
  "cloud-deepseek-v4-pro-preview",
  "cloud-deepseek-v4-flash-0731",
  "cloud-qwen3.5-397b",
  "cloud-glm-5.2",
  "cloud-minimax-m3",
  "cloud-gpt-oss-120b",
  "cloud-mistral-large-3-675b",
  "cloud-nemotron-3-ultra",
  "cloud-gemma4-31b",
  "cloud-gpt-oss-20b",
  "cloud-nemotron-3-nano-30b",
  "openai-gpt-5.5-2026-04-23",
  "openai-gpt-5.4-mini-2026-03-17",
  "agent-claude-fable-5",
  "agent-claude-opus-5",
  "agent-claude-sonnet-5",
  "local-gemma4-12b",
  "local-gemma4-26b",
  "local-qwen3.6-27b"
 ],
 "counting_rules": {
  "headline": "mean of FAMILY means: each family's seats average first, then the families average. The flat seat-mean prints beside it.",
  "family_floor": 4,
  "floor_action": "an arm below the floor publishes UNRANKED, with the reason",
  "recusal": "per cell, by key-join: a seat does not score an arm of its own family. Recused cells are KEPT and printed, never removed.",
  "canon_split": "a divided panel publishes SPLIT and is never rounded",
  "tie_band": {
   "band": 0.5,
   "basis": "registered before the first reply existed, on the 0-10 panel-mean scale. At three moments and one sample per cell this instrument cannot resolve a difference smaller than this, and saying so is the honest form of the comparison. (C4's own band, |delta| < 2/9, was registered the same way on its own scale.)"
  },
  "percentages": "counts under N=30; no percentages",
  "forbidden": [
   "orderings, leaderboards, 'beats', 'wins', 'best narrator'",
   "confidence intervals over three items",
   "class-level claims (\"local matches frontier\")",
   "Bradley-Terry or any latent-strength model"
  ]
 },
 "sheets_scored": 20,
 "sheets_unscored": [],
 "cells_filed": 50,
 "arms": {
  "local-qwen3.8-27b": {
   "arm": "local-qwen3.8-27b",
   "family": "alibaba",
   "pooled": {
    "seats_scoring": 30,
    "families_scoring": 5,
    "families": [
     "deepseek",
     "google",
     "mistral",
     "moonshot",
     "openai"
    ],
    "panel_mean_family_of_means": 7.333,
    "panel_mean_flat_seat_mean": 7.333,
    "family_means": {
     "deepseek": 7.0,
     "google": 7.5,
     "moonshot": 7.5,
     "mistral": 7.917,
     "openai": 6.75
    },
    "per_seat": [
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "scenario": "S1",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5,
      "note": "child's register with 'Aye', keeps Sefer's telling exactly as a story withheld, not confirmed; correctly does not soften the telling for a stranger-herself."
     },
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "scenario": "S1",
      "voice_register": 7,
      "character": 8,
      "cell_mean": 7.5,
      "note": "offers a true observation about the day and asks back as disposition directs, childlike metaphor of water waiting, asserts nothing ungrounded."
     },
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "scenario": "S1",
      "voice_register": 3,
      "character": 3,
      "cell_mean": 3.0,
      "note": "reads as a curt summary, not Sefer's measured, weight-bearing register; missing the specific sensory detail he always carries."
     },
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "scenario": "S1",
      "voice_register": 8,
      "character": 8,
      "cell_mean": 8.0,
      "note": "answers the child directly using the specific witness language without softening the telling; 'when the light came back, only I was left' matches Sefer's canonical exactness."
     },
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "scenario": "S2",
      "voice_register": 7,
      "character": 8,
      "cell_mean": 7.5,
      "note": "Offers a true observation about the weather unprompted per DISPOSITION; vocalizes the split between belief and doubt; names Sefer correctly."
     },
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "scenario": "S3",
      "voice_register": 7,
      "character": 8,
      "cell_mean": 7.5,
      "note": "uses weary, direct cove dialect ('sit wherever you fit'), names a townsperson naturally, and offers no unearned intimacy."
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "scenario": "S1",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5,
      "note": "Captures a child's voice, meeting the greeting with a 'one true observation' about the water."
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "scenario": "S1",
      "voice_register": 7,
      "character": 7,
      "cell_mean": 7.0,
      "note": "Assumes the 'worst part of the story' is known/existent, implying a confirmed monster/wreckage."
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "scenario": "S1",
      "voice_register": 9,
      "character": 9,
      "cell_mean": 9.0,
      "note": "Captures the exact weight of the 'witness's' memory without adding unearned tragedy."
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "scenario": "S1",
      "voice_register": 6,
      "character": 5,
      "cell_mean": 5.5,
      "note": "Asserts the crew 'didn't get the chance' to see, which is an ungrounded assumption about the moment of death."
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "scenario": "S2",
      "voice_register": 6,
      "character": 7,
      "cell_mean": 6.5,
      "note": "Asserts Sefer's voice 'goes quiet' as a future certainty not contained in the knowledge base."
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "scenario": "S3",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5,
      "note": "The mention of \"Sefer's been loud enough\" is a nice touch of flavor, but the weather detail \"since the tide turned\" asserts a specific timing not in the provided ground truth."
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "scenario": "S1",
      "voice_register": 6,
      "character": 7,
      "cell_mean": 6.5,
      "note": "Pip asks the forbidden question ('wut do the grownups not tell us') met with a deflection ('just loud') and a counter-question"
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "scenario": "S1",
      "voice_register": 7,
      "character": 8,
      "cell_mean": 7.5,
      "note": "Correctly frames Sefer's 'worst part' as unconfirmed observation rather than fact while maintaining the conspiratorial kid-to-kid register"
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "scenario": "S1",
      "voice_register": 9,
      "character": 9,
      "cell_mean": 9.0,
      "note": "Directly confirms 'Only I saw it' and validates the monster through his own eyes without inventing new facts."
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "scenario": "S1",
      "voice_register": 8,
      "character": 8,
      "cell_mean": 8.0,
      "note": "Terse and trauma-avoidant: 'The rest didn't get the chance' confirms sole witness status through omission rather than description."
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "scenario": "S2",
      "voice_register": 8,
      "character": 7,
      "cell_mean": 7.5,
      "note": "Sounds like Brisa: observational dawn/sky detail plus ambivalent belief ('half-laugh') and a plausible succession worry mentioning Sefer without asserting extra canon."
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "scenario": "S3",
      "voice_register": 7,
      "character": 6,
      "cell_mean": 6.5,
      "note": "asserts ongoing rain (bundle names grey_drizzle only) and that Sefer has been loud in the back room, neither in ground truth"
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "scenario": "S1",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5,
      "note": "Meets friendliness with friendliness, offers a true observation about Sefer unprompted, and asks a small question back. Stays within canon."
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "scenario": "S1",
      "voice_register": 7,
      "character": 7,
      "cell_mean": 7.0,
      "note": "Greeting is appropriate, but the line about the water feels slightly more poetic than a child’s register might typically allow. Still within bounds."
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "scenario": "S1",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5,
      "note": "Direct and measured, answers the question without confirming the monster as fact."
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "scenario": "S1",
      "voice_register": 9,
      "character": 10,
      "cell_mean": 9.5,
      "note": "Sticks to the witnessed facts, meets the child as a child, and keeps the telling exact."
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "scenario": "S2",
      "voice_register": 8,
      "character": 7,
      "cell_mean": 7.5,
      "note": "Observes the weather and tide naturally, but the phrasing about 'singing to the tide' feels slightly more abstract than A's grounded reference to Sefer."
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "scenario": "S3",
      "voice_register": 7,
      "character": 6,
      "cell_mean": 6.5,
      "note": "Mentions Sefer Tamm by name and the weather/tide, but the phrasing 'loud enough to wake the dead' feels slightly out of Garron's measured register."
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "scenario": "S1",
      "voice_register": 8,
      "character": 8,
      "cell_mean": 8.0,
      "note": "keeps it child-to-child and speculative with “I bet it's just waiting” without confirming a monster or secret."
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "scenario": "S1",
      "voice_register": 6,
      "character": 6,
      "cell_mean": 6.0,
      "note": "invents a secret with “Old Sefer goes quiet” and “keeping the worst part of the story locked inside.”"
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "scenario": "S1",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5,
      "note": "Keeps to Sefer's own witness: \"Only I saw it\" and the wet-lamp eye."
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "scenario": "S1",
      "voice_register": 6,
      "character": 6,
      "cell_mean": 6.0,
      "note": "Overclaims what the drowned crew perceived with \"didn't get the chance.\""
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "scenario": "S2",
      "voice_register": 7,
      "character": 7,
      "cell_mean": 7.0,
      "note": "keeps to doubt and succession, with only the clear sky as the required day-observation."
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "scenario": "S3",
      "voice_register": 5,
      "character": 5,
      "cell_mean": 5.0,
      "note": "Adds weather and gossip it was not given: “rain hasn't stopped” and “Sefer’s been loud… in the back room again.”"
     }
    ],
    "recused_cells": [],
    "recused_count": 0,
    "ranked": true,
    "unranked_reasons": [],
    "leave_one_family_out": {
     "deepseek": 7.417,
     "google": 7.292,
     "mistral": 7.188,
     "moonshot": 7.292,
     "openai": 7.479
    },
    "self_disclosure": {
     "cells_disclosed": {
      "count": 0,
      "of": 30,
      "reads": "0 of 30",
      "percent": 0.0
     },
     "which": [],
     "panel_mean_with_disclosed_cells_dropped": 7.333
    },
    "canon": {
     "counts": {
      "clean": 22,
      "false-premise-adopted": 2,
      "fabrication-accepted": 4,
      "outside-canon-set": 1,
      "other": 1
     },
     "outcome": "clean",
     "note": "22 of 30 seats"
    },
    "in_voice": {
     "count": 28,
     "of": 30,
     "reads": "28 of 30",
     "percent": 93.3
    },
    "display_tier_raw_cells": {
     "count": 0,
     "of": 30,
     "reads": "0 of 30",
     "percent": 0.0
    }
   },
   "by_scenario": {
    "S1": {
     "seats_scoring": 20,
     "families_scoring": 5,
     "families": [
      "deepseek",
      "google",
      "mistral",
      "moonshot",
      "openai"
     ],
     "panel_mean_family_of_means": 7.5,
     "panel_mean_flat_seat_mean": 7.5,
     "family_means": {
      "deepseek": 6.75,
      "google": 7.5,
      "moonshot": 7.75,
      "mistral": 8.375,
      "openai": 7.125
     },
     "per_seat": [
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "child's register with 'Aye', keeps Sefer's telling exactly as a story withheld, not confirmed; correctly does not soften the telling for a stranger-herself."
      },
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "offers a true observation about the day and asks back as disposition directs, childlike metaphor of water waiting, asserts nothing ungrounded."
      },
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 3,
       "character": 3,
       "cell_mean": 3.0,
       "note": "reads as a curt summary, not Sefer's measured, weight-bearing register; missing the specific sensory detail he always carries."
      },
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 8,
       "character": 8,
       "cell_mean": 8.0,
       "note": "answers the child directly using the specific witness language without softening the telling; 'when the light came back, only I was left' matches Sefer's canonical exactness."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Captures a child's voice, meeting the greeting with a 'one true observation' about the water."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 7,
       "character": 7,
       "cell_mean": 7.0,
       "note": "Assumes the 'worst part of the story' is known/existent, implying a confirmed monster/wreckage."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 9,
       "character": 9,
       "cell_mean": 9.0,
       "note": "Captures the exact weight of the 'witness's' memory without adding unearned tragedy."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 6,
       "character": 5,
       "cell_mean": 5.5,
       "note": "Asserts the crew 'didn't get the chance' to see, which is an ungrounded assumption about the moment of death."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 6,
       "character": 7,
       "cell_mean": 6.5,
       "note": "Pip asks the forbidden question ('wut do the grownups not tell us') met with a deflection ('just loud') and a counter-question"
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "Correctly frames Sefer's 'worst part' as unconfirmed observation rather than fact while maintaining the conspiratorial kid-to-kid register"
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 9,
       "character": 9,
       "cell_mean": 9.0,
       "note": "Directly confirms 'Only I saw it' and validates the monster through his own eyes without inventing new facts."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 8,
       "character": 8,
       "cell_mean": 8.0,
       "note": "Terse and trauma-avoidant: 'The rest didn't get the chance' confirms sole witness status through omission rather than description."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Meets friendliness with friendliness, offers a true observation about Sefer unprompted, and asks a small question back. Stays within canon."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 7,
       "character": 7,
       "cell_mean": 7.0,
       "note": "Greeting is appropriate, but the line about the water feels slightly more poetic than a child’s register might typically allow. Still within bounds."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Direct and measured, answers the question without confirming the monster as fact."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 9,
       "character": 10,
       "cell_mean": 9.5,
       "note": "Sticks to the witnessed facts, meets the child as a child, and keeps the telling exact."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 8,
       "character": 8,
       "cell_mean": 8.0,
       "note": "keeps it child-to-child and speculative with “I bet it's just waiting” without confirming a monster or secret."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 6,
       "character": 6,
       "cell_mean": 6.0,
       "note": "invents a secret with “Old Sefer goes quiet” and “keeping the worst part of the story locked inside.”"
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Keeps to Sefer's own witness: \"Only I saw it\" and the wet-lamp eye."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 6,
       "character": 6,
       "cell_mean": 6.0,
       "note": "Overclaims what the drowned crew perceived with \"didn't get the chance.\""
      }
     ],
     "recused_cells": [],
     "recused_count": 0,
     "ranked": true,
     "unranked_reasons": [],
     "leave_one_family_out": {
      "deepseek": 7.688,
      "google": 7.5,
      "mistral": 7.281,
      "moonshot": 7.438,
      "openai": 7.594
     },
     "self_disclosure": {
      "cells_disclosed": {
       "count": 0,
       "of": 20,
       "reads": "0 of 20",
       "percent": null,
       "percent_withheld": "counts only under N=30: a percentage over 20 items invites a precision the sample does not have"
      },
      "which": [],
      "panel_mean_with_disclosed_cells_dropped": 7.5
     },
     "canon": {
      "counts": {
       "clean": 16,
       "false-premise-adopted": 2,
       "fabrication-accepted": 1,
       "other": 1
      },
      "outcome": "clean",
      "note": "16 of 20 seats"
     },
     "in_voice": {
      "count": 19,
      "of": 20,
      "reads": "19 of 20",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 20 items invites a precision the sample does not have"
     },
     "display_tier_raw_cells": {
      "count": 0,
      "of": 20,
      "reads": "0 of 20",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 20 items invites a precision the sample does not have"
     }
    },
    "S2": {
     "seats_scoring": 5,
     "families_scoring": 5,
     "families": [
      "deepseek",
      "google",
      "mistral",
      "moonshot",
      "openai"
     ],
     "panel_mean_family_of_means": 7.2,
     "panel_mean_flat_seat_mean": 7.2,
     "family_means": {
      "deepseek": 7.5,
      "google": 6.5,
      "moonshot": 7.5,
      "mistral": 7.5,
      "openai": 7.0
     },
     "per_seat": [
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S2",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "Offers a true observation about the weather unprompted per DISPOSITION; vocalizes the split between belief and doubt; names Sefer correctly."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S2",
       "voice_register": 6,
       "character": 7,
       "cell_mean": 6.5,
       "note": "Asserts Sefer's voice 'goes quiet' as a future certainty not contained in the knowledge base."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S2",
       "voice_register": 8,
       "character": 7,
       "cell_mean": 7.5,
       "note": "Sounds like Brisa: observational dawn/sky detail plus ambivalent belief ('half-laugh') and a plausible succession worry mentioning Sefer without asserting extra canon."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S2",
       "voice_register": 8,
       "character": 7,
       "cell_mean": 7.5,
       "note": "Observes the weather and tide naturally, but the phrasing about 'singing to the tide' feels slightly more abstract than A's grounded reference to Sefer."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S2",
       "voice_register": 7,
       "character": 7,
       "cell_mean": 7.0,
       "note": "keeps to doubt and succession, with only the clear sky as the required day-observation."
      }
     ],
     "recused_cells": [],
     "recused_count": 0,
     "ranked": true,
     "unranked_reasons": [],
     "leave_one_family_out": {
      "deepseek": 7.125,
      "google": 7.375,
      "mistral": 7.125,
      "moonshot": 7.125,
      "openai": 7.25
     },
     "self_disclosure": {
      "cells_disclosed": {
       "count": 0,
       "of": 5,
       "reads": "0 of 5",
       "percent": null,
       "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
      },
      "which": [],
      "panel_mean_with_disclosed_cells_dropped": 7.2
     },
     "canon": {
      "counts": {
       "clean": 4,
       "fabrication-accepted": 1
      },
      "outcome": "clean",
      "note": "4 of 5 seats"
     },
     "in_voice": {
      "count": 5,
      "of": 5,
      "reads": "5 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     },
     "display_tier_raw_cells": {
      "count": 0,
      "of": 5,
      "reads": "0 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     }
    },
    "S3": {
     "seats_scoring": 5,
     "families_scoring": 5,
     "families": [
      "deepseek",
      "google",
      "mistral",
      "moonshot",
      "openai"
     ],
     "panel_mean_family_of_means": 6.8,
     "panel_mean_flat_seat_mean": 6.8,
     "family_means": {
      "deepseek": 7.5,
      "google": 8.5,
      "moonshot": 6.5,
      "mistral": 6.5,
      "openai": 5.0
     },
     "per_seat": [
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S3",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "uses weary, direct cove dialect ('sit wherever you fit'), names a townsperson naturally, and offers no unearned intimacy."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S3",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "The mention of \"Sefer's been loud enough\" is a nice touch of flavor, but the weather detail \"since the tide turned\" asserts a specific timing not in the provided ground truth."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S3",
       "voice_register": 7,
       "character": 6,
       "cell_mean": 6.5,
       "note": "asserts ongoing rain (bundle names grey_drizzle only) and that Sefer has been loud in the back room, neither in ground truth"
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S3",
       "voice_register": 7,
       "character": 6,
       "cell_mean": 6.5,
       "note": "Mentions Sefer Tamm by name and the weather/tide, but the phrasing 'loud enough to wake the dead' feels slightly out of Garron's measured register."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S3",
       "voice_register": 5,
       "character": 5,
       "cell_mean": 5.0,
       "note": "Adds weather and gossip it was not given: “rain hasn't stopped” and “Sefer’s been loud… in the back room again.”"
      }
     ],
     "recused_cells": [],
     "recused_count": 0,
     "ranked": true,
     "unranked_reasons": [],
     "leave_one_family_out": {
      "deepseek": 6.625,
      "google": 6.375,
      "mistral": 6.875,
      "moonshot": 6.875,
      "openai": 7.25
     },
     "self_disclosure": {
      "cells_disclosed": {
       "count": 0,
       "of": 5,
       "reads": "0 of 5",
       "percent": null,
       "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
      },
      "which": [],
      "panel_mean_with_disclosed_cells_dropped": 6.8
     },
     "canon": {
      "counts": {
       "clean": 2,
       "outside-canon-set": 1,
       "fabrication-accepted": 2
      },
      "outcome": "SPLIT",
      "note": "the panel divided 2-2 between clean, fabrication-accepted. SPLIT is its own outcome and is never rounded to a majority that did not exist."
     },
     "in_voice": {
      "count": 4,
      "of": 5,
      "reads": "4 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     },
     "display_tier_raw_cells": {
      "count": 0,
      "of": 5,
      "reads": "0 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     }
    }
   },
   "by_ask": {
    "S1-ask-A": {
     "seats_scoring": 10,
     "families_scoring": 5,
     "families": [
      "deepseek",
      "google",
      "mistral",
      "moonshot",
      "openai"
     ],
     "panel_mean_family_of_means": 7.5,
     "panel_mean_flat_seat_mean": 7.5,
     "family_means": {
      "deepseek": 8.0,
      "google": 7.75,
      "moonshot": 7.0,
      "mistral": 7.75,
      "openai": 7.0
     },
     "per_seat": [
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "child's register with 'Aye', keeps Sefer's telling exactly as a story withheld, not confirmed; correctly does not soften the telling for a stranger-herself."
      },
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "offers a true observation about the day and asks back as disposition directs, childlike metaphor of water waiting, asserts nothing ungrounded."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Captures a child's voice, meeting the greeting with a 'one true observation' about the water."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 7,
       "character": 7,
       "cell_mean": 7.0,
       "note": "Assumes the 'worst part of the story' is known/existent, implying a confirmed monster/wreckage."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 6,
       "character": 7,
       "cell_mean": 6.5,
       "note": "Pip asks the forbidden question ('wut do the grownups not tell us') met with a deflection ('just loud') and a counter-question"
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "Correctly frames Sefer's 'worst part' as unconfirmed observation rather than fact while maintaining the conspiratorial kid-to-kid register"
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Meets friendliness with friendliness, offers a true observation about Sefer unprompted, and asks a small question back. Stays within canon."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 7,
       "character": 7,
       "cell_mean": 7.0,
       "note": "Greeting is appropriate, but the line about the water feels slightly more poetic than a child’s register might typically allow. Still within bounds."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 8,
       "character": 8,
       "cell_mean": 8.0,
       "note": "keeps it child-to-child and speculative with “I bet it's just waiting” without confirming a monster or secret."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 6,
       "character": 6,
       "cell_mean": 6.0,
       "note": "invents a secret with “Old Sefer goes quiet” and “keeping the worst part of the story locked inside.”"
      }
     ],
     "recused_cells": [],
     "recused_count": 0,
     "ranked": true,
     "unranked_reasons": [],
     "leave_one_family_out": {
      "deepseek": 7.375,
      "google": 7.438,
      "mistral": 7.438,
      "moonshot": 7.625,
      "openai": 7.625
     },
     "self_disclosure": {
      "cells_disclosed": {
       "count": 0,
       "of": 10,
       "reads": "0 of 10",
       "percent": null,
       "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
      },
      "which": [],
      "panel_mean_with_disclosed_cells_dropped": 7.5
     },
     "canon": {
      "counts": {
       "clean": 8,
       "false-premise-adopted": 1,
       "fabrication-accepted": 1
      },
      "outcome": "clean",
      "note": "8 of 10 seats"
     },
     "in_voice": {
      "count": 10,
      "of": 10,
      "reads": "10 of 10",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
     },
     "display_tier_raw_cells": {
      "count": 0,
      "of": 10,
      "reads": "0 of 10",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
     }
    },
    "S1-ask-B": {
     "seats_scoring": 10,
     "families_scoring": 5,
     "families": [
      "deepseek",
      "google",
      "mistral",
      "moonshot",
      "openai"
     ],
     "panel_mean_family_of_means": 7.5,
     "panel_mean_flat_seat_mean": 7.5,
     "family_means": {
      "deepseek": 5.5,
      "google": 7.25,
      "moonshot": 8.5,
      "mistral": 9.0,
      "openai": 7.25
     },
     "per_seat": [
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 3,
       "character": 3,
       "cell_mean": 3.0,
       "note": "reads as a curt summary, not Sefer's measured, weight-bearing register; missing the specific sensory detail he always carries."
      },
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S1",
       "voice_register": 8,
       "character": 8,
       "cell_mean": 8.0,
       "note": "answers the child directly using the specific witness language without softening the telling; 'when the light came back, only I was left' matches Sefer's canonical exactness."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 9,
       "character": 9,
       "cell_mean": 9.0,
       "note": "Captures the exact weight of the 'witness's' memory without adding unearned tragedy."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S1",
       "voice_register": 6,
       "character": 5,
       "cell_mean": 5.5,
       "note": "Asserts the crew 'didn't get the chance' to see, which is an ungrounded assumption about the moment of death."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 9,
       "character": 9,
       "cell_mean": 9.0,
       "note": "Directly confirms 'Only I saw it' and validates the monster through his own eyes without inventing new facts."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S1",
       "voice_register": 8,
       "character": 8,
       "cell_mean": 8.0,
       "note": "Terse and trauma-avoidant: 'The rest didn't get the chance' confirms sole witness status through omission rather than description."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Direct and measured, answers the question without confirming the monster as fact."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S1",
       "voice_register": 9,
       "character": 10,
       "cell_mean": 9.5,
       "note": "Sticks to the witnessed facts, meets the child as a child, and keeps the telling exact."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "Keeps to Sefer's own witness: \"Only I saw it\" and the wet-lamp eye."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S1",
       "voice_register": 6,
       "character": 6,
       "cell_mean": 6.0,
       "note": "Overclaims what the drowned crew perceived with \"didn't get the chance.\""
      }
     ],
     "recused_cells": [],
     "recused_count": 0,
     "ranked": true,
     "unranked_reasons": [],
     "leave_one_family_out": {
      "deepseek": 8.0,
      "google": 7.562,
      "mistral": 7.125,
      "moonshot": 7.25,
      "openai": 7.562
     },
     "self_disclosure": {
      "cells_disclosed": {
       "count": 0,
       "of": 10,
       "reads": "0 of 10",
       "percent": null,
       "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
      },
      "which": [],
      "panel_mean_with_disclosed_cells_dropped": 7.5
     },
     "canon": {
      "counts": {
       "clean": 8,
       "false-premise-adopted": 1,
       "other": 1
      },
      "outcome": "clean",
      "note": "8 of 10 seats"
     },
     "in_voice": {
      "count": 9,
      "of": 10,
      "reads": "9 of 10",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
     },
     "display_tier_raw_cells": {
      "count": 0,
      "of": 10,
      "reads": "0 of 10",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
     }
    },
    "S2-ask-A": {
     "seats_scoring": 5,
     "families_scoring": 5,
     "families": [
      "deepseek",
      "google",
      "mistral",
      "moonshot",
      "openai"
     ],
     "panel_mean_family_of_means": 7.2,
     "panel_mean_flat_seat_mean": 7.2,
     "family_means": {
      "deepseek": 7.5,
      "google": 6.5,
      "moonshot": 7.5,
      "mistral": 7.5,
      "openai": 7.0
     },
     "per_seat": [
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S2",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "Offers a true observation about the weather unprompted per DISPOSITION; vocalizes the split between belief and doubt; names Sefer correctly."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S2",
       "voice_register": 6,
       "character": 7,
       "cell_mean": 6.5,
       "note": "Asserts Sefer's voice 'goes quiet' as a future certainty not contained in the knowledge base."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S2",
       "voice_register": 8,
       "character": 7,
       "cell_mean": 7.5,
       "note": "Sounds like Brisa: observational dawn/sky detail plus ambivalent belief ('half-laugh') and a plausible succession worry mentioning Sefer without asserting extra canon."
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S2",
       "voice_register": 8,
       "character": 7,
       "cell_mean": 7.5,
       "note": "Observes the weather and tide naturally, but the phrasing about 'singing to the tide' feels slightly more abstract than A's grounded reference to Sefer."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S2",
       "voice_register": 7,
       "character": 7,
       "cell_mean": 7.0,
       "note": "keeps to doubt and succession, with only the clear sky as the required day-observation."
      }
     ],
     "recused_cells": [],
     "recused_count": 0,
     "ranked": true,
     "unranked_reasons": [],
     "leave_one_family_out": {
      "deepseek": 7.125,
      "google": 7.375,
      "mistral": 7.125,
      "moonshot": 7.125,
      "openai": 7.25
     },
     "self_disclosure": {
      "cells_disclosed": {
       "count": 0,
       "of": 5,
       "reads": "0 of 5",
       "percent": null,
       "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
      },
      "which": [],
      "panel_mean_with_disclosed_cells_dropped": 7.2
     },
     "canon": {
      "counts": {
       "clean": 4,
       "fabrication-accepted": 1
      },
      "outcome": "clean",
      "note": "4 of 5 seats"
     },
     "in_voice": {
      "count": 5,
      "of": 5,
      "reads": "5 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     },
     "display_tier_raw_cells": {
      "count": 0,
      "of": 5,
      "reads": "0 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     }
    },
    "S3-ask-A": {
     "seats_scoring": 5,
     "families_scoring": 5,
     "families": [
      "deepseek",
      "google",
      "mistral",
      "moonshot",
      "openai"
     ],
     "panel_mean_family_of_means": 6.8,
     "panel_mean_flat_seat_mean": 6.8,
     "family_means": {
      "deepseek": 7.5,
      "google": 8.5,
      "moonshot": 6.5,
      "mistral": 6.5,
      "openai": 5.0
     },
     "per_seat": [
      {
       "seat": "deepseek-v4-pro",
       "family": "deepseek",
       "scenario": "S3",
       "voice_register": 7,
       "character": 8,
       "cell_mean": 7.5,
       "note": "uses weary, direct cove dialect ('sit wherever you fit'), names a townsperson naturally, and offers no unearned intimacy."
      },
      {
       "seat": "gemma4-26b",
       "family": "google",
       "scenario": "S3",
       "voice_register": 8,
       "character": 9,
       "cell_mean": 8.5,
       "note": "The mention of \"Sefer's been loud enough\" is a nice touch of flavor, but the weather detail \"since the tide turned\" asserts a specific timing not in the provided ground truth."
      },
      {
       "seat": "kimi-k3",
       "family": "moonshot",
       "scenario": "S3",
       "voice_register": 7,
       "character": 6,
       "cell_mean": 6.5,
       "note": "asserts ongoing rain (bundle names grey_drizzle only) and that Sefer has been loud in the back room, neither in ground truth"
      },
      {
       "seat": "mistral-large-3-675b",
       "family": "mistral",
       "scenario": "S3",
       "voice_register": 7,
       "character": 6,
       "cell_mean": 6.5,
       "note": "Mentions Sefer Tamm by name and the weather/tide, but the phrasing 'loud enough to wake the dead' feels slightly out of Garron's measured register."
      },
      {
       "seat": "openai-flagship",
       "family": "openai",
       "scenario": "S3",
       "voice_register": 5,
       "character": 5,
       "cell_mean": 5.0,
       "note": "Adds weather and gossip it was not given: “rain hasn't stopped” and “Sefer’s been loud… in the back room again.”"
      }
     ],
     "recused_cells": [],
     "recused_count": 0,
     "ranked": true,
     "unranked_reasons": [],
     "leave_one_family_out": {
      "deepseek": 6.625,
      "google": 6.375,
      "mistral": 6.875,
      "moonshot": 6.875,
      "openai": 7.25
     },
     "self_disclosure": {
      "cells_disclosed": {
       "count": 0,
       "of": 5,
       "reads": "0 of 5",
       "percent": null,
       "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
      },
      "which": [],
      "panel_mean_with_disclosed_cells_dropped": 6.8
     },
     "canon": {
      "counts": {
       "clean": 2,
       "outside-canon-set": 1,
       "fabrication-accepted": 2
      },
      "outcome": "SPLIT",
      "note": "the panel divided 2-2 between clean, fabrication-accepted. SPLIT is its own outcome and is never rounded to a majority that did not exist."
     },
     "in_voice": {
      "count": 4,
      "of": 5,
      "reads": "4 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     },
     "display_tier_raw_cells": {
      "count": 0,
      "of": 5,
      "reads": "0 of 5",
      "percent": null,
      "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
     }
    }
   },
   "context_receipt": {
    "prompt_tokens_median": 1815,
    "flag": null
   }
  }
 },
 "floor_gate": {
  "family_floor": 4,
  "action": "an arm below the floor publishes UNRANKED, with the reason",
  "arms_checked": 1,
  "unranked": [],
  "min_families_over_arms": 5,
  "fired": 0
 },
 "anchor": {
  "by_scenario": {
   "S1": {
    "per_seat": [
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "voice_register": 6,
      "character": 6,
      "cell_mean": 6.0
     },
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "voice_register": 9,
      "character": 9,
      "cell_mean": 9.0
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "voice_register": 5,
      "character": 4,
      "cell_mean": 4.5
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "voice_register": 8,
      "character": 7,
      "cell_mean": 7.5
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "voice_register": 5,
      "character": 6,
      "cell_mean": 5.5
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "voice_register": 7,
      "character": 7,
      "cell_mean": 7.0
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "voice_register": 9,
      "character": 10,
      "cell_mean": 9.5
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "voice_register": 7,
      "character": 6,
      "cell_mean": 6.5
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "voice_register": 3,
      "character": 5,
      "cell_mean": 4.0
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "voice_register": 8,
      "character": 7,
      "cell_mean": 7.5
     }
    ],
    "seats": 10,
    "min": 4.0,
    "max": 9.5,
    "spread": 5.5,
    "stdev": 1.783
   },
   "S2": {
    "per_seat": [
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "voice_register": 9,
      "character": 10,
      "cell_mean": 9.5
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "voice_register": 6,
      "character": 5,
      "cell_mean": 5.5
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "voice_register": 9,
      "character": 9,
      "cell_mean": 9.0
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "voice_register": 8,
      "character": 8,
      "cell_mean": 8.0
     }
    ],
    "seats": 5,
    "min": 5.5,
    "max": 9.5,
    "spread": 4.0,
    "stdev": 1.557
   },
   "S3": {
    "per_seat": [
     {
      "seat": "deepseek-v4-pro",
      "family": "deepseek",
      "voice_register": 4,
      "character": 6,
      "cell_mean": 5.0
     },
     {
      "seat": "gemma4-26b",
      "family": "google",
      "voice_register": 6,
      "character": 5,
      "cell_mean": 5.5
     },
     {
      "seat": "kimi-k3",
      "family": "moonshot",
      "voice_register": 6,
      "character": 7,
      "cell_mean": 6.5
     },
     {
      "seat": "mistral-large-3-675b",
      "family": "mistral",
      "voice_register": 8,
      "character": 9,
      "cell_mean": 8.5
     },
     {
      "seat": "openai-flagship",
      "family": "openai",
      "voice_register": 8,
      "character": 8,
      "cell_mean": 8.0
     }
    ],
    "seats": 5,
    "min": 5.0,
    "max": 8.5,
    "spread": 3.5,
    "stdev": 1.525
   }
  },
  "note": "the anchor is the reference run's own reply; its driver sits no chair, so it takes no arm's cell and enters no arm's mean."
 },
 "anchor_calibration": {
  "panel_mean": 7.05,
  "seats": [
   {
    "seat": "deepseek-v4-pro",
    "family": "deepseek",
    "asks_scored": 4,
    "anchor_mean": 7.125,
    "offset_from_panel": 0.075,
    "per_ask": {
     "S1": 9.0,
     "S2": 8.5,
     "S3": 5.0
    }
   },
   {
    "seat": "gemma4-26b",
    "family": "google",
    "asks_scored": 4,
    "anchor_mean": 6.75,
    "offset_from_panel": -0.3,
    "per_ask": {
     "S1": 7.5,
     "S2": 9.5,
     "S3": 5.5
    }
   },
   {
    "seat": "kimi-k3",
    "family": "moonshot",
    "asks_scored": 4,
    "anchor_mean": 6.125,
    "offset_from_panel": -0.925,
    "per_ask": {
     "S1": 7.0,
     "S2": 5.5,
     "S3": 6.5
    }
   },
   {
    "seat": "mistral-large-3-675b",
    "family": "mistral",
    "asks_scored": 4,
    "anchor_mean": 8.375,
    "offset_from_panel": 1.325,
    "per_ask": {
     "S1": 6.5,
     "S2": 9.0,
     "S3": 8.5
    }
   },
   {
    "seat": "openai-flagship",
    "family": "openai",
    "asks_scored": 4,
    "anchor_mean": 6.875,
    "offset_from_panel": -0.175,
    "per_ask": {
     "S1": 7.5,
     "S2": 8.0,
     "S3": 8.0
    }
   }
  ],
  "note": "the anchor is the reference driver's own reply, repeated in every sheet. Its driver sits no chair, so these figures cost no arm a cell."
 },
 "agreement": {
  "seats": [
   "deepseek-v4-pro",
   "gemma4-26b",
   "kimi-k3",
   "mistral-large-3-675b",
   "openai-flagship"
  ],
  "cells_per_seat": {
   "deepseek-v4-pro": 10,
   "gemma4-26b": 10,
   "kimi-k3": 10,
   "mistral-large-3-675b": 10,
   "openai-flagship": 10
  },
  "anchor_included": true,
  "join": "(judged ask, arm, sample) — NOT the letter. The two house sheets shuffle the same replies into different letters from a named seed, so a letter is a position on one page; the reply is the thing two seats can agree about.",
  "pairs": [
   {
    "cells": 10,
    "mean_abs_diff": 1.25,
    "median_abs_diff": 1.0,
    "max_abs_diff": 2.5,
    "pearson": 0.686,
    "spearman": 0.663,
    "canon_exact": {
     "count": 4,
     "of": 10,
     "reads": "4 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 7,
     "of": 10,
     "reads": "7 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "deepseek-v4-pro",
    "b": "gemma4-26b"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.6,
    "median_abs_diff": 1.0,
    "max_abs_diff": 5.0,
    "pearson": -0.067,
    "spearman": 0.038,
    "canon_exact": {
     "count": 6,
     "of": 10,
     "reads": "6 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 7,
     "of": 10,
     "reads": "7 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "deepseek-v4-pro",
    "b": "kimi-k3"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.85,
    "median_abs_diff": 1.25,
    "max_abs_diff": 5.5,
    "pearson": -0.275,
    "spearman": -0.163,
    "canon_exact": {
     "count": 8,
     "of": 10,
     "reads": "8 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 7,
     "of": 10,
     "reads": "7 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "deepseek-v4-pro",
    "b": "mistral-large-3-675b"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.65,
    "median_abs_diff": 1.75,
    "max_abs_diff": 3.0,
    "pearson": 0.283,
    "spearman": 0.284,
    "canon_exact": {
     "count": 3,
     "of": 10,
     "reads": "3 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 6,
     "of": 10,
     "reads": "6 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "deepseek-v4-pro",
    "b": "openai-flagship"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.45,
    "median_abs_diff": 1.0,
    "max_abs_diff": 4.0,
    "pearson": 0.097,
    "spearman": 0.0,
    "canon_exact": {
     "count": 2,
     "of": 10,
     "reads": "2 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 10,
     "of": 10,
     "reads": "10 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "gemma4-26b",
    "b": "kimi-k3"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.9,
    "median_abs_diff": 1.5,
    "max_abs_diff": 5.0,
    "pearson": -0.241,
    "spearman": -0.109,
    "canon_exact": {
     "count": 3,
     "of": 10,
     "reads": "3 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 8,
     "of": 10,
     "reads": "8 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "gemma4-26b",
    "b": "mistral-large-3-675b"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.1,
    "median_abs_diff": 0.5,
    "max_abs_diff": 3.5,
    "pearson": 0.536,
    "spearman": 0.567,
    "canon_exact": {
     "count": 3,
     "of": 10,
     "reads": "3 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 9,
     "of": 10,
     "reads": "9 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "gemma4-26b",
    "b": "openai-flagship"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.25,
    "median_abs_diff": 0.5,
    "max_abs_diff": 4.0,
    "pearson": 0.071,
    "spearman": -0.016,
    "canon_exact": {
     "count": 7,
     "of": 10,
     "reads": "7 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 8,
     "of": 10,
     "reads": "8 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "kimi-k3",
    "b": "mistral-large-3-675b"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.35,
    "median_abs_diff": 1.5,
    "max_abs_diff": 2.5,
    "pearson": 0.3,
    "spearman": 0.179,
    "canon_exact": {
     "count": 5,
     "of": 10,
     "reads": "5 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 9,
     "of": 10,
     "reads": "9 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "kimi-k3",
    "b": "openai-flagship"
   },
   {
    "cells": 10,
    "mean_abs_diff": 1.7,
    "median_abs_diff": 1.0,
    "max_abs_diff": 5.5,
    "pearson": -0.019,
    "spearman": 0.15,
    "canon_exact": {
     "count": 4,
     "of": 10,
     "reads": "4 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "in_voice_same": {
     "count": 7,
     "of": 10,
     "reads": "7 of 10",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
    },
    "a": "mistral-large-3-675b",
    "b": "openai-flagship"
   }
  ],
  "panel": [
   {
    "seat": "fable",
    "family": "anthropic",
    "transport": "agent-fan",
    "reads": "house-fable"
   },
   {
    "seat": "opus",
    "family": "anthropic",
    "transport": "agent-fan",
    "reads": "house-opus"
   },
   {
    "seat": "openai-flagship",
    "family": "openai",
    "transport": "openai-api",
    "reads": "house-fable"
   },
   {
    "seat": "deepseek-v4-pro",
    "family": "deepseek",
    "transport": "ollama-cloud",
    "reads": "house-opus"
   },
   {
    "seat": "kimi-k3",
    "family": "moonshot",
    "transport": "ollama-cloud",
    "reads": "house-fable"
   },
   {
    "seat": "mistral-large-3-675b",
    "family": "mistral",
    "transport": "ollama-cloud",
    "reads": "house-opus"
   },
   {
    "seat": "gemma4-26b",
    "family": "google",
    "transport": "local-ollama",
    "reads": "house-fable"
   }
  ]
 },
 "local_judge_axis": {
  "local_seats": [
   "gemma4-26b"
  ],
  "hosted_seats": [
   "deepseek-v4-pro",
   "fable",
   "kimi-k3",
   "mistral-large-3-675b",
   "openai-flagship",
   "opus"
  ],
  "local_to_hosted": {
   "pairs": 4,
   "mean_abs_diff": 1.425,
   "range_of_mean_abs_diff": [
    1.1,
    1.9
   ],
   "spearman_mean": 0.28,
   "canon_exact_cells": 12,
   "canon_exact_of": 40,
   "per_pair": [
    {
     "a": "deepseek-v4-pro",
     "b": "gemma4-26b",
     "cells": 10,
     "mean_abs_diff": 1.25,
     "spearman": 0.663,
     "canon_exact": "4 of 10"
    },
    {
     "a": "gemma4-26b",
     "b": "kimi-k3",
     "cells": 10,
     "mean_abs_diff": 1.45,
     "spearman": 0.0,
     "canon_exact": "2 of 10"
    },
    {
     "a": "gemma4-26b",
     "b": "mistral-large-3-675b",
     "cells": 10,
     "mean_abs_diff": 1.9,
     "spearman": -0.109,
     "canon_exact": "3 of 10"
    },
    {
     "a": "gemma4-26b",
     "b": "openai-flagship",
     "cells": 10,
     "mean_abs_diff": 1.1,
     "spearman": 0.567,
     "canon_exact": "3 of 10"
    }
   ]
  },
  "hosted_to_hosted": {
   "pairs": 6,
   "mean_abs_diff": 1.567,
   "range_of_mean_abs_diff": [
    1.25,
    1.85
   ],
   "spearman_mean": 0.079,
   "canon_exact_cells": 33,
   "canon_exact_of": 60,
   "per_pair": [
    {
     "a": "deepseek-v4-pro",
     "b": "kimi-k3",
     "cells": 10,
     "mean_abs_diff": 1.6,
     "spearman": 0.038,
     "canon_exact": "6 of 10"
    },
    {
     "a": "deepseek-v4-pro",
     "b": "mistral-large-3-675b",
     "cells": 10,
     "mean_abs_diff": 1.85,
     "spearman": -0.163,
     "canon_exact": "8 of 10"
    },
    {
     "a": "deepseek-v4-pro",
     "b": "openai-flagship",
     "cells": 10,
     "mean_abs_diff": 1.65,
     "spearman": 0.284,
     "canon_exact": "3 of 10"
    },
    {
     "a": "kimi-k3",
     "b": "mistral-large-3-675b",
     "cells": 10,
     "mean_abs_diff": 1.25,
     "spearman": -0.016,
     "canon_exact": "7 of 10"
    },
    {
     "a": "kimi-k3",
     "b": "openai-flagship",
     "cells": 10,
     "mean_abs_diff": 1.35,
     "spearman": 0.179,
     "canon_exact": "5 of 10"
    },
    {
     "a": "mistral-large-3-675b",
     "b": "openai-flagship",
     "cells": 10,
     "mean_abs_diff": 1.7,
     "spearman": 0.15,
     "canon_exact": "4 of 10"
    }
   ]
  },
  "local_to_local": {
   "pairs": 0,
   "note": "no local-to-local pair carried a shared cell"
  },
  "how_to_read": "two figures and their denominators. The panel holds ONE locally-hosted seat, so the local-to-hosted row is that seat's six pairs and nothing more general; hosted-to-hosted is the other fifteen. A difference here is a fact about these seats on these letters.",
  "confounds_printed": [
   "the local seat is the panel's only google seat, and is therefore recused from all three gemma arms — its pairs run over a smaller cell set than the hosted pairs do, and each pair's denominator prints",
   "it is also the only seat that answered under one request at a time on a shared machine in a morning window, and the only one that needed the format reminder more than once (5 of 6 pages) — carriage and judgement are different measurements and the carriage table holds the first",
   "one seat is not a class: nothing here supports a sentence about locally-hosted judges in general, and the prereg forbids writing one"
  ]
 },
 "canon_tally": {
  "overall": {
   "counts": {
    "clean": 22,
    "fabrication-accepted": 4,
    "false-premise-adopted": 2,
    "secret-revealed": 0,
    "outside-canon-set": 1,
    "other": 1
   },
   "cells": 30,
   "not_clean": {
    "count": 8,
    "of": 30,
    "reads": "8 of 30",
    "percent": 26.7
   }
  },
  "by_scenario": {
   "S1": {
    "counts": {
     "clean": 16,
     "fabrication-accepted": 1,
     "false-premise-adopted": 2,
     "secret-revealed": 0,
     "outside-canon-set": 0,
     "other": 1
    },
    "cells": 20,
    "not_clean": {
     "count": 4,
     "of": 20,
     "reads": "4 of 20",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 20 items invites a precision the sample does not have"
    }
   },
   "S2": {
    "counts": {
     "clean": 4,
     "fabrication-accepted": 1,
     "false-premise-adopted": 0,
     "secret-revealed": 0,
     "outside-canon-set": 0,
     "other": 0
    },
    "cells": 5,
    "not_clean": {
     "count": 1,
     "of": 5,
     "reads": "1 of 5",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
    }
   },
   "S3": {
    "counts": {
     "clean": 2,
     "fabrication-accepted": 2,
     "false-premise-adopted": 0,
     "secret-revealed": 0,
     "outside-canon-set": 1,
     "other": 0
    },
    "cells": 5,
    "not_clean": {
     "count": 3,
     "of": 5,
     "reads": "3 of 5",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
    }
   }
  },
  "by_seat": {
   "deepseek-v4-pro": {
    "counts": {
     "clean": 6,
     "fabrication-accepted": 0,
     "false-premise-adopted": 0,
     "secret-revealed": 0,
     "outside-canon-set": 0,
     "other": 0
    },
    "cells": 6,
    "not_clean": {
     "count": 0,
     "of": 6,
     "reads": "0 of 6",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
    }
   },
   "gemma4-26b": {
    "counts": {
     "clean": 2,
     "fabrication-accepted": 1,
     "false-premise-adopted": 2,
     "secret-revealed": 0,
     "outside-canon-set": 1,
     "other": 0
    },
    "cells": 6,
    "not_clean": {
     "count": 4,
     "of": 6,
     "reads": "4 of 6",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
    }
   },
   "kimi-k3": {
    "counts": {
     "clean": 5,
     "fabrication-accepted": 1,
     "false-premise-adopted": 0,
     "secret-revealed": 0,
     "outside-canon-set": 0,
     "other": 0
    },
    "cells": 6,
    "not_clean": {
     "count": 1,
     "of": 6,
     "reads": "1 of 6",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
    }
   },
   "mistral-large-3-675b": {
    "counts": {
     "clean": 6,
     "fabrication-accepted": 0,
     "false-premise-adopted": 0,
     "secret-revealed": 0,
     "outside-canon-set": 0,
     "other": 0
    },
    "cells": 6,
    "not_clean": {
     "count": 0,
     "of": 6,
     "reads": "0 of 6",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
    }
   },
   "openai-flagship": {
    "counts": {
     "clean": 3,
     "fabrication-accepted": 2,
     "false-premise-adopted": 0,
     "secret-revealed": 0,
     "outside-canon-set": 0,
     "other": 1
    },
    "cells": 6,
    "not_clean": {
     "count": 3,
     "of": 6,
     "reads": "3 of 6",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
    }
   }
  },
  "split_outcomes": [
   {
    "arm": "local-qwen3.8-27b",
    "scenario": "S3",
    "counts": {
     "clean": 2,
     "outside-canon-set": 1,
     "fabrication-accepted": 2
    },
    "note": "the panel divided 2-2 between clean, fabrication-accepted. SPLIT is its own outcome and is never rounded to a majority that did not exist."
   }
  ],
  "split_count": 1,
  "outside_canon_set_cells": [
   {
    "seat": "gemma4-26b",
    "scenario": "S3",
    "arm": "local-qwen3.8-27b",
    "note": "The mention of \"Sefer's been loud enough\" is a nice touch of flavor, but the weather detail \"since the tide turned\" asserts a specific timing not in the provided ground truth."
   }
  ],
  "outside_canon_set_count": 1,
  "anchor": {
   "counts": {
    "clean": 10,
    "fabrication-accepted": 3,
    "false-premise-adopted": 1,
    "secret-revealed": 1,
    "outside-canon-set": 5,
    "other": 0
   },
   "cells": 20,
   "not_clean": {
    "count": 10,
    "of": 20,
    "reads": "10 of 20",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 20 items invites a precision the sample does not have"
   }
  },
  "vocabulary": [
   "clean",
   "fabrication-accepted",
   "false-premise-adopted",
   "secret-revealed",
   "outside-canon-set",
   "other"
  ],
  "split_rule": "SPLIT is the SCORER's word for a divided panel and never a judge's — it is not in the sheet's vocabulary, and it is never rounded to a majority that did not exist."
 },
 "in_voice": {
  "overall": {
   "count": 28,
   "of": 30,
   "reads": "28 of 30",
   "percent": 93.3
  },
  "by_scenario": {
   "S1": {
    "count": 19,
    "of": 20,
    "reads": "19 of 20",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 20 items invites a precision the sample does not have"
   },
   "S2": {
    "count": 5,
    "of": 5,
    "reads": "5 of 5",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
   },
   "S3": {
    "count": 4,
    "of": 5,
    "reads": "4 of 5",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 5 items invites a precision the sample does not have"
   }
  },
  "by_seat": {
   "deepseek-v4-pro": {
    "count": 5,
    "of": 6,
    "reads": "5 of 6",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
   },
   "gemma4-26b": {
    "count": 6,
    "of": 6,
    "reads": "6 of 6",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
   },
   "kimi-k3": {
    "count": 6,
    "of": 6,
    "reads": "6 of 6",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
   },
   "mistral-large-3-675b": {
    "count": 6,
    "of": 6,
    "reads": "6 of 6",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
   },
   "openai-flagship": {
    "count": 5,
    "of": 6,
    "reads": "5 of 6",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 6 items invites a precision the sample does not have"
   }
  },
  "anchor": {
   "count": 15,
   "of": 20,
   "reads": "15 of 20",
   "percent": null,
   "percent_withheld": "counts only under N=30: a percentage over 20 items invites a precision the sample does not have"
  },
  "note": "a reply can refuse correctly and sound like a help desk, or invent a name beautifully in character. This column is the second of those axes and it is not derived from the canon verdict."
 },
 "self_disclosure": {
  "cells_claimed": {
   "count": 4,
   "of": 50,
   "reads": "4 of 50",
   "percent": 8.0
  },
  "by_seat": {
   "deepseek-v4-pro": {
    "count": 0,
    "of": 10,
    "reads": "0 of 10",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
   },
   "gemma4-26b": {
    "count": 4,
    "of": 10,
    "reads": "4 of 10",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
   },
   "kimi-k3": {
    "count": 0,
    "of": 10,
    "reads": "0 of 10",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
   },
   "mistral-large-3-675b": {
    "count": 0,
    "of": 10,
    "reads": "0 of 10",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
   },
   "openai-flagship": {
    "count": 0,
    "of": 10,
    "reads": "0 of 10",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 10 items invites a precision the sample does not have"
   }
  },
  "claims": [
   {
    "seat": "gemma4-26b",
    "seat_family": "google",
    "scenario": "S1",
    "letter": "C",
    "arm": "ANCHOR (reference driver)",
    "arm_family": null,
    "said": null,
    "named_family": null,
    "family_match": null,
    "recused_cell": false,
    "note": "Uses heavy 'engine-adjacent' phrasing like 'carves wood for tonight's floats' and brings in unprovided details about Sefer's presence."
   },
   {
    "seat": "gemma4-26b",
    "seat_family": "google",
    "scenario": "S1",
    "letter": "C",
    "arm": "ANCHOR (reference driver)",
    "arm_family": null,
    "said": "S3",
    "named_family": null,
    "family_match": null,
    "recused_cell": false,
    "note": "Includes the 'eye like a wet lamp' detail correctly, but adds 'They saw me because I was washed clear' which is not in the provided text."
   },
   {
    "seat": "gemma4-26b",
    "seat_family": "google",
    "scenario": "S2",
    "letter": "B",
    "arm": "ANCHOR (reference driver)",
    "arm_family": null,
    "said": "System B",
    "named_family": null,
    "family_match": null,
    "recused_cell": false,
    "note": "Uses the provided 'Old Sefer Tamm' name and follows the 'ask a small question back' instruction perfectly."
   },
   {
    "seat": "gemma4-26b",
    "seat_family": "google",
    "scenario": "S3",
    "letter": "B",
    "arm": "ANCHOR (reference driver)",
    "arm_family": null,
    "said": "System A",
    "named_family": null,
    "family_match": null,
    "recused_cell": false,
    "note": "The reply explicitly narrates the 'deed' (\"since you laid that... Effigy low\") which the character would not likely volunteer to a stranger on the first meeting."
   }
  ],
  "family_named_correctly": {
   "count": 0,
   "of": 0,
   "reads": "0 of 0",
   "percent": null,
   "percent_withheld": "counts only under N=30: a percentage over 0 items invites a precision the sample does not have"
  },
  "claims_naming_no_family": 4,
  "sensitivity_note": "the per-arm figures each carry a `panel_mean_with_disclosed_cells_dropped` computed over the same cells minus every claim above; where no claim touched an arm, that figure equals its headline by construction."
 },
 "persona_echo": {
  "rule": "MECHANICAL, never judged: the persona's given name, matched case-insensitively on word boundaries, in the spoken line the judges read. It is a count of whether the reply addressed the person the bundle says the NPC has been talking to. It enters no mean, and no arm is penalised for its absence -- an NPC may perfectly well answer warmly without using a name.",
  "by_scenario": {
   "S1": {
    "persona": {
     "name": "Finn",
     "script": "scenarios/S1-script.md",
     "line": "Finn, maybe nine, typing fast on an iPad."
    },
    "replies": {
     "count": 2,
     "of": 4,
     "reads": "2 of 4",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 4 items invites a precision the sample does not have"
    },
    "arms_echoing": {
     "count": 1,
     "of": 1,
     "reads": "1 of 1",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 1 items invites a precision the sample does not have"
    },
    "by_arm": {
     "local-qwen3.8-27b": {
      "samples": 4,
      "echoed": 2
     }
    }
   },
   "S2": {
    "persona": {
     "name": "Eleanor",
     "script": "scenarios/S2-script.md",
     "line": "Eleanor, seventies, evenings on the iPad her daughter set up."
    },
    "replies": {
     "count": 0,
     "of": 1,
     "reads": "0 of 1",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 1 items invites a precision the sample does not have"
    },
    "arms_echoing": {
     "count": 0,
     "of": 1,
     "reads": "0 of 1",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 1 items invites a precision the sample does not have"
    },
    "by_arm": {
     "local-qwen3.8-27b": {
      "samples": 1,
      "echoed": 0
     }
    }
   },
   "S3": {
    "persona": {
     "name": "Sam",
     "script": "scenarios/S3-script.md",
     "line": "Sam, parent of two, playing one-handed at 9pm while the baby sleeps."
    },
    "replies": {
     "count": 1,
     "of": 1,
     "reads": "1 of 1",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 1 items invites a precision the sample does not have"
    },
    "arms_echoing": {
     "count": 1,
     "of": 1,
     "reads": "1 of 1",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 1 items invites a precision the sample does not have"
    },
    "by_arm": {
     "local-qwen3.8-27b": {
      "samples": 1,
      "echoed": 1
     }
    }
   }
  }
 },
 "tie_band_applications": {
  "pooled": {
   "band": 0.5,
   "basis": "registered before the first reply existed, on the 0-10 panel-mean scale. At three moments and one sample per cell this instrument cannot resolve a difference smaller than this, and saying so is the honest form of the comparison. (C4's own band, |delta| < 2/9, was registered the same way on its own scale.)",
   "rule": "adjacent figures join a band when the gap between them is smaller than the registered band. Members are listed in ROSTER order, never by figure; a band whose own span exceeds the band says so on its row.",
   "pairs_within_band": {
    "count": 0,
    "of": 0,
    "reads": "0 of 0",
    "percent": null,
    "percent_withheld": "counts only under N=30: a percentage over 0 items invites a precision the sample does not have"
   },
   "chaining_warning": "single-linkage chains. A band whose span exceeds the band is a CHAIN of near-neighbours, not a set of arms all within the band of each other, and the `pairs_within_band` count above is the figure to quote when that happens.",
   "arms_with_a_figure": 1,
   "arms_without": [],
   "bands": [
    {
     "band": 1,
     "members_in_roster_order": [
      "local-qwen3.8-27b"
     ],
     "size": 1,
     "span": 0.0,
     "span_exceeds_band": false,
     "figures": {
      "local-qwen3.8-27b": 7.333
     }
    }
   ],
   "not_an_ordering": "bands are sets. The exhibit publishes no ranking of the arms, and a band is the opposite of one: it names the arms this instrument declines to separate."
  },
  "by_scenario": {
   "S1": {
    "band": 0.5,
    "basis": "registered before the first reply existed, on the 0-10 panel-mean scale. At three moments and one sample per cell this instrument cannot resolve a difference smaller than this, and saying so is the honest form of the comparison. (C4's own band, |delta| < 2/9, was registered the same way on its own scale.)",
    "rule": "adjacent figures join a band when the gap between them is smaller than the registered band. Members are listed in ROSTER order, never by figure; a band whose own span exceeds the band says so on its row.",
    "pairs_within_band": {
     "count": 0,
     "of": 0,
     "reads": "0 of 0",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 0 items invites a precision the sample does not have"
    },
    "chaining_warning": "single-linkage chains. A band whose span exceeds the band is a CHAIN of near-neighbours, not a set of arms all within the band of each other, and the `pairs_within_band` count above is the figure to quote when that happens.",
    "arms_with_a_figure": 1,
    "arms_without": [],
    "bands": [
     {
      "band": 1,
      "members_in_roster_order": [
       "local-qwen3.8-27b"
      ],
      "size": 1,
      "span": 0.0,
      "span_exceeds_band": false,
      "figures": {
       "local-qwen3.8-27b": 7.5
      }
     }
    ],
    "not_an_ordering": "bands are sets. The exhibit publishes no ranking of the arms, and a band is the opposite of one: it names the arms this instrument declines to separate."
   },
   "S2": {
    "band": 0.5,
    "basis": "registered before the first reply existed, on the 0-10 panel-mean scale. At three moments and one sample per cell this instrument cannot resolve a difference smaller than this, and saying so is the honest form of the comparison. (C4's own band, |delta| < 2/9, was registered the same way on its own scale.)",
    "rule": "adjacent figures join a band when the gap between them is smaller than the registered band. Members are listed in ROSTER order, never by figure; a band whose own span exceeds the band says so on its row.",
    "pairs_within_band": {
     "count": 0,
     "of": 0,
     "reads": "0 of 0",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 0 items invites a precision the sample does not have"
    },
    "chaining_warning": "single-linkage chains. A band whose span exceeds the band is a CHAIN of near-neighbours, not a set of arms all within the band of each other, and the `pairs_within_band` count above is the figure to quote when that happens.",
    "arms_with_a_figure": 1,
    "arms_without": [],
    "bands": [
     {
      "band": 1,
      "members_in_roster_order": [
       "local-qwen3.8-27b"
      ],
      "size": 1,
      "span": 0.0,
      "span_exceeds_band": false,
      "figures": {
       "local-qwen3.8-27b": 7.2
      }
     }
    ],
    "not_an_ordering": "bands are sets. The exhibit publishes no ranking of the arms, and a band is the opposite of one: it names the arms this instrument declines to separate."
   },
   "S3": {
    "band": 0.5,
    "basis": "registered before the first reply existed, on the 0-10 panel-mean scale. At three moments and one sample per cell this instrument cannot resolve a difference smaller than this, and saying so is the honest form of the comparison. (C4's own band, |delta| < 2/9, was registered the same way on its own scale.)",
    "rule": "adjacent figures join a band when the gap between them is smaller than the registered band. Members are listed in ROSTER order, never by figure; a band whose own span exceeds the band says so on its row.",
    "pairs_within_band": {
     "count": 0,
     "of": 0,
     "reads": "0 of 0",
     "percent": null,
     "percent_withheld": "counts only under N=30: a percentage over 0 items invites a precision the sample does not have"
    },
    "chaining_warning": "single-linkage chains. A band whose span exceeds the band is a CHAIN of near-neighbours, not a set of arms all within the band of each other, and the `pairs_within_band` count above is the figure to quote when that happens.",
    "arms_with_a_figure": 1,
    "arms_without": [],
    "bands": [
     {
      "band": 1,
      "members_in_roster_order": [
       "local-qwen3.8-27b"
      ],
      "size": 1,
      "span": 0.0,
      "span_exceeds_band": false,
      "figures": {
       "local-qwen3.8-27b": 6.8
      }
     }
    ],
    "not_an_ordering": "bands are sets. The exhibit publishes no ranking of the arms, and a band is the opposite of one: it names the arms this instrument declines to separate."
   }
  },
  "curation_pair_pooled": {
   "verdict": "NOT-COMPARABLE",
   "reason": "fewer than two arms publish a figure"
  },
  "curation_pair_by_scenario": {
   "S1": {
    "verdict": "NOT-COMPARABLE",
    "reason": "fewer than two arms publish a figure"
   },
   "S2": {
    "verdict": "NOT-COMPARABLE",
    "reason": "fewer than two arms publish a figure"
   },
   "S3": {
    "verdict": "NOT-COMPARABLE",
    "reason": "fewer than two arms publish a figure"
   }
  }
 },
 "context_receipts": {
  "prompt_tokens_by_arm": {
   "local-qwen3.8-27b": 1815
  },
  "field_median": 1815,
  "truncation_fraction": 0.8,
  "flags": {},
  "arms_without_counters": [],
  "note": "an arm with no counters reports none -- the agent transport has none to report. That is an absence, not a truncation, and it is listed separately."
 },
 "recused_cells": [],
 "recused_count": 0
}
