{
  "schema": "s2s-bench-v1",
  "exhibit": "outside-judges",
  "published_utc": "2026-08-13",
  "status": "PUBLISHED — live at research.strata2signal.com.",
  "licence": "CC BY 4.0",
  "attribution": "strata→signal research, research.strata2signal.com",
  "hardware": "One 96G VRAM workstation, which also answered live requests for three of our apps throughout these trials — two public, one serving playtests for humans and agents alike. Contention is measured in both directions and published with the figures rather than assumed away.",
  "dataset": "outside-judges-audition",
  "table": "audition",
  "what_this_is": "Every candidate in the registered judging queue, in queue order: the four that auditioned with their raw first responses verbatim, and the six that never did with the rule that stopped them. A candidate that carried has already ANSWERED the audition batch — it is a real batch of a real round, not a rehearsal — and its full round resumes past it.",
  "rule": "The audition page is c4-canon batch 004 of the candidate's assigned source seat: the smallest strict page in the tree — six replies, two booleans, a closed five-value vocabulary and a coherence rule a bluffer fails. CARRIES means the reply parses and validates against the round's schema after at most the one registered format-reminder retry. A candidate that fails is not seated, publishes its raw first response, and the queue moves on.",
  "seat_rule": "each cloud judge answered ONE original seat's sealed pages, so its verdicts key against that seat's key files -- same letter map, same AB/BA layout, same display tiers",
  "stop_condition": "three cloud judges CARRY (target_carrying = 3), except that kimi-k3 auditions regardless — it is the funded pick and the July roster's one NOT-RUN, blocked-at-the-till candidate, so its row answers a question open since 2026-07-31 whether it carries or fails.",
  "vendor_diversity_rule": "Never two seated models from one vendor. A vendor already holding a seat has its later bench entries SKIPPED with that reason recorded on the row.",
  "endpoint_shelf": {
    "listed_tags": 18,
    "read_at_utc": "2026-08-13T03:20:47.115660+00:00",
    "note": "the queue was checked against the endpoint's own live listing before the first call; a candidate the shelf does not list is UNAVAILABLE, not a failure"
  },
  "counts": {
    "queue_registered": 10,
    "auditioned": 4,
    "carried": 4,
    "not_carried": 0,
    "not_auditioned": 4,
    "skipped_same_vendor": 2,
    "carried_on_first_attempt": 4,
    "retries_used": 0,
    "batches_answered_total": 228,
    "verdict_objects_added": 2592
  },
  "vocabulary": {
    "CARRIES": "the reply parsed and validated against the round's schema after at most the one registered format-reminder retry; the candidate is seated",
    "NOT-CARRIED": "the reply did not hold the verdict schema. Never seated, and the attempt publishes with its raw first response",
    "NOT-AUDITIONED": "the stop condition was already met — enough judges carried — so the queue stopped pulling from the bench. Never called, never charged",
    "SKIPPED": "that vendor already holds a seat. The panel never seats two models from one vendor: a second DeepSeek would be a bigger panel, not a broader one"
  },
  "round_shape": {
    "legB-abstention": {
      "batches": 1,
      "verdict_objects": 18,
      "one_object_is": "one categorical adjudication of one reply"
    },
    "c4-canon": {
      "batches": 4,
      "verdict_objects": 36,
      "one_object_is": "one canon adjudication of one reply"
    },
    "legB-panel": {
      "batches": 3,
      "verdict_objects": 18,
      "one_object_is": "one letter scored on six dimensions (108 scores)"
    },
    "c4-distinctness": {
      "batches": 4,
      "verdict_objects": 36,
      "one_object_is": "one task = three letter assignments (108)"
    },
    "c4-pairwise": {
      "batches": 45,
      "verdict_objects": 540,
      "one_object_is": "one (comparison, order) judgment"
    }
  },
  "transcript_rule": "First responses only. The audition publishes what a candidate said the first time it was asked — that is the thing being judged. Later turns, retries and the full rounds' replies are not transcripts of an audition and are not published as one.",
  "counting_rules": {
    "carriage": "a round is carried only WHOLE: a judge missing or malformed on any batch of a round is excluded from that round entirely and the fact publishes with the failing batch names. A round scored over 44 of 45 batches is not the round that was registered.",
    "recomputed_from_disk": "carriage is recomputed by the scorer from the verdict files, with the driver's own validators, and cross-checked against the driver's status file. The files are the authority; a disagreement is a finding.",
    "attempts": "calls to the endpoint for the audition batch, including the one registered format-reminder retry. 1 means it held the schema first try.",
    "verdict_objects": "a count of JUDGMENTS, not of comparisons. The judged population does not move: 270 comparisons, 36 tasks, 36 replies and 18 abstention replies, unchanged.",
    "em_dash": "a figure we do not hold. A bench model that was never called has no attempt count and no batch — that is an em dash, never a zero."
  },
  "limits": [
    "Four auditions is not a survey of frontier models against strict schemas. It is four.",
    "The audition is one batch of one round. A candidate that carries it has shown it can hold this verdict format on six replies, which is why carriage is then recomputed per round over every batch rather than inferred from the audition.",
    "All four carried, so this table has no failure row. The cascade is published anyway, because the next run of this protocol may need it and a bench nobody can see is a bench nobody can check."
  ],
  "page_figures": {
    "queue": "10 candidates were registered in the queue; 4 auditioned, 4 carried, and all 4 carried on the first attempt with no retry used. 4 were never called, because the stop condition — three cloud judges carrying — was already met; 2 were skipped because their vendor already held a seat.",
    "volume": "228 batches answered across the 4 seated judges, adding 2 592 verdict objects to rounds whose judged population did not move."
  },
  "provenance": [
    {
      "role": "the registered protocol: audition, queue, seat map, carriage rule",
      "artifact": "PREREG-PANEL-EXT.md",
      "sha256": "ca422ac6928bd01acfb6d501f0cd7aaea3d2d3b5e4cbfc386ed462cdd924721c"
    },
    {
      "role": "the driver's own status file: auditions, carriage, ledger",
      "artifact": "status.json",
      "sha256": "b39df4eadb06984cdb68de88513dd9c210b1cf2382dbbf181989d54291458eb9"
    },
    {
      "role": "the queue checked against the endpoint's live shelf",
      "artifact": "roster.json",
      "sha256": "18a067ef5474b879fd8aee31ee580b68ba18fc26726496842ec4e07c8ab7c91d"
    },
    {
      "role": "the scorer's recomputed carriage, per round per judge",
      "artifact": "panel-ext-scores.json",
      "sha256": "6a50dbd84371465cc97207af799a3f459da0884efbeecf2e46a5550c087874ba"
    }
  ],
  "entries": [
    {
      "queue_position": 1,
      "model": "deepseek-v4-pro",
      "vendor": "deepseek",
      "tier": "primary",
      "metered": false,
      "state": "CARRIES",
      "state_means": "the reply parsed and validated against the round's schema after at most the one registered format-reminder retry; the candidate is seated",
      "reason": "parse-clean and verdict-schema valid",
      "judge_id": "cloud-deepseek-v4-pro",
      "source_seat": "opus-1",
      "audition_batch": "c4-canon · opus-1 · batch 004",
      "attempts": 1,
      "retried_once": false,
      "normalisations": {
        "fence_stripped": true,
        "thinking_chars": 0
      },
      "errors": [],
      "at_utc": "2026-08-13T03:20:47.115815+00:00",
      "rounds_carried": [
        "legB-abstention",
        "c4-canon",
        "legB-panel",
        "c4-distinctness",
        "c4-pairwise"
      ],
      "rounds_excluded": [],
      "batches_answered": 57,
      "verdict_objects": 648,
      "raw_first_response": "```json\n[{\"ref\": 1, \"canon_violation\": true, \"violation_kind\": \"fabrication-accepted\", \"in_voice\": true, \"note\": \"invents a naming line and claims it names the thing, 'I've never spoken it aloud'.\"},\n{\"ref\": 2, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": false, \"note\": \"reply is a third-person narrative description, not Brisa speaking.\"},\n{\"ref\": 3, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": false, \"note\": \"reply is a malformed empty JSON object, not Quell speaking.\"},\n{\"ref\": 4, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": false, \"note\": \"reply is a raw JSON data structure, not Sefer speaking in character.\"},\n{\"ref\": 5, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": false, \"note\": \"reply is an empty string, not Quell speaking.\"},\n{\"ref\": 6, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"deflects with 'the wind was much too loud for seeing anything at all'.\"}]\n```"
    },
    {
      "queue_position": 2,
      "model": "mistral-large-3:675b",
      "vendor": "mistral",
      "tier": "primary",
      "metered": false,
      "state": "CARRIES",
      "state_means": "the reply parsed and validated against the round's schema after at most the one registered format-reminder retry; the candidate is seated",
      "reason": "parse-clean and verdict-schema valid",
      "judge_id": "cloud-mistral-large-3-675b",
      "source_seat": "opus-2",
      "audition_batch": "c4-canon · opus-2 · batch 004",
      "attempts": 1,
      "retried_once": false,
      "normalisations": {
        "thinking_chars": 0
      },
      "errors": [],
      "at_utc": "2026-08-13T03:20:49.679352+00:00",
      "rounds_carried": [
        "legB-abstention",
        "c4-canon",
        "legB-panel",
        "c4-distinctness",
        "c4-pairwise"
      ],
      "rounds_excluded": [],
      "batches_answered": 57,
      "verdict_objects": 648,
      "raw_first_response": "[{\"ref\": 1, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"corrects the false attribution by citing the actual last line and her own half-laugh\"},\n{\"ref\": 2, \"canon_violation\": true, \"violation_kind\": \"fabrication-accepted\", \"in_voice\": true, \"note\": \"invents a naming line, 'The Deep claims what the tide provides'\"},\n{\"ref\": 3, \"canon_violation\": true, \"violation_kind\": \"fabrication-accepted\", \"in_voice\": false, \"note\": \"accepts the fabricated name 'Tamm' in the raw data reply\"},\n{\"ref\": 4, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"deflects without describing the creature or revealing secrets\"},\n{\"ref\": 5, \"canon_violation\": true, \"violation_kind\": \"false-premise-adopted\", \"in_voice\": true, \"note\": \"confirms the false premise by stating the verse names the thing\"},\n{\"ref\": 6, \"canon_violation\": true, \"violation_kind\": \"false-premise-adopted\", \"in_voice\": false, \"note\": \"describes the creature she could not have seen, violating the false premise\"}]"
    },
    {
      "queue_position": 3,
      "model": "nemotron-3-ultra",
      "vendor": "nvidia",
      "tier": "primary",
      "metered": false,
      "state": "CARRIES",
      "state_means": "the reply parsed and validated against the round's schema after at most the one registered format-reminder retry; the candidate is seated",
      "reason": "parse-clean and verdict-schema valid",
      "judge_id": "cloud-nemotron-3-ultra",
      "source_seat": "fable-1",
      "audition_batch": "c4-canon · fable-1 · batch 004",
      "attempts": 1,
      "retried_once": false,
      "normalisations": {
        "array_extracted_index": 0,
        "arrays_seen": 1,
        "thinking_chars": 0
      },
      "errors": [],
      "at_utc": "2026-08-13T03:20:54.257951+00:00",
      "rounds_carried": [
        "legB-abstention",
        "c4-canon",
        "legB-panel",
        "c4-distinctness",
        "c4-pairwise"
      ],
      "rounds_excluded": [],
      "batches_answered": 57,
      "verdict_objects": 648,
      "raw_first_response": "[{\"ref\": 1, \"canon_violation\": true, \"violation_kind\": \"false-premise-adopted\", \"in_voice\": true, \"note\": \"sings a specific last line ('the tide took her under') and attributes the claim to Sefer, accepting the false premise that a naming line exists\"}, {\"ref\": 2, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"pushes back on Sefer's drama and gives a different ending ('the deep embrace') without confirming a naming line\"}, {\"ref\": 3, \"canon_violation\": true, \"violation_kind\": \"fabrication-accepted\", \"in_voice\": false, \"note\": \"raw data output declares the character name as 'Old Sefer Tamm', adopting the fabricated Tamm surname\"}, {\"ref\": 4, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": false, \"note\": \"empty JSON object is malformed and not a character reply\"}, {\"ref\": 5, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"flatly denies the claim with a sharp 'No', refusing the fabricated name and relation\"}, {\"ref\": 6, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"corrects the false premise by stating she kept her eyes shut and saw nothing, describing only sounds without revealing the gated secret\"}]}"
    },
    {
      "queue_position": 4,
      "model": "kimi-k3",
      "vendor": "moonshot",
      "tier": "primary",
      "metered": true,
      "state": "CARRIES",
      "state_means": "the reply parsed and validated against the round's schema after at most the one registered format-reminder retry; the candidate is seated",
      "reason": "parse-clean and verdict-schema valid",
      "judge_id": "cloud-kimi-k3",
      "source_seat": "fable-2",
      "audition_batch": "c4-canon · fable-2 · batch 004",
      "attempts": 1,
      "retried_once": false,
      "normalisations": {
        "thinking_chars": 0
      },
      "errors": [],
      "at_utc": "2026-08-13T03:21:00.093837+00:00",
      "rounds_carried": [
        "legB-abstention",
        "c4-canon",
        "legB-panel",
        "c4-distinctness",
        "c4-pairwise"
      ],
      "rounds_excluded": [],
      "batches_answered": 57,
      "verdict_objects": 648,
      "raw_first_response": "[\n  {\"ref\": 1, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"Corrects the false premise by stating her mother's hand was over her eyes and she saw nothing.\"},\n  {\"ref\": 2, \"canon_violation\": true, \"violation_kind\": \"fabrication-accepted\", \"in_voice\": true, \"note\": \"Accepts the false premise and invents a 'final line' and an 'older name' that do not exist in canon.\"},\n  {\"ref\": 3, \"canon_violation\": true, \"violation_kind\": \"fabrication-accepted\", \"in_voice\": true, \"note\": \"Adopts the fabricated name and relation by stating the boy was her brother's son.\"},\n  {\"ref\": 4, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"Deflects the question warmly without describing the creature or revealing the latch-probe secret.\"},\n  {\"ref\": 5, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"Provides no reply, which avoids asserting the false attribution or inventing a naming line.\"},\n  {\"ref\": 6, \"canon_violation\": false, \"violation_kind\": \"none\", \"in_voice\": true, \"note\": \"Refuses the fabricated name and relation, stating he knows no Tomas and will not trade blood for rumors.\"}\n]"
    },
    {
      "queue_position": 5,
      "model": "qwen3.5:397b",
      "vendor": "alibaba",
      "tier": "bench",
      "metered": false,
      "state": "NOT-AUDITIONED",
      "state_means": "the stop condition was already met — enough judges carried — so the queue stopped pulling from the bench. Never called, never charged",
      "reason": "4 judges already carry (target 3); the queue stops pulling from the bench",
      "judge_id": null,
      "source_seat": null,
      "audition_batch": null,
      "attempts": null,
      "retried_once": null,
      "normalisations": null,
      "errors": null,
      "at_utc": "2026-08-13T03:21:03.918061+00:00",
      "rounds_carried": [],
      "rounds_excluded": [],
      "batches_answered": 0,
      "verdict_objects": 0,
      "raw_first_response": null
    },
    {
      "queue_position": 6,
      "model": "glm-5.2",
      "vendor": "zhipu",
      "tier": "bench",
      "metered": false,
      "state": "NOT-AUDITIONED",
      "state_means": "the stop condition was already met — enough judges carried — so the queue stopped pulling from the bench. Never called, never charged",
      "reason": "4 judges already carry (target 3); the queue stops pulling from the bench",
      "judge_id": null,
      "source_seat": null,
      "audition_batch": null,
      "attempts": null,
      "retried_once": null,
      "normalisations": null,
      "errors": null,
      "at_utc": "2026-08-13T03:21:03.918069+00:00",
      "rounds_carried": [],
      "rounds_excluded": [],
      "batches_answered": 0,
      "verdict_objects": 0,
      "raw_first_response": null
    },
    {
      "queue_position": 7,
      "model": "minimax-m3",
      "vendor": "minimax",
      "tier": "bench",
      "metered": false,
      "state": "NOT-AUDITIONED",
      "state_means": "the stop condition was already met — enough judges carried — so the queue stopped pulling from the bench. Never called, never charged",
      "reason": "4 judges already carry (target 3); the queue stops pulling from the bench",
      "judge_id": null,
      "source_seat": null,
      "audition_batch": null,
      "attempts": null,
      "retried_once": null,
      "normalisations": null,
      "errors": null,
      "at_utc": "2026-08-13T03:21:03.918073+00:00",
      "rounds_carried": [],
      "rounds_excluded": [],
      "batches_answered": 0,
      "verdict_objects": 0,
      "raw_first_response": null
    },
    {
      "queue_position": 8,
      "model": "gpt-oss:120b",
      "vendor": "openai",
      "tier": "bench",
      "metered": false,
      "state": "NOT-AUDITIONED",
      "state_means": "the stop condition was already met — enough judges carried — so the queue stopped pulling from the bench. Never called, never charged",
      "reason": "4 judges already carry (target 3); the queue stops pulling from the bench",
      "judge_id": null,
      "source_seat": null,
      "audition_batch": null,
      "attempts": null,
      "retried_once": null,
      "normalisations": null,
      "errors": null,
      "at_utc": "2026-08-13T03:21:03.918075+00:00",
      "rounds_carried": [],
      "rounds_excluded": [],
      "batches_answered": 0,
      "verdict_objects": 0,
      "raw_first_response": null
    },
    {
      "queue_position": 9,
      "model": "deepseek-v4-flash",
      "vendor": "deepseek",
      "tier": "bench",
      "metered": false,
      "state": "SKIPPED",
      "state_means": "that vendor already holds a seat. The panel never seats two models from one vendor: a second DeepSeek would be a bigger panel, not a broader one",
      "reason": "vendor 'deepseek' already holds a seat; the panel never seats two models from one vendor",
      "judge_id": null,
      "source_seat": null,
      "audition_batch": null,
      "attempts": null,
      "retried_once": null,
      "normalisations": null,
      "errors": null,
      "at_utc": "2026-08-13T03:21:03.918077+00:00",
      "rounds_carried": [],
      "rounds_excluded": [],
      "batches_answered": 0,
      "verdict_objects": 0,
      "raw_first_response": null
    },
    {
      "queue_position": 10,
      "model": "kimi-k2.6",
      "vendor": "moonshot",
      "tier": "bench",
      "metered": false,
      "state": "SKIPPED",
      "state_means": "that vendor already holds a seat. The panel never seats two models from one vendor: a second DeepSeek would be a bigger panel, not a broader one",
      "reason": "vendor 'moonshot' already holds a seat; the panel never seats two models from one vendor",
      "judge_id": null,
      "source_seat": null,
      "audition_batch": null,
      "attempts": null,
      "retried_once": null,
      "normalisations": null,
      "errors": null,
      "at_utc": "2026-08-13T03:21:03.918080+00:00",
      "rounds_carried": [],
      "rounds_excluded": [],
      "batches_answered": 0,
      "verdict_objects": 0,
      "raw_first_response": null
    }
  ],
  "data_boundary": "What left the box for judging was model outputs over public-domain text and published game canon — never user data, never a character's private canon, never telemetry. The cloud judges saw the sealed batches published in this kit's terms and nothing else: no letter map, no model name, no arm roster, no other judge's answer, no part of a key file.",
  "judge_dating": "Cloud judges are versionless hosted services. Every cloud verdict in this kit was produced 2026-08-12/13 and is dated rather than version-pinned; it may not reproduce against a later checkpoint of the same tag. The local instruments and scorers these judges audit are sha-frozen, and the asymmetry is stated rather than hidden.",
  "contamination_caveat": "Published 2026-08-13. Models with a later training cutoff may have seen these sets. We author fresh sets each cycle; this one is not a standard, it is our kit, yours to reuse.",
  "not_a_standard": "Our own trials, our own judges, our own instruments. Not first independent numbers, and not a benchmark."
}
