{
  "schema": "s2s-bench-v1",
  "exhibit": "outside-judges",
  "published_utc": "2026-08-13",
  "status": "PUBLISHED — live at research.strata2signal.com.",
  "licence": "CC BY 4.0",
  "attribution": "strata→signal research, research.strata2signal.com",
  "hardware": "One 96G VRAM workstation, which also answered live requests for three of our apps throughout these trials — two public, one serving playtests for humans and agents alike. Contention is measured in both directions and published with the figures rather than assumed away.",
  "dataset": "outside-judges-bill",
  "table": "the bill",
  "what_this_is": "Every duty this night bought, on both sides of the table: four cloud judges re-judging five sealed rounds, and the one metered model's three candidate legs besides. Published because nobody else does, and because \"we can't afford outside judges\" should now require a footnote.",
  "headline": {
    "metered_total_usd": 1.5068,
    "metered_total_display": "$1.51",
    "of_credit_usd": 20.0,
    "metered_calls": 289,
    "judge_side_usd": 0.868,
    "candidate_side_usd": 0.6388,
    "plan_included_judges": [
      "deepseek-v4-pro",
      "mistral-large-3:675b",
      "nemotron-3-ultra"
    ],
    "reading": "one metered model, 289 metered calls, both sides of the table, $1.51 of a $20 credit. Multi-vendor judging is not an enterprise budget line; it is pocket change and a registered protocol."
  },
  "pricing": {
    "metered_model": "kimi-k3",
    "usd_per_mtok": {
      "input": 3.0,
      "cached_input": 0.3,
      "output": 15.0
    },
    "credit_usd": 20.0,
    "upper_bound_rule": "every metered figure is priced at the UNCACHED input rate. Cached input bills at a tenth and is not reported by the API, so each figure is an upper bound and never an under-count — with one named exception, the gauntlet's four unbanked probe calls.",
    "unmetered_rule": "three of the four cloud judges are included in a plan the workshop already pays for and burn no credit on this key: they print $0.00*. The asterisk is load-bearing: $0.00* means no per-token charge on these calls — the access itself is a monthly subscription with rotating token limits, a real cost that does not itemize per call. The agent-harness rows print an em dash, because we hold no figure for them at all."
  },
  "budget": {
    "soft_cap_usd": 18.0,
    "candidate_reserve_usd": 6.0,
    "judge_budget_usd": 12.0,
    "trims": [],
    "registered_projection_usd": {
      "judge_side": 0.91,
      "candidate_side": 3.4,
      "note": "the pre-registered projection, for comparison with what actually happened. The judge side came in under it; the cap never bit and no round was trimmed."
    },
    "trim_order_if_it_had_bitten": "keep legB-abstention, c4-canon and legB-panel whole; trim c4-pairwise BY BATCH INDEX, published as 'judged N of 45 batches, batches 1..N'; the candidate legs are never trimmed. Pre-registered by index, never a silent subset.",
    "hard_abort": "HTTP 402 stops everything, unfinished rows publish PENDING, and a 402 on the judge side means the candidate legs are not started at all",
    "trims_taken": [],
    "payment_stop": null
  },
  "estimates": {
    "rule": "a cost ESTIMATE printed and logged before every metered round — ceil(chars/4) on the sealed prompt bytes plus a registered per-object output allowance. It is labelled an estimate everywhere it appears and it never becomes the published figure.",
    "per_round": [
      {
        "round": "legB-abstention",
        "model": "kimi-k3",
        "metered": true,
        "batches": 1,
        "est_input_tokens": 2710,
        "est_output_tokens": 1030,
        "est_usd": 0.0236,
        "basis": "4 chars/token on the sealed prompt bytes, plus a registered per-object output allowance. An ESTIMATE; the ledger totals the endpoint's own counters."
      },
      {
        "round": "c4-canon",
        "model": "kimi-k3",
        "metered": true,
        "batches": 4,
        "est_input_tokens": 8791,
        "est_output_tokens": 2140,
        "est_usd": 0.0585,
        "basis": "4 chars/token on the sealed prompt bytes, plus a registered per-object output allowance. An ESTIMATE; the ledger totals the endpoint's own counters."
      },
      {
        "round": "legB-panel",
        "model": "kimi-k3",
        "metered": true,
        "batches": 3,
        "est_input_tokens": 8774,
        "est_output_tokens": 930,
        "est_usd": 0.0403,
        "basis": "4 chars/token on the sealed prompt bytes, plus a registered per-object output allowance. An ESTIMATE; the ledger totals the endpoint's own counters."
      },
      {
        "round": "c4-distinctness",
        "model": "kimi-k3",
        "metered": true,
        "batches": 4,
        "est_input_tokens": 9900,
        "est_output_tokens": 1240,
        "est_usd": 0.0483,
        "basis": "4 chars/token on the sealed prompt bytes, plus a registered per-object output allowance. An ESTIMATE; the ledger totals the endpoint's own counters."
      },
      {
        "round": "c4-pairwise",
        "model": "kimi-k3",
        "metered": true,
        "batches": 45,
        "est_input_tokens": 114496,
        "est_output_tokens": 26100,
        "est_usd": 0.735,
        "basis": "4 chars/token on the sealed prompt bytes, plus a registered per-object output allowance. An ESTIMATE; the ledger totals the endpoint's own counters."
      }
    ]
  },
  "counting_rules": {
    "calls": "one request on the wire, retries included.",
    "tokens": "the endpoint's own prompt_eval_count and eval_count, summed over the calls that banked them. Never estimated, never modelled.",
    "cost": "tokens x the uncached price list, to four decimals; displayed to the cent.",
    "em_dash": "a figure we do not hold — an unmetered transport with no counters. It is never a zero, and 0.00 never means unmeasured."
  },
  "limits": [
    "These are our prices on our key on one night. A different plan, a different endpoint or a different day prices the same work differently.",
    "The gauntlet's four pre-run probe calls were charged and banked no counters, so the candidate-side figure is short by four small calls.",
    "The agent-harness duties are real work with a real cost that this instrument cannot see. Nothing here should be read as their being free."
  ],
  "page_figures": {
    "candidate_side": "kimi-k3 also sat 129 seat-exam (Leg A) calls ($0.25), 63 judge-leg (C1) calls ($0.11), 40 assistant-leg (C2) calls ($0.28) — $0.64 across the three legs (Leg A is the seat exam's registered id; C1 and C2 are the chair trials' judge and assistant legs). The two Claude candidates sat 129 calls each on the agent harness, unmetered on this key and printed as an em dash rather than a zero.",
    "total": "$1.51 of a $20 credit, across 289 metered calls on both sides of the table — one metered model, the whole night.",
    "undercount": "the gauntlet's four pre-run probe calls were charged and banked no counters, so the candidate-side figure is short by four small calls. It is the one place in this bill that under-states, and it is named rather than absorbed."
  },
  "provenance": [
    {
      "role": "the registered budget, the cap and the trim order",
      "artifact": "PREREG-GAUNTLET.md",
      "sha256": "98bdc43a22ae2ec2791b44202e8479635145926547b22a83cbec0dc45e5dd0b5"
    },
    {
      "role": "the judge-side ledgers and the pre-call estimates",
      "artifact": "status.json",
      "sha256": "b39df4eadb06984cdb68de88513dd9c210b1cf2382dbbf181989d54291458eb9"
    },
    {
      "role": "the candidate-side ledger as written",
      "artifact": "kimi-candidate.json",
      "sha256": "609920f8f32e929f244d553b9f2dd52f9ca717d62f3f4d8e8bc3f01428295bec"
    },
    {
      "role": "the run's own log — the witness for the overwritten C1 record",
      "artifact": "panel-ext.log",
      "sha256": "2ca4b9c7f376d408cfd5c43e4e5b66574cfb412badfef00bae1f33e06b9103ef"
    },
    {
      "role": "the gauntlet's append-only rows, with their per-call counters",
      "artifact": "judges-gauntlet-cloud-kimi-k3.jsonl",
      "sha256": "6336ddb5d73a873e2fdbcde3e1a60af5fa5ed73d519ca22369b4791cbde32317"
    }
  ],
  "judge_side": [
    {
      "duty_kind": "judge",
      "judge_id": "cloud-deepseek-v4-pro",
      "model": "deepseek-v4-pro",
      "vendor": "deepseek",
      "source_seat": "opus-1",
      "metered": false,
      "calls": 57,
      "prompt_tokens": 143493,
      "completion_tokens": 30889,
      "usd_upper_bound": 0.0,
      "pricing_usd_per_mtok": {
        "input": 0.0,
        "cached_input": 0.0,
        "output": 0.0
      },
      "basis": "actual token counts from the endpoint's own prompt_eval_count / eval_count, priced at the UNCACHED input rate. Cached input bills at a tenth and is not reported by the API, so this is an upper bound, never an under-count.",
      "evidence": "the driver's own committed ledger, re-priced here from its own counters"
    },
    {
      "duty_kind": "judge",
      "judge_id": "cloud-mistral-large-3-675b",
      "model": "mistral-large-3:675b",
      "vendor": "mistral",
      "source_seat": "opus-2",
      "metered": false,
      "calls": 57,
      "prompt_tokens": 143974,
      "completion_tokens": 30031,
      "usd_upper_bound": 0.0,
      "pricing_usd_per_mtok": {
        "input": 0.0,
        "cached_input": 0.0,
        "output": 0.0
      },
      "basis": "actual token counts from the endpoint's own prompt_eval_count / eval_count, priced at the UNCACHED input rate. Cached input bills at a tenth and is not reported by the API, so this is an upper bound, never an under-count.",
      "evidence": "the driver's own committed ledger, re-priced here from its own counters"
    },
    {
      "duty_kind": "judge",
      "judge_id": "cloud-nemotron-3-ultra",
      "model": "nemotron-3-ultra",
      "vendor": "nvidia",
      "source_seat": "fable-1",
      "metered": false,
      "calls": 57,
      "prompt_tokens": 144658,
      "completion_tokens": 29132,
      "usd_upper_bound": 0.0,
      "pricing_usd_per_mtok": {
        "input": 0.0,
        "cached_input": 0.0,
        "output": 0.0
      },
      "basis": "actual token counts from the endpoint's own prompt_eval_count / eval_count, priced at the UNCACHED input rate. Cached input bills at a tenth and is not reported by the API, so this is an upper bound, never an under-count.",
      "evidence": "the driver's own committed ledger, re-priced here from its own counters"
    },
    {
      "duty_kind": "judge",
      "judge_id": "cloud-kimi-k3",
      "model": "kimi-k3",
      "vendor": "moonshot",
      "source_seat": "fable-2",
      "metered": true,
      "calls": 57,
      "prompt_tokens": 145192,
      "completion_tokens": 28825,
      "usd_upper_bound": 0.868,
      "pricing_usd_per_mtok": {
        "input": 3.0,
        "cached_input": 0.3,
        "output": 15.0
      },
      "basis": "actual token counts from the endpoint's own prompt_eval_count / eval_count, priced at the UNCACHED input rate. Cached input bills at a tenth and is not reported by the API, so this is an upper bound, never an under-count.",
      "evidence": "the driver's own committed ledger, re-priced here from its own counters"
    }
  ],
  "candidate_side": [
    {
      "duty_kind": "candidate",
      "leg": "Leg A — the gauntlet",
      "model": "kimi-k3",
      "vendor": "moonshot",
      "metered": true,
      "calls": 129,
      "prompt_tokens": 48264,
      "completion_tokens": 7001,
      "usd_upper_bound": 0.2498,
      "pricing_usd_per_mtok": {
        "input": 3.0,
        "cached_input": 0.3,
        "output": 15.0
      },
      "what": "43 claim-cases x 3 repeats on the frozen seat instrument",
      "evidence": "recounted from the endpoint's own per-call counters in the frozen runner's append-only rows: that runner writes no ledger of its own",
      "not_counted": "the four pre-run probe calls banked no counters, so this figure covers the 129 scored calls only. The probes were charged and are not in it — the one place in this bill that is an under-count, named rather than absorbed"
    },
    {
      "duty_kind": "candidate",
      "leg": "C1 — the judge chair",
      "model": "kimi-k3",
      "vendor": "moonshot",
      "metered": true,
      "calls": 63,
      "prompt_tokens": 22938,
      "completion_tokens": 2847,
      "usd_upper_bound": 0.1115,
      "pricing_usd_per_mtok": {
        "input": 3.0,
        "cached_input": 0.3,
        "output": 15.0
      },
      "what": "21 cases x 3 repeats, EXPLORATORY (the protocol fence is unevaluated on this transport, so the row publishes and is never ranked)",
      "evidence": "recounted from the banked per-call counters and checked against the run log's ledger line. This leg's ledger RECORD was overwritten by the next leg's on the way out — the file holds C2's — so the recount is the artifact and the log line is its witness"
    },
    {
      "duty_kind": "candidate",
      "leg": "C2 — the assistant",
      "model": "kimi-k3",
      "vendor": "moonshot",
      "metered": true,
      "calls": 40,
      "prompt_tokens": 72454,
      "completion_tokens": 4008,
      "usd_upper_bound": 0.2775,
      "pricing_usd_per_mtok": {
        "input": 3.0,
        "cached_input": 0.3,
        "output": 15.0
      },
      "what": "20 items x 2 repeats at think:false, DESCRIPTIVE as the leg is for every arm",
      "evidence": "recounted from the banked counters and checked against the committed ledger"
    },
    {
      "duty_kind": "candidate",
      "leg": "Leg A — the gauntlet",
      "model": "claude-opus-5",
      "vendor": "anthropic",
      "metered": false,
      "calls": 129,
      "prompt_tokens": null,
      "completion_tokens": null,
      "usd_upper_bound": null,
      "pricing_usd_per_mtok": null,
      "what": "the same 43 cases x 3 repeats, on the agent-harness protocol class",
      "evidence": "the agent harness reports no token counters and bills against no key of this run's. The cost is UNMEASURED, not zero, and prints as an em dash"
    },
    {
      "duty_kind": "candidate",
      "leg": "Leg A — the gauntlet",
      "model": "claude-fable-5",
      "vendor": "anthropic",
      "metered": false,
      "calls": 129,
      "prompt_tokens": null,
      "completion_tokens": null,
      "usd_upper_bound": null,
      "pricing_usd_per_mtok": null,
      "what": "the same 43 cases x 3 repeats, on the agent-harness protocol class",
      "evidence": "as above — unmeasured on this key, never a zero"
    }
  ],
  "data_boundary": "What left the box for judging was model outputs over public-domain text and published game canon — never user data, never a character's private canon, never telemetry. The cloud judges saw the sealed batches published in this kit's terms and nothing else: no letter map, no model name, no arm roster, no other judge's answer, no part of a key file.",
  "judge_dating": "Cloud judges are versionless hosted services. Every cloud verdict in this kit was produced 2026-08-12/13 and is dated rather than version-pinned; it may not reproduce against a later checkpoint of the same tag. The local instruments and scorers these judges audit are sha-frozen, and the asymmetry is stated rather than hidden.",
  "contamination_caveat": "Published 2026-08-13. Models with a later training cutoff may have seen these sets. We author fresh sets each cycle; this one is not a standard, it is our kit, yours to reuse.",
  "not_a_standard": "Our own trials, our own judges, our own instruments. Not first independent numbers, and not a benchmark."
}
