{
 "schema": "s2s-bench-v1",
 "kind": "reference-arm-row",
 "exhibit": "the-same-sixteen",
 "what_this_is": "ONE row, for one model, on one day. It exists so an index that lists which exhibits a model appears in can read this exhibit's own words for the arm instead of parsing prose — and so that whatever reads it cannot drop the class label, which is the whole fence. Every field here is copied from RESULTS-TABLES.md and the narrative beside it; nothing is computed.",
 "tag": "glm-5.3-flash:cloud",
 "class": "REFERENCE ARM · cloud · dated",
 "class_registered_in": "prereg-reference-arm.md (operator ruling D-20260828-18)",
 "ranked_against": null,
 "ranked_against_note": "Nothing. The scorer assigns outcome states mechanically and stamped the toolbench row RANKED; the class label overrides it. This row carries no ordering, no margin and no comparison verdict, and it sets no threshold.",
 "read_on": "2026-08-28",
 "window_utc": {
  "start": "2026-08-28T16:37:59Z",
  "end": "2026-08-28T16:49:43Z"
 },
 "vendor_stated": "Z.ai GLM-5.3-Flash; 320B total / 18B active per token, 1M context, MIT licence, published to ollama's cloud 2026-08-26 — read off the vendor's library page 2026-08-28 and labelled as vendor figures, because a hosted model cannot be weighed here.",
 "posture": "think:true",
 "posture_why": "measured, not stylistic: `think` does not switch this model's reasoning off on this path, it only decides whether ollama parses the reasoning out of the reply. Under think:false the raw reasoning lands in message.content, which every leg here scores — a measurement failure, not a low score.",
 "legs": [
  {
   "leg": "C3 toolbench",
   "disposition": "READ",
   "result": "16/19",
   "interval": "Wilson95 [62.4%, 94.5%]",
   "discipline": "DESCRIPTIVE",
   "detail": "single 5/6 · chains 5/5 · distractor 2/2 · honesty 3/5 · error-recovery 1/1",
   "source": "RESULTS-TABLES.md · C3-RESULTS.md"
  },
  {
   "leg": "field exam",
   "disposition": "READ (40 of 60 valid)",
   "result": "39/40",
   "interval": "Wilson95 [87.1%, 99.6%]",
   "discipline": "DESCRIPTIVE",
   "detail": "T1 20/20 · T2 19/20 · T3 schema-extract (20 items) EXCLUDED as a MEASUREMENT FAILURE (cloud: `format` unenforced). The 60-item headline is not quotable on this path.",
   "source": "field/glm-5.3-flash-cloud.json · RESULTS-TABLES.md"
  },
  {
   "leg": "fit (24 GB)",
   "disposition": "NOT-APPLICABLE",
   "why": "hosted; no local weights, nothing to fit"
  },
  {
   "leg": "C0 polarity",
   "disposition": "NOT-RUN",
   "why": "reference arm; only C3 and the field exam are carried"
  },
  {
   "leg": "C1 judge",
   "disposition": "NOT-RUN",
   "why": "cloud path invalid — the verdict binds on a `format` JSON-schema grammar the cloud path accepts and silently ignores, confirmed on this date at 16:49:43Z"
  },
  {
   "leg": "C2 assistant",
   "disposition": "NOT-RUN",
   "why": "cloud path invalid — same `format` defect"
  },
  {
   "leg": "C5 tok/s",
   "disposition": "NOT-RUN",
   "why": "not our silicon — a decode rate taken over the public internet against someone else's fleet under someone else's batching is unattributable"
  },
  {
   "leg": "C7 filing",
   "disposition": "NOT-RUN",
   "why": "not our silicon — the long-context recall leg is bound to a model resident on a card we can name"
  },
  {
   "leg": "seat-43",
   "disposition": "NOT-RUN",
   "why": "reference arm; seat exams do not apply to a model that can never hold a seat"
  }
 ],
 "secondary_counts": {
  "argument_fidelity": "49/66 (74.2%) · Wilson95 [62.6%, 83.3%]",
  "spurious_calls": "39/88 (44.3%) · Wilson95 [34.4%, 54.7%] — inside the local 23.8-47.3% range, and dominated by round-cap loops, so the calls are nothing like independent trials",
  "chains": "right final answer 5/5; the pre-registered call sequence 1/5"
 },
 "cost_signals": {
  "note": "ollama exposes no per-call price; these are the response's own counters, summed verbatim. No currency figure is quoted because none was taken.",
  "c3": {
   "eval_count": 8115,
   "prompt_eval_count": 195532,
   "total_duration_s": 155.4
  },
  "field": {
   "eval_count": 4166,
   "prompt_eval_count": 14424,
   "wall_s": 67.6
  },
  "arm_total": {
   "eval_count": 12281,
   "prompt_eval_count": 209956
  }
 },
 "the_finding_is_about_the_ruler": "At nineteen items this bench separates a 16/19 row only from a comparator seven tasks or more below it (Fisher exact, α = 0.05: 16-vs-9 separates, 16-vs-10 does not), and the field exam is saturated at this level. Two of the nineteen tasks are scored by literal-string checkers that fail correct answers, so every row they touch is a floor rather than a reading. Those are findings about the instruments, not about any model on them."
}
