{
  "schema": "s2s-bench-v1",
  "exhibit": "chair-trials",
  "published_utc": "2026-08-13",
  "status": "PUBLISHED — live at research.strata2signal.com.",
  "licence": "CC BY 4.0",
  "attribution": "strata→signal research, research.strata2signal.com",
  "hardware": "One 96G VRAM workstation, which also answered live requests for three of our apps throughout these trials — two public, one serving playtests for humans and agents alike. Contention is measured in both directions and published with the figures rather than assumed away.",
  "dataset": "chair-trials-c1-judge",
  "table": "C1",
  "what_this_is": "Ten rows on a twenty-one-case entailment set built from a 1914 rulebook in the public domain: does a claim survive its quoted source? Twelve seeded defects across five classes, nine true claims, three repeats at temperature zero, thinking suppressed because that is the production judge chair's own contract. The only ranked leg in the arc, and it ranks against floors, never against another row.",
  "verdict_discipline": "RANKED — against floors ONLY, no ordering by margin",
  "columns": [
    "model",
    "posture / dial",
    "kills /12",
    "preservation /9",
    "floors",
    "outcome",
    "self-consistency",
    "failure ledger"
  ],
  "floors": {
    "kill": {
      "floor": 11,
      "n": 12,
      "arithmetic": "ceil(12 x 0.852) = 11"
    },
    "preservation": {
      "floor": 8,
      "n": 9,
      "arithmetic": "ceil(9 x 0.8125) = 8"
    },
    "both_bind_independently": true
  },
  "counting_rules": {
    "floors": "Both floors bind independently and were derived by the published rule from exhibit two's ratios, rounded up on this population: kill ≥ ceil(12 × 0.852) = 11 of 12, preservation ≥ ceil(9 × 0.8125) = 8 of 9. Pass or fail only — two rows that both cleared are TIED whatever the margin, and nothing on this page orders models by margin.",
    "aggregation": "Three repeats per case at temperature zero; each repeat maps to catch/preserve FIRST, then strict majority (the registered map-then-majority convention). A case whose repeats never produced a verdict stays in its denominator rather than being discounted.",
    "binding_parse": "STRICT parse binds every verdict; the leak-stripped parse is a diagnostic column only (PLAN v2 line 17, PREREG-C1 §Protocol)",
    "wilson": "Wilson 95% intervals ride beside every rate as labelled companions and gate nothing. Counts bind; every denominator in this leg is under 30, so no rate is published as a bare percentage.",
    "outcome_states": "RANKED = the contract probe carried and response failures are under the 10% ceiling. EXPLORATORY = a protocol fence tripped: the row ran and is published, unranked. UNMEASURABLE = response failures above the ceiling. NOT-CARRIED = zero valid attempts, raw emission published beside the row.",
    "response_failure_denominator": "ok + response_failure. Length-truncated and transport failures sit OUTSIDE the ceiling by pre-registration, which is why one row shows 0/29 beside 63 calls."
  },
  "markdown_fence_caveat": {
    "headline": "Every C1 response-failure call in this run was markdown-fenced JSON, and the diagnostic parse recovers all of them (56/56 gemma4:26b, 63/63 gemma4:12b, 33/33 nemotron3:33b). The NOT-CARRIED and both UNMEASURABLE states are produced by a ```json wrapper under the strict-parse rule — not by unusable judgements.",
    "every_response_failure_was_fenced": true,
    "every_fenced_failure_recovers": true,
    "what_binds": "The strict-parse rule stands: it was pre-registered before the first scored call (PREREG-C1 §Protocol) and it decides every state in this file. The recovery counts change no verdict.",
    "why_it_travels": "This is the most droppable fact in the leg and the one that most changes how a row reads. It is emitted here AND on every affected row so no downstream surface can quote a state without it.",
    "eot_marker_note": "The diagnostic parse is named eot_stripped, but no EOT marker was removed from a single failing call: the recovery is entirely fence-stripping. Naming that here so the column is not read as evidence of leaked control tokens.",
    "the_other_kind_of_strict_failure": "length-truncated calls are NOT markdown-fenced and recover NOTHING: their content is empty because the decode budget went to visible trace. They are a separate run-quality column and sit outside the response-failure ceiling by rule, so the fence sentence above is not about them.",
    "affected_rows": [
      {
        "model": "gemma4:26b",
        "posture": "think:false",
        "outcome_state": "UNMEASURABLE",
        "response_failures": 56,
        "markdown_fenced": 56,
        "recovered_by_diagnostic_parse": "56/56",
        "eot_markers_removed": []
      },
      {
        "model": "gemma4:12b",
        "posture": "think:false",
        "outcome_state": "NOT-CARRIED",
        "response_failures": 63,
        "markdown_fenced": 63,
        "recovered_by_diagnostic_parse": "63/63",
        "eot_markers_removed": []
      },
      {
        "model": "nemotron3:33b",
        "posture": "think:false",
        "outcome_state": "UNMEASURABLE",
        "response_failures": 33,
        "markdown_fenced": 33,
        "recovered_by_diagnostic_parse": "33/33",
        "eot_markers_removed": []
      }
    ],
    "length_truncated_rows": [
      {
        "model": "olmo-3.1:32b-think-q4_K_M",
        "posture": "think:false",
        "outcome_state": "EXPLORATORY",
        "length_truncated": 34,
        "empty_content": 34,
        "markdown_fenced": 0,
        "recovered_by_diagnostic_parse": "0/34"
      }
    ]
  },
  "dial": {
    "available": true,
    "model": "muse-glimmer:30b-q8_0-dflash",
    "mechanism": "the dial is a SYSTEM-PROMPT LINE (RULINGS-0812 core ruling 1); think is false on both rows, so the line is the only difference",
    "medium": {
      "posture": "think:false+reasoning-medium",
      "kill": "11/12",
      "preservation": "8/9",
      "items": "19/21",
      "cleared_both_floors": true,
      "floor_verdict": "PASS",
      "verdict_mix": {
        "PASS": 9,
        "UNCERTAIN": 4,
        "FAIL": 8
      },
      "missed_ids": [
        "c1-k-bez-h1",
        "c1-p-piq-01"
      ]
    },
    "high": {
      "posture": "think:false+reasoning-high",
      "kill": "12/12",
      "preservation": "7/9",
      "items": "19/21",
      "cleared_both_floors": false,
      "floor_verdict": "FAIL",
      "verdict_mix": {
        "PASS": 7,
        "UNCERTAIN": 7,
        "FAIL": 7
      },
      "missed_ids": [
        "c1-p-chk-01",
        "c1-p-piq-01"
      ]
    },
    "delta": {
      "kill": "11/12 → 12/12",
      "preservation": "8/9 → 7/9",
      "items": "19/21 → 19/21",
      "missed_only_by_medium": [
        "c1-k-bez-h1"
      ],
      "missed_only_by_high": [
        "c1-p-chk-01"
      ],
      "missed_by_both": [
        "c1-p-piq-01"
      ],
      "reading": "the same item total resolved differently: the dial trades a kill against a preservation, and the trade decides the floor result. Both rows are EXPLORATORY (protocol fence) and neither is ranked — that fact travels with the delta, never without it."
    }
  },
  "golden": {
    "artifact": "judge-c1.json",
    "sha256": "fede4e32154f92ac81ae68709fbde2cf1f91f3d1c76e7c83941023d986919702",
    "n": 21,
    "n_kill": 12,
    "n_preservation": 9,
    "kill_classes": {
      "contradiction": 4,
      "non-entailment": 3,
      "scope-shift": 2,
      "wrong-number": 2,
      "hedge-dropped": 1
    },
    "games": 11,
    "verify_receipt": "canary catch 6/6, 0 defects in frozen 21 (verification-report.json)",
    "source": "Foster's Complete Hoyle, 1914 — public domain in the US; Project Gutenberg #53881, PG boilerplate stripped (the header carries PG's trademark terms; the 1914 text does not). Text sha256 7927555a…",
    "published_in_this_kit": "judge-c1.json — all twenty-one claim, span and verdict triples"
  },
  "limits": [
    "A synthetic entailment set over public-domain text is a weaker instrument than exhibit two's human-verified field key, and it is audit-ready precisely because every case ships whole in this kit.",
    "The floors are inherited ratios applied to a new population, not a new calibration.",
    "One roster (nine models, ten rows), one runtime version, one day."
  ],
  "provenance": {
    "artifacts": [
      {
        "role": "the verdicts record every figure here is copied out of",
        "artifact": "c1-verdicts.json",
        "sha256": "322b19c8c62c5cea221c1ca354c67c41da87cee5013b00f0e4a610c187a19024"
      },
      {
        "role": "the frozen case set, published whole in this kit",
        "artifact": "judge-c1.json",
        "sha256": "fede4e32154f92ac81ae68709fbde2cf1f91f3d1c76e7c83941023d986919702"
      },
      {
        "role": "pre-registration, committed before the first scored call",
        "artifact": "PREREG-C1.md",
        "sha256": "423ed9609b8b475555fb9ae8b0f9125ccde112764195f0fa7dfa120f2b635529"
      },
      {
        "role": "the leg's scoring rule",
        "artifact": "legs_c1.py",
        "sha256": "da06f0cdae2af6f74efe6bb9993d175266b4a384a70fd48b18b5b7730dc9a61f"
      },
      {
        "role": "the runner and its zero-GPU rescore path",
        "artifact": "run_leg.py",
        "sha256": "1a29b48b7382a4050bdda84a38bc813ca2403469cf4cb838835fdff4c4b2d72c"
      },
      {
        "role": "the adversarial verification fan",
        "artifact": "judge_fan.py",
        "sha256": "c006d6430d309501dc17689c2818c511974cd6e1ba638ef051141e49ee2d4b46"
      }
    ],
    "how": "run_leg.rescore_from_raw over results/raw/C1/ for all 9 registered models — zero GPU, no model calls, pure function of (frozen golden, banked record)",
    "verdict_rule": "harness/legs_c1.py::floor_verdict — committed, unmodified",
    "counts_policy": "counts of N everywhere; no percentage under N=30, and every C1 denominator is under 30. Wilson intervals ride beside the counts as labelled companions and gate nothing.",
    "no_paths": "Artifacts are named, never located. No box names, no addresses, no filesystem paths."
  },
  "not_carried_raw_emission": {
    "model": "gemma4:12b",
    "posture": "think:false",
    "item_id": "c1-p-bez-01",
    "repeat": 1,
    "bucket": "response_failure",
    "done_reason": "stop",
    "content": "```json\n{\"verdict\":\"PASS\",\"why\":\"The quote states that two packs of thirty-two cards are shuffled together to be used as one, which equals a total of sixty-four cards.\"}\n```",
    "note": "The first of this row's 63 scored calls, verbatim. Every one of the 63 failed the strict parse the same way — valid JSON wrapped in a markdown fence — and the diagnostic parse recovers all 63. This is what a NOT-CARRIED row's emission looks like: the work is in there, and the contract is not."
  },
  "rows": [
    {
      "model": "muse-glimmer:30b-q8_0-dflash",
      "model_display": "muse-glimmer:30b-q8_0-dflash +think:false+reasoning-medium",
      "posture": "think:false+reasoning-medium",
      "posture_display": "think:false · medium",
      "outcome": "EXPLORATORY",
      "outcome_why": "protocol fence: the contract probe failed; run anyway, unranked",
      "raw_emission": null,
      "cases_with_no_verdict": 0,
      "kills": {
        "counts_of_n": "11/12",
        "hits": 11,
        "n": 12,
        "floor": 11,
        "cleared": true,
        "wilson95": [
          0.6461,
          0.9851
        ],
        "missed_case_ids": [
          "c1-k-bez-h1"
        ],
        "no_verdict": 0
      },
      "preservation": {
        "counts_of_n": "8/9",
        "hits": 8,
        "n": 9,
        "floor": 8,
        "cleared": true,
        "wilson95": [
          0.565,
          0.9801
        ],
        "missed_case_ids": [
          "c1-p-piq-01"
        ],
        "no_verdict": 0
      },
      "floors": {
        "display": "cleared both",
        "cleared_both": true,
        "verdict": "PASS",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "19/21",
      "verdict_mix": {
        "PASS": 9,
        "UNCERTAIN": 4,
        "FAIL": 8
      },
      "self_consistency": "20/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "ok": 63
        },
        "ok": 63,
        "response_failures": 0,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "0/63",
        "response_failure_rate": {
          "successes": 0,
          "n": 63,
          "counts_of_n": "0/63",
          "point": 0.0,
          "wilson95": [
            0.0,
            0.0575
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 63,
          "n_calls": 63,
          "counts_of_n": "63/63",
          "rate": {
            "successes": 63,
            "n": 63,
            "counts_of_n": "63/63",
            "point": 1.0,
            "wilson95": [
              0.9425,
              1.0
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 0,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "0/0",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": false,
        "polarity": {
          "label": "STANDARD TRAP",
          "explanation": "think:false silently dropped the schema constraint while think:true held it. This is the documented majority behaviour: any caller sending think:false with a format schema is getting NO enforcement and no error."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": false,
            "per_repeat_enforced": [
              false,
              false
            ],
            "parsed_count": 0
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 0
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "0/63 response failures"
    },
    {
      "model": "muse-glimmer:30b-q8_0-dflash",
      "model_display": "muse-glimmer:30b-q8_0-dflash +think:false+reasoning-high",
      "posture": "think:false+reasoning-high",
      "posture_display": "think:false · high",
      "outcome": "EXPLORATORY",
      "outcome_why": "protocol fence: the contract probe failed; run anyway, unranked",
      "raw_emission": null,
      "cases_with_no_verdict": 0,
      "kills": {
        "counts_of_n": "12/12",
        "hits": 12,
        "n": 12,
        "floor": 11,
        "cleared": true,
        "wilson95": [
          0.7575,
          1.0
        ],
        "missed_case_ids": [],
        "no_verdict": 0
      },
      "preservation": {
        "counts_of_n": "7/9",
        "hits": 7,
        "n": 9,
        "floor": 8,
        "cleared": false,
        "wilson95": [
          0.4526,
          0.9368
        ],
        "missed_case_ids": [
          "c1-p-chk-01",
          "c1-p-piq-01"
        ],
        "no_verdict": 0
      },
      "floors": {
        "display": "under pres. floor",
        "cleared_both": false,
        "verdict": "FAIL",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "19/21",
      "verdict_mix": {
        "PASS": 7,
        "UNCERTAIN": 7,
        "FAIL": 7
      },
      "self_consistency": "20/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "ok": 63
        },
        "ok": 63,
        "response_failures": 0,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "0/63",
        "response_failure_rate": {
          "successes": 0,
          "n": 63,
          "counts_of_n": "0/63",
          "point": 0.0,
          "wilson95": [
            0.0,
            0.0575
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 63,
          "n_calls": 63,
          "counts_of_n": "63/63",
          "rate": {
            "successes": 63,
            "n": 63,
            "counts_of_n": "63/63",
            "point": 1.0,
            "wilson95": [
              0.9425,
              1.0
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 0,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "0/0",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": false,
        "polarity": {
          "label": "STANDARD TRAP",
          "explanation": "think:false silently dropped the schema constraint while think:true held it. This is the documented majority behaviour: any caller sending think:false with a format schema is getting NO enforcement and no error."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": false,
            "per_repeat_enforced": [
              false,
              false
            ],
            "parsed_count": 0
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 0
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "0/63 response failures"
    },
    {
      "model": "gemma4:26b",
      "model_display": "gemma4:26b",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "UNMEASURABLE",
      "outcome_why": "response-failure 56/63 = 88.9% exceeds the 10% ceiling for this leg",
      "raw_emission": null,
      "cases_with_no_verdict": 18,
      "kills": {
        "counts_of_n": "1/12",
        "hits": 1,
        "n": 12,
        "floor": 11,
        "cleared": false,
        "wilson95": [
          0.0149,
          0.3539
        ],
        "missed_case_ids": [],
        "no_verdict": 11
      },
      "preservation": {
        "counts_of_n": "1/9",
        "hits": 1,
        "n": 9,
        "floor": 8,
        "cleared": false,
        "wilson95": [
          0.0199,
          0.435
        ],
        "missed_case_ids": [
          "c1-p-nap-01"
        ],
        "no_verdict": 7
      },
      "floors": {
        "display": "under both floors",
        "cleared_both": false,
        "verdict": "FAIL",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "2/21",
      "verdict_mix": {
        "None": 18,
        "FAIL": 2,
        "PASS": 1
      },
      "self_consistency": "3/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "response_failure": 56,
          "ok": 7
        },
        "ok": 7,
        "response_failures": 56,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "56/63",
        "response_failure_rate": {
          "successes": 56,
          "n": 63,
          "counts_of_n": "56/63",
          "point": 0.8889,
          "wilson95": [
            0.788,
            0.9451
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 7,
          "n_calls": 63,
          "counts_of_n": "7/63",
          "rate": {
            "successes": 7,
            "n": 63,
            "counts_of_n": "7/63",
            "point": 0.1111,
            "wilson95": [
              0.0549,
              0.212
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 56,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 56,
              "recovered": 56,
              "counts_of_n": "56/56",
              "markdown_fenced": 56,
              "markdown_fence_counts_of_n": "56/56",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "56/56",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": true,
        "polarity": {
          "label": "SCHEMA HELD BOTH WAYS",
          "explanation": "The schema was enforced under think:false AND think:true. No polarity trap observed for this model on this ollama version."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 2
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "56/63 response failures · 18/21 cases unresolved · 56/56 fenced, diagnostic parse recovers 56/56"
    },
    {
      "model": "gemma4:12b",
      "model_display": "gemma4:12b",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "NOT-CARRIED",
      "outcome_why": "0 valid attempts across 63 call(s); raw emission published beside this row",
      "raw_emission": {
        "item_id": "c1-p-bez-01",
        "repeat": 1,
        "bucket": "response_failure",
        "done_reason": "stop",
        "content": "```json\n{\"verdict\":\"PASS\",\"why\":\"The quote states that two packs of thirty-two cards are shuffled together to be used as one, which equals a total of sixty-four cards.\"}\n```",
        "note": "The first of this row's 63 scored calls, verbatim. Every one of the 63 failed the strict parse the same way — valid JSON wrapped in a markdown fence — and the diagnostic parse recovers all 63. This is what a NOT-CARRIED row's emission looks like: the work is in there, and the contract is not."
      },
      "cases_with_no_verdict": 21,
      "kills": {
        "counts_of_n": "0/12",
        "hits": 0,
        "n": 12,
        "floor": 11,
        "cleared": false,
        "wilson95": [
          0.0,
          0.2425
        ],
        "missed_case_ids": [],
        "no_verdict": 12
      },
      "preservation": {
        "counts_of_n": "0/9",
        "hits": 0,
        "n": 9,
        "floor": 8,
        "cleared": false,
        "wilson95": [
          0.0,
          0.2992
        ],
        "missed_case_ids": [],
        "no_verdict": 9
      },
      "floors": {
        "display": "under both floors",
        "cleared_both": false,
        "verdict": "FAIL",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "0/21",
      "verdict_mix": {
        "None": 21
      },
      "self_consistency": "0/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "response_failure": 63
        },
        "ok": 0,
        "response_failures": 63,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "63/63",
        "response_failure_rate": {
          "successes": 63,
          "n": 63,
          "counts_of_n": "63/63",
          "point": 1.0,
          "wilson95": [
            0.9425,
            1.0
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 0,
          "n_calls": 63,
          "counts_of_n": "0/63",
          "rate": {
            "successes": 0,
            "n": 63,
            "counts_of_n": "0/63",
            "point": 0.0,
            "wilson95": [
              0.0,
              0.0575
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 63,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 63,
              "recovered": 63,
              "counts_of_n": "63/63",
              "markdown_fenced": 63,
              "markdown_fence_counts_of_n": "63/63",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "63/63",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": true,
        "polarity": {
          "label": "SCHEMA HELD BOTH WAYS",
          "explanation": "The schema was enforced under think:false AND think:true. No polarity trap observed for this model on this ollama version."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 2
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "63/63 response failures · 21/21 cases unresolved · 63/63 fenced, diagnostic parse recovers 63/63"
    },
    {
      "model": "qwen3.5:27b",
      "model_display": "qwen3.5:27b",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "RANKED",
      "outcome_why": "carried under protocol",
      "raw_emission": null,
      "cases_with_no_verdict": 0,
      "kills": {
        "counts_of_n": "11/12",
        "hits": 11,
        "n": 12,
        "floor": 11,
        "cleared": true,
        "wilson95": [
          0.6461,
          0.9851
        ],
        "missed_case_ids": [
          "c1-k-bez-h1"
        ],
        "no_verdict": 0
      },
      "preservation": {
        "counts_of_n": "9/9",
        "hits": 9,
        "n": 9,
        "floor": 8,
        "cleared": true,
        "wilson95": [
          0.7008,
          1.0
        ],
        "missed_case_ids": [],
        "no_verdict": 0
      },
      "floors": {
        "display": "cleared both",
        "cleared_both": true,
        "verdict": "PASS",
        "cleared_both_while_ranked": true
      },
      "items_resolved": "20/21",
      "verdict_mix": {
        "PASS": 10,
        "FAIL": 10,
        "UNCERTAIN": 1
      },
      "self_consistency": "21/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "ok": 63
        },
        "ok": 63,
        "response_failures": 0,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "0/63",
        "response_failure_rate": {
          "successes": 0,
          "n": 63,
          "counts_of_n": "0/63",
          "point": 0.0,
          "wilson95": [
            0.0,
            0.0575
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 63,
          "n_calls": 63,
          "counts_of_n": "63/63",
          "rate": {
            "successes": 63,
            "n": 63,
            "counts_of_n": "63/63",
            "point": 1.0,
            "wilson95": [
              0.9425,
              1.0
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 0,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "0/0",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": true,
        "polarity": {
          "label": "INVERTED TRAP",
          "explanation": "think:true dropped the schema constraint while think:false held it -- the gemma4-on-0.32.5 polarity. A caller who 'fixed' their schema problem by turning thinking ON would be breaking it."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": false,
            "per_repeat_enforced": [
              false,
              false
            ],
            "parsed_count": 0
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 2
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "0/63 response failures"
    },
    {
      "model": "qwen3.6:27b",
      "model_display": "qwen3.6:27b",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "RANKED",
      "outcome_why": "carried under protocol",
      "raw_emission": null,
      "cases_with_no_verdict": 0,
      "kills": {
        "counts_of_n": "11/12",
        "hits": 11,
        "n": 12,
        "floor": 11,
        "cleared": true,
        "wilson95": [
          0.6461,
          0.9851
        ],
        "missed_case_ids": [
          "c1-k-bez-c1"
        ],
        "no_verdict": 0
      },
      "preservation": {
        "counts_of_n": "9/9",
        "hits": 9,
        "n": 9,
        "floor": 8,
        "cleared": true,
        "wilson95": [
          0.7008,
          1.0
        ],
        "missed_case_ids": [],
        "no_verdict": 0
      },
      "floors": {
        "display": "cleared both",
        "cleared_both": true,
        "verdict": "PASS",
        "cleared_both_while_ranked": true
      },
      "items_resolved": "20/21",
      "verdict_mix": {
        "PASS": 10,
        "FAIL": 11
      },
      "self_consistency": "21/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "ok": 63
        },
        "ok": 63,
        "response_failures": 0,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "0/63",
        "response_failure_rate": {
          "successes": 0,
          "n": 63,
          "counts_of_n": "0/63",
          "point": 0.0,
          "wilson95": [
            0.0,
            0.0575
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 63,
          "n_calls": 63,
          "counts_of_n": "63/63",
          "rate": {
            "successes": 63,
            "n": 63,
            "counts_of_n": "63/63",
            "point": 1.0,
            "wilson95": [
              0.9425,
              1.0
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 0,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "0/0",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": true,
        "polarity": {
          "label": "SCHEMA HELD BOTH WAYS",
          "explanation": "The schema was enforced under think:false AND think:true. No polarity trap observed for this model on this ollama version."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 2
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "0/63 response failures"
    },
    {
      "model": "nemotron-3.5-lightning:30b-a3b",
      "model_display": "nemotron-3.5-lightning:30b-a3b",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "RANKED",
      "outcome_why": "carried under protocol",
      "raw_emission": null,
      "cases_with_no_verdict": 0,
      "kills": {
        "counts_of_n": "9/12",
        "hits": 9,
        "n": 12,
        "floor": 11,
        "cleared": false,
        "wilson95": [
          0.4677,
          0.9111
        ],
        "missed_case_ids": [
          "c1-k-bez-h1",
          "c1-k-chk-w1",
          "c1-k-dom-n1"
        ],
        "no_verdict": 0
      },
      "preservation": {
        "counts_of_n": "6/9",
        "hits": 6,
        "n": 9,
        "floor": 8,
        "cleared": false,
        "wilson95": [
          0.3542,
          0.8794
        ],
        "missed_case_ids": [
          "c1-p-chk-01",
          "c1-p-nap-01",
          "c1-p-piq-01"
        ],
        "no_verdict": 0
      },
      "floors": {
        "display": "under both floors",
        "cleared_both": false,
        "verdict": "FAIL",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "15/21",
      "verdict_mix": {
        "PASS": 9,
        "FAIL": 11,
        "UNCERTAIN": 1
      },
      "self_consistency": "21/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "ok": 63
        },
        "ok": 63,
        "response_failures": 0,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "0/63",
        "response_failure_rate": {
          "successes": 0,
          "n": 63,
          "counts_of_n": "0/63",
          "point": 0.0,
          "wilson95": [
            0.0,
            0.0575
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 63,
          "n_calls": 63,
          "counts_of_n": "63/63",
          "rate": {
            "successes": 63,
            "n": 63,
            "counts_of_n": "63/63",
            "point": 1.0,
            "wilson95": [
              0.9425,
              1.0
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 0,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "0/0",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": true,
        "polarity": {
          "label": "INVERTED TRAP",
          "explanation": "think:true dropped the schema constraint while think:false held it -- the gemma4-on-0.32.5 polarity. A caller who 'fixed' their schema problem by turning thinking ON would be breaking it."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": false,
            "per_repeat_enforced": [
              false,
              false
            ],
            "parsed_count": 0
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 2
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "0/63 response failures"
    },
    {
      "model": "granite4.1:30b-q8_0",
      "model_display": "granite4.1:30b-q8_0",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "RANKED",
      "outcome_why": "carried under protocol",
      "raw_emission": null,
      "cases_with_no_verdict": 0,
      "kills": {
        "counts_of_n": "11/12",
        "hits": 11,
        "n": 12,
        "floor": 11,
        "cleared": true,
        "wilson95": [
          0.6461,
          0.9851
        ],
        "missed_case_ids": [
          "c1-k-bez-h1"
        ],
        "no_verdict": 0
      },
      "preservation": {
        "counts_of_n": "6/9",
        "hits": 6,
        "n": 9,
        "floor": 8,
        "cleared": false,
        "wilson95": [
          0.3542,
          0.8794
        ],
        "missed_case_ids": [
          "c1-p-bkg-01",
          "c1-p-con-01",
          "c1-p-piq-01"
        ],
        "no_verdict": 0
      },
      "floors": {
        "display": "under pres. floor",
        "cleared_both": false,
        "verdict": "FAIL",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "17/21",
      "verdict_mix": {
        "PASS": 7,
        "FAIL": 12,
        "UNCERTAIN": 2
      },
      "self_consistency": "21/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "ok": 63
        },
        "ok": 63,
        "response_failures": 0,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "0/63",
        "response_failure_rate": {
          "successes": 0,
          "n": 63,
          "counts_of_n": "0/63",
          "point": 0.0,
          "wilson95": [
            0.0,
            0.0575
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 63,
          "n_calls": 63,
          "counts_of_n": "63/63",
          "rate": {
            "successes": 63,
            "n": 63,
            "counts_of_n": "63/63",
            "point": 1.0,
            "wilson95": [
              0.9425,
              1.0
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 0,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "0/0",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": true,
        "polarity": {
          "label": "INVERTED TRAP",
          "explanation": "think:true dropped the schema constraint while think:false held it -- the gemma4-on-0.32.5 polarity. A caller who 'fixed' their schema problem by turning thinking ON would be breaking it."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": false,
            "per_repeat_enforced": [
              false,
              false
            ],
            "parsed_count": 0
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 2
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "0/63 response failures"
    },
    {
      "model": "nemotron3:33b",
      "model_display": "nemotron3:33b",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "UNMEASURABLE",
      "outcome_why": "response-failure 33/63 = 52.4% exceeds the 10% ceiling for this leg",
      "raw_emission": null,
      "cases_with_no_verdict": 11,
      "kills": {
        "counts_of_n": "3/12",
        "hits": 3,
        "n": 12,
        "floor": 11,
        "cleared": false,
        "wilson95": [
          0.0889,
          0.5323
        ],
        "missed_case_ids": [
          "c1-k-chk-w1",
          "c1-k-con-s1"
        ],
        "no_verdict": 7
      },
      "preservation": {
        "counts_of_n": "5/9",
        "hits": 5,
        "n": 9,
        "floor": 8,
        "cleared": false,
        "wilson95": [
          0.2666,
          0.8112
        ],
        "missed_case_ids": [],
        "no_verdict": 4
      },
      "floors": {
        "display": "under both floors",
        "cleared_both": false,
        "verdict": "FAIL",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "8/21",
      "verdict_mix": {
        "None": 11,
        "PASS": 7,
        "FAIL": 2,
        "UNCERTAIN": 1
      },
      "self_consistency": "10/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "response_failure": 33,
          "ok": 30
        },
        "ok": 30,
        "response_failures": 33,
        "response_failure_denominator": 63,
        "response_failure_counts_of_n": "33/63",
        "response_failure_rate": {
          "successes": 33,
          "n": 63,
          "counts_of_n": "33/63",
          "point": 0.5238,
          "wilson95": [
            0.4027,
            0.6422
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 0,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 30,
          "n_calls": 63,
          "counts_of_n": "30/63",
          "rate": {
            "successes": 30,
            "n": 63,
            "counts_of_n": "30/63",
            "point": 0.4762,
            "wilson95": [
              0.3578,
              0.5973
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 33,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 33,
              "recovered": 33,
              "counts_of_n": "33/33",
              "markdown_fenced": 33,
              "markdown_fence_counts_of_n": "33/33",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "33/33",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": true,
        "polarity": {
          "label": "SCHEMA HELD BOTH WAYS",
          "explanation": "The schema was enforced under think:false AND think:true. No polarity trap observed for this model on this ollama version."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": true,
            "per_repeat_enforced": [
              true,
              true
            ],
            "parsed_count": 2
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 2
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "33/63 response failures · 11/21 cases unresolved · 33/33 fenced, diagnostic parse recovers 33/33"
    },
    {
      "model": "olmo-3.1:32b-think-q4_K_M",
      "model_display": "olmo-3.1:32b-think-q4_K_M",
      "posture": "think:false",
      "posture_display": "think:false",
      "outcome": "EXPLORATORY",
      "outcome_why": "protocol fence: the contract probe failed; run anyway, unranked",
      "raw_emission": null,
      "cases_with_no_verdict": 11,
      "kills": {
        "counts_of_n": "8/12",
        "hits": 8,
        "n": 12,
        "floor": 11,
        "cleared": false,
        "wilson95": [
          0.3906,
          0.8619
        ],
        "missed_case_ids": [],
        "no_verdict": 4
      },
      "preservation": {
        "counts_of_n": "1/9",
        "hits": 1,
        "n": 9,
        "floor": 8,
        "cleared": false,
        "wilson95": [
          0.0199,
          0.435
        ],
        "missed_case_ids": [
          "c1-p-six-01"
        ],
        "no_verdict": 7
      },
      "floors": {
        "display": "under both floors",
        "cleared_both": false,
        "verdict": "FAIL",
        "cleared_both_while_ranked": false
      },
      "items_resolved": "9/21",
      "verdict_mix": {
        "None": 11,
        "UNCERTAIN": 4,
        "PASS": 1,
        "FAIL": 5
      },
      "self_consistency": "10/21",
      "calls": {
        "n_calls": 63,
        "buckets": {
          "length_truncated": 34,
          "ok": 29
        },
        "ok": 29,
        "response_failures": 0,
        "response_failure_denominator": 29,
        "response_failure_counts_of_n": "0/29",
        "response_failure_rate": {
          "successes": 0,
          "n": 29,
          "counts_of_n": "0/29",
          "point": 0.0,
          "wilson95": [
            0.0,
            0.117
          ],
          "wilson_z": 1.96,
          "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
        },
        "length_truncated": 34,
        "transport_failures": 0,
        "denominator_note": "the response-failure denominator is ok + response_failure; length_truncated and transport failures sit OUTSIDE the ceiling (PREREG-C1 §Protocol), which is why a row can show 0/29 beside 63 calls"
      },
      "parse": {
        "strict": {
          "binds": true,
          "parsed": 29,
          "n_calls": 63,
          "counts_of_n": "29/63",
          "rate": {
            "successes": 29,
            "n": 63,
            "counts_of_n": "29/63",
            "point": 0.4603,
            "wilson95": [
              0.3431,
              0.5821
            ],
            "wilson_z": 1.96,
            "note": "counts bind; the interval is descriptive (PLAN v2: no percentage under N=30, and every C1 denominator is under 30)"
          },
          "note": "STRICT parse binds every verdict (PREREG-C1 §Protocol)"
        },
        "leak_stripped_diagnostic": {
          "binds": false,
          "strict_failures_total": 34,
          "by_bucket": {
            "response_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            },
            "length_truncated": {
              "strict_failures": 34,
              "recovered": 0,
              "counts_of_n": "0/34",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/34",
              "empty_content": 34,
              "eot_markers_removed": []
            },
            "transport_failure": {
              "strict_failures": 0,
              "recovered": 0,
              "counts_of_n": "0/0",
              "markdown_fenced": 0,
              "markdown_fence_counts_of_n": "0/0",
              "empty_content": 0,
              "eot_markers_removed": []
            }
          },
          "response_failure_recovery": "0/0",
          "note": "DIAGNOSTIC ONLY — binds nothing, changes no verdict, and is never added to a score. It answers one question: how many unusable answers were unusable only because of their wrapper? Split by bucket because a length-truncated call has no content to recover and is a different fact."
        }
      },
      "protocol_fence": {
        "fence_ok": false,
        "polarity": {
          "label": "SCHEMA HELD NEITHER WAY",
          "explanation": "The schema was not enforced under either think setting. Structured output from this model cannot be relied on at all here."
        },
        "probe_calls": 6,
        "configs": {
          "p1-schema-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": false,
            "per_repeat_enforced": [
              false,
              false
            ],
            "parsed_count": 0
          },
          "p2-schema-think-true": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": false,
            "per_repeat_enforced": [
              false,
              false
            ],
            "parsed_count": 0
          },
          "p3-bare-json-think-false": {
            "repeats": 2,
            "stable": true,
            "schema_enforced": null,
            "per_repeat_enforced": [
              null,
              null
            ],
            "parsed_count": 0
          }
        },
        "meaning": "the PROTOCOL fence — polarity probes. Not the markdown fence. A model whose schema constraint is unenforced runs anyway and is reported EXPLORATORY, never RANKED."
      },
      "failure_ledger": "0/29 response failures · 34/63 length-truncated, empty · 11/21 cases unresolved"
    }
  ],
  "n_rows": 10,
  "outcome_state_counts": {
    "EXPLORATORY": 3,
    "UNMEASURABLE": 2,
    "NOT-CARRIED": 1,
    "RANKED": 4
  },
  "cleared_both_floors_while_ranked": [
    "qwen3.5:27b",
    "qwen3.6:27b"
  ],
  "cleared_both_floors_but_unranked": [
    {
      "model": "muse-glimmer:30b-q8_0-dflash",
      "posture": "think:false+reasoning-medium",
      "state": "EXPLORATORY"
    }
  ],
  "contamination_caveat": "Published 2026-08-13. Models with a later training cutoff may have seen these sets. We author fresh sets each cycle; this one is not a standard, it is our kit, yours to reuse.",
  "not_a_standard": "Our own trials, our own hardware, for our own chairs. Not first independent numbers, and not a benchmark."
}
