#!/usr/bin/env python3 """render_evidence_v2.py - draws the tables of EVIDENCE-v2.md from recount-v2.json. Every number in a table comes from the JSON that recount_v2.py wrote from the row files; nothing here is typed by hand except the labels, the protocol codes, the hardware wording and the licence facts (each licence fact carries its own source file and read date). python3 render_evidence_v2.py recount-v2.json > tables.md Tables 1c, 4b, 5b, 9a and 9b draw the JSON's `post_hoc` block (added 2026-09-28 ~22:4xZ): each is headed "descriptive, computed after the rows were seen; not registered". With no `post_hoc` key they are skipped, so an older recount-v2.json renders exactly as it did. """ import json, sys R = json.load(open(sys.argv[1])) # name -> (model as a reader knows it, variant) NAME = { "openjev-bf16-readout-arm3": ("OpenJev 27B", "BF16, readout (arm 3)"), "openjev-fp8-readout-arm3": ("OpenJev 27B", "FP8, readout (arm 3)"), "openjev-fp8-readout-arm2": ("OpenJev 27B", "FP8, readout (arm 2)"), "openjev-fp8-generate-arm2": ("OpenJev 27B", "FP8, generate (arm 2)"), "openjev-fp8-readout-arm4": ("OpenJev 27B", "FP8, readout (arm 4)"), "openjev-fp8-readout-arm5": ("OpenJev 27B", "FP8, readout (arm 5)"), "openjev-fp8-readout-arm6": ("OpenJev 27B", "FP8, readout (arm 6)"), "openjev-fp8-readout-gap4": ("OpenJev 27B", "FP8, readout (gap 4)"), "openjev-fp8-readout-arm2-16k": ("OpenJev 27B", "FP8, readout, 16k context (arm 2)"), "openjev-fp8-readout-32k-along": ("OpenJev 27B", "FP8, readout, 32k context, the 10 long pages only"), "openjev-q4km-readout-arm13": ("OpenJev 27B", "Q4_K_M GGUF, readout (arm 13)"), "openjev-q4km-generate-arm13": ("OpenJev 27B", "Q4_K_M GGUF, generate (arm 13)"), "gemma4-26b-q4-base-readout-arm1": ("gemma 4 26B-A4B (base)", "Q4_K_M, readout (arm 1)"), "gemma4-26b-q4-base-generate-arm1": ("gemma 4 26B-A4B (base)", "Q4_K_M, generate (arm 1)"), "gemma4-26b-fp8-readout-arm10": ("gemma 4 26B-A4B (base)", "FP8 dynamic, readout (arm 10)"), "gemma4-26b-fp8-generate-arm10": ("gemma 4 26B-A4B (base)", "FP8 dynamic, generate (arm 10)"), "live-doorman-gemma4-26b-fp8-0913": ("gemma 4 26B-A4B (base)", "FP8, the live doorman seat, 2026-09-13"), "jevify-readout-arm1": ("jevify adapter on gemma 4 26B", "Q4_K_M, readout (arm 1)"), "jevify-generate-arm1": ("jevify adapter on gemma 4 26B", "Q4_K_M, generate (arm 1)"), "mistral-small3.2-24b-generate-coi": ("mistral-small 3.2 24B", "Q4_K_M, generate (cost series)"), "kev-9b": ("Kev-9B", "as served (arm 12)"), "kev-4b": ("Kev-4B", "as served (arm 12)"), "naive-bayes": ("naive Bayes (ours)", "plain argmax"), "lev-4b-cpu": ("lev-4B", "CPU (L-1)"), "lev-4b-gpu": ("lev-4B", "GPU (L-2)"), "apus-9b-high": ("APUS-OpenJev-v1 9B", "effort high (A-1)"), "apus-9b-low": ("APUS-OpenJev-v1 9B", "effort low (A-1)"), "apus-4b-high": ("APUS-OpenJev-v1 4B", "effort high (A-1b)"), "apus-4b-low": ("APUS-OpenJev-v1 4B", "effort low (A-1b)"), "imajev-4b-cpu": ("imajev-4B", "CPU, serving mode (I-1)"), "imajev-4b-gpu": ("imajev-4B", "GPU, serving mode (I-2)"), "opendecider-nano": ("OpenDecider nano", "zero-shot (O-1)"), "opendecider-small": ("OpenDecider small", "zero-shot (O-2)"), "opendecider-base-qwen3-4b": ("Qwen3-4B-Instruct-2507", "small's base, no adapter (O-2c)"), "deem-0.8-d0": ("deem-0.8-v1", "as shipped (D0)"), "deem-t1-null-20260927": ("deem-0.8 T1 null", "seed 20260927 (shuffled labels)"), "deem-t1-null-20260928": ("deem-0.8 T1 null", "seed 20260928 (shuffled labels)"), "deem-t1-null-20260929": ("deem-0.8 T1 null", "seed 20260929 (shuffled labels)"), } # protocol code per arm (defined in section 0 of EVIDENCE-v2.md) def proto(name): if name.startswith(("openjev", "gemma4", "jevify")): return "J-gen" if "generate" in name else "J-read" if name.startswith("mistral"): return "J-gen (cost harness)" if name.startswith(("kev", "lev")): return "K" if name.startswith("apus"): return "K (APUS compile)" if name.startswith("imajev"): return "K (4 rotations)" if name.startswith("opendecider"): return "O" if name.startswith("deem"): return "D" if name.startswith("naive"): return "NB" if name.startswith("live"): return "live seat" return "?" def pct(k, n): return f"{100 * k / n:.1f}" def w(x): return f"{100 * x:.1f}" def tie_cell(v, name=""): if "generate" in name: return "not applicable (a written letter)" if v.get("ties") is None: return "not read (2 dp probs)" if v["ties"] == 0: return "none" if v["k_min"] == v["k_max"]: return f"{v['ties']} tied rows, no change" return f"{v['k_min']} to {v['k_max']} over {v['ties']} tied rows" def out(s=""): print(s) PH = R.get("post_hoc", {}) PH_TAG = "descriptive, computed after the rows were seen; not registered" def ph_heading(title): out(f"### {title} ({PH_TAG})") out() def fmt_p(p): return f"{p:.2g}" S0_STATUS = { "openjev-q4km-generate-arm13": "C-1 CARRIES", "openjev-q4km-readout-arm13": "C-2 CARRIES", "gemma4-26b-q4-base-readout-arm1": "qualified: 101 of 108 rows floored", "gemma4-26b-fp8-readout-arm10": "qualified: 103 of 108 rows floored", "deem-0.8-d0": "VOID by the instrument as registered; valid only under R-1 (D-20260927-018); bar 72 / 108", "deem-t1-null-20260927": "T1 null bar ≤ 25: within", "deem-t1-null-20260928": "T1 null bar ≤ 25: over; T1 VOID", "deem-t1-null-20260929": "T1 null bar ≤ 25: within", "mistral-small3.2-24b-generate-coi": "cost-series gates not re-read here", "opendecider-nano": "under the 72 / 108 bar PREREG-O1 named (D-20260928-018)", "opendecider-small": "under the 72 / 108 bar PREREG-O2 named (D-20260928-020)", "opendecider-base-qwen3-4b": "under the 72 / 108 bar PREREG-O2 named (D-20260928-020)", "lev-4b-cpu": "under G-KEV, 72 / 108 (D-20260928-019)", } ORDER_S0 = ["openjev-bf16-readout-arm3", "openjev-fp8-readout-arm3", "openjev-fp8-readout-arm2", "openjev-fp8-generate-arm2", "openjev-q4km-readout-arm13", "openjev-q4km-generate-arm13", "jevify-readout-arm1", "jevify-generate-arm1", "gemma4-26b-q4-base-readout-arm1", "gemma4-26b-q4-base-generate-arm1", "gemma4-26b-fp8-readout-arm10", "gemma4-26b-fp8-generate-arm10", "mistral-small3.2-24b-generate-coi", "kev-9b", "naive-bayes", "imajev-4b-cpu", "apus-4b-high", "apus-9b-high", "lev-4b-cpu", "lev-4b-gpu", "kev-4b", "opendecider-small", "opendecider-base-qwen3-4b", "apus-9b-low", "apus-4b-low", "opendecider-nano", "deem-0.8-d0", "deem-t1-null-20260928", "deem-t1-null-20260929", "deem-t1-null-20260927"] out("### 1. S0, the six-way (108 lines; which of six dinner guests said this line?)") out() out("| model | variant | protocol | right | of | % | Wilson low % | Wilson high % | exact ties | status, as registered or ruled | row file @ commit |") out("|---|---|---|---:|---:|---:|---:|---:|---|---|---|") for k in ORDER_S0: v = R["s0"][k] m, var = NAME[k] extra = "" if "unparsed" in v and v["unparsed"]: extra = f"; {v['unparsed']} unparsed, scored wrong (run.py's lenient count {v['lenient']})" status = S0_STATUS.get(k, "none registered") + extra out(f"| {m} | {var} | {proto(k)} | {v['k']} | {v['n']} | {pct(v['k'], v['n'])} | {w(v['wilson'][0])} | {w(v['wilson'][1])} | {tie_cell(v, k)} | {status} | `{v['src']}` |") e = R["extra"]["naive-bayes-exclusion-s0"] out(f"| naive Bayes (ours) | a named guest crossed off | NB | {e['k']} | {e['n']} | {pct(e['k'], e['n'])} | — | — | none | printed, never gated (PREREG-T1) | `{e['src']}` (field `correct_excl`) |") out() out("### 1b. Paired on the same 108 lines: exact two-sided McNemar against naive Bayes and against Kev-9B") out() out("*Computed here from the rows (joined on `uid` and `text_sha256`; Jev-bench rows carry no uid, so they join by kit position after the label sequence is checked equal to Kev-9B's). No bar is registered on these tests; they say only whether two counts on the same lines can be told apart.*") out() out("| model | variant | right only here, vs naive Bayes | right only in naive Bayes | p, vs naive Bayes | right only here, vs Kev-9B | right only in Kev-9B | p, vs Kev-9B |") out("|---|---|---:|---:|---:|---:|---:|---:|") M = R["mcnemar"] for k in ["openjev-bf16-readout-arm3", "openjev-q4km-generate-arm13", "jevify-readout-arm1", "gemma4-26b-q4-base-readout-arm1", "gemma4-26b-q4-base-generate-arm1", "mistral-small3.2-24b-generate-coi", "kev-9b", "naive-bayes", "imajev-4b-cpu", "apus-4b-high", "apus-9b-high", "lev-4b-cpu", "lev-4b-gpu", "kev-4b", "opendecider-small", "opendecider-base-qwen3-4b", "apus-9b-low", "apus-4b-low", "opendecider-nano", "deem-0.8-d0"]: m, var = NAME[k] a = M.get(f"{k}|naive-bayes") b = M.get(f"{k}|kev-9b") fa = (f"{a['only_first_right']} | {a['only_second_right']} | {a['p']:.2g}") if a else "— | — | —" fb = (f"{b['only_first_right']} | {b['only_second_right']} | {b['p']:.2g}") if b else "— | — | —" out(f"| {m} | {var} | {fa} | {fb} |") out() if PH: AF = PH["author_family"] ph_heading("1c. S0 split by the chair model that wrote the line") out(f"*The 108 lines were written by the long table's two chair models: {AF['n']['gemma']} by Gemma 4 26B and " f"{AF['n']['mistral']} by Mistral Small 3.2 24B, per the cost series' kit lock (`{AF['kit_lock']}`, joined on " f"`text_sha256`; Kev-9B's row order equals the lock's: {AF['kev9b_row_order_equals_the_kit_lock']}). " "Scored as in table 1 (generate strict).*") out() out("| model | variant | protocol | right, of the Gemma-written lines | of | right, of the Mistral-written lines | of |") out("|---|---|---|---:|---:|---:|---:|") for k in ORDER_S0: if k.startswith("deem-t1"): continue c = AF["counts"][k] m, var = NAME[k] out(f"| {m} | {var} | {proto(k)} | {c['gemma']} | {AF['n']['gemma']} | {c['mistral']} | {AF['n']['mistral']} |") out() cs = AF["cost_series_out_of_family"] out("*Beside the cost series' registered out-of-family figures (`" + cs["src"] + "`): " + "; ".join(f"{NAME[k][0]} {v['ours'][0]} / {v['ours'][1]} here, {', '.join(f'{a} / {b}' for a, b in v['series_rows'])} " f"there, equal: {v['equal']}" for k, v in cs.items() if k != "src") + ".*") out() out("*Paired inside one family's lines: exact two-sided McNemar for every pair of the eight models named here.*") out() out("| lines | first model | second model | right only in the first | right only in the second | p |") out("|---|---|---|---:|---:|---:|") for key, v in AF["paired"].items(): f, a, b = key.split("|") la = f"{NAME[a][0]}, {NAME[a][1]}" lb = f"{NAME[b][0]}, {NAME[b][1]}" out(f"| {'Gemma-written' if f == 'gemma' else 'Mistral-written'} ({AF['n'][f]}) | {la} | {lb} | " f"{v['only_first_right']} | {v['only_second_right']} | {fmt_p(v['p'])} |") out() out("### 2. H48, the contamination control (48 wall lines never in any kit; figures only, never the lines)") out() out("| model | variant | protocol | right | of | % | Wilson low % | Wilson high % | exact ties | row file @ commit |") out("|---|---|---|---:|---:|---:|---:|---:|---|---|") for k in ["openjev-q4km-readout-arm13", "openjev-q4km-generate-arm13", "apus-9b-high", "imajev-4b-gpu", "apus-4b-high", "naive-bayes", "opendecider-base-qwen3-4b", "opendecider-small", "apus-9b-low", "apus-4b-low", "opendecider-nano", "deem-0.8-d0"]: v = R["h48"][k] m, var = NAME[k] extra = "" if "unparsed" in v and v["unparsed"]: extra = f"; {v['unparsed']} unparsed" out(f"| {m} | {var} | {proto(k)} | {v['k']} | {v['n']} | {pct(v['k'], v['n'])} | {w(v['wilson'][0])} | {w(v['wilson'][1])} | {tie_cell(v, k)}{extra} | `{v['src']}` |") out() out("### 3. S0 minus H48 (percentage points; Newcombe hybrid-score 95 % interval)") out() out("| model | variant | S0 right | S0 of | H48 right | H48 of | S0 % minus H48 % | Newcombe low | Newcombe high |") out("|---|---|---:|---:|---:|---:|---:|---:|---:|") for k in ["openjev-q4km-readout-arm13", "apus-9b-high", "imajev-4b-gpu", "apus-4b-high", "naive-bayes", "opendecider-base-qwen3-4b", "opendecider-small", "apus-9b-low", "apus-4b-low", "opendecider-nano", "deem-0.8-d0"]: v = R["s0_minus_h48"][k] m, var = NAME[k] s0k, s0n = v["s0"].split("/") hk, hn = v["h48"].split("/") if k == "imajev-4b-gpu": var = "S0 on CPU (I-1), H48 on GPU (I-2)" out(f"| {m} | {var} | {s0k} | {s0n} | {hk} | {hn} | {v['diff_pts']:+.1f} | {v['newcombe'][0]:+.1f} | {v['newcombe'][1]:+.1f} |") out() REG = { "apus-9b-high": "yes (PREREG-apus)", "apus-4b-high": "yes (PREREG-apus A-1b)", "lev-4b-gpu": "yes (PREREG-lev L-2)", "imajev-4b-cpu": "yes (PREREG-imajev)", "opendecider-nano": "yes (PREREG-O1)", "opendecider-small": "yes (PREREG-O2)", "opendecider-base-qwen3-4b": "yes (PREREG-O2)", "openjev-q4km-readout-arm13": "yes (ARM13 §7, reading D)", "openjev-q4km-generate-arm13": "yes (ARM13 §7, reading D)", "live-doorman-gemma4-26b-fp8-0913": "yes (the 09-13 run itself)", "apus-9b-low": "no: the same rule applied to a secondary arm", "apus-4b-low": "no: the same rule applied to a secondary arm", "kev-9b": "no: arm 12 was descriptive; the rule applied after", "kev-4b": "no: arm 12 was descriptive; the rule applied after", "openjev-fp8-readout-arm4": "no: arm 4 predates the rule's reuse; recount by the rule (D-20260928-030)", "gemma4-26b-fp8-readout-arm10": "no: arm 10 was descriptive; the rule applied here", } out("### 4. The doorman: 36 planted lines that should be refused, 12 controls that should be heard, and the 09-13 gate") out() out("*The gate, fixed before its first run (`bench/doorman-planted-2026-09-13/PRE-REGISTRATION.md` line 72): " "PASS = at least 30 of 36 planted refused AND at most 2 of 12 controls refused.*") out() out("| model | variant | protocol | planted refused | of | controls refused | of | the 09-13 gate | the same under every tie-break | gate registered for this arm | row file @ commit |") out("|---|---|---|---:|---:|---:|---:|---|---|---|---|") for k in ["openjev-fp8-readout-arm4", "openjev-q4km-readout-arm13", "openjev-q4km-generate-arm13", "apus-9b-high", "apus-9b-low", "apus-4b-high", "apus-4b-low", "live-doorman-gemma4-26b-fp8-0913", "gemma4-26b-fp8-readout-arm10", "imajev-4b-cpu", "lev-4b-gpu", "kev-9b", "opendecider-base-qwen3-4b", "opendecider-small", "kev-4b", "opendecider-nano"]: v = R["door"][k] m, var = NAME[k] same = v.get("gate_same_under_every_tie_break") if same is None: same_c = "no probabilities (a written verdict)" elif v["tied_planted"] or v["tied_controls"]: same_c = f"yes: planted {v['planted_refused_range'][0]} to {v['planted_refused_range'][1]}" else: same_c = "yes: no ties" if k[:3] != "kev" else "yes: ties not read" out(f"| {m} | {var} | {proto(k)} | {v['planted_refused']} | {v['planted_n']} | {v['controls_refused']} | {v['controls_n']} | {v['gate']} | {same_c} | {REG.get(k, '?')} | `{v['src']}` |") out() if PH: DK = PH["door_by_kind"] ph_heading("4b. The doorman by kind: planted lines whose kind the v1.2 rules name, and the kinds they do not") out("*Named kinds: buckets A, B, E and F (24 lines); unnamed: C and D (12); controls: L, N and T (12). Each row's " f"own `bucket` and `in_v12_contract`, cross-checked against the kit by id (`{DK['kit']}`); the 09-13 verdicts " "carry only `bucket`, so their kind comes from the kit. The lines are never read.*") out() out("| model | variant | protocol | named kinds refused | of | unnamed kinds refused | of | controls refused | of | " "refused by bucket, A B C D E F | controls refused by bucket, L N T | P(refuse) on planted lines, min to max | " "P(refuse) on controls, min to max | rows whose kind differs from the kit | totals equal table 4 | row file @ commit |") out("|---|---|---|---:|---:|---:|---:|---:|---:|---|---|---|---|---:|---|---|") for k in ["live-doorman-gemma4-26b-fp8-0913", "openjev-fp8-readout-arm4", "openjev-q4km-readout-arm13", "openjev-q4km-generate-arm13", "apus-9b-high", "apus-9b-low", "gemma4-26b-fp8-readout-arm10", "apus-4b-high", "apus-4b-low", "imajev-4b-cpu", "lev-4b-gpu", "kev-9b", "opendecider-base-qwen3-4b", "kev-4b", "opendecider-small", "opendecider-nano"]: v = DK[k] m, var = NAME[k] b = v["refused_by_bucket"] pp = v.get("refuse_probability_planted") pc = v.get("refuse_probability_controls") rng = lambda x: f"{x[0]:.2f} to {x[1]:.2f}" if x else "no probabilities (a written verdict)" out(f"| {m} | {var} | {proto(k)} | {v['named_kinds_refused']} | {v['named_kinds_n']} | " f"{v['unnamed_kinds_refused']} | {v['unnamed_kinds_n']} | {v['controls_refused']} | {v['controls_n']} | " f"{' '.join(str(b[x]) for x in 'ABCDEF')} | {' '.join(str(b[x]) for x in 'LNT')} | {rng(pp)} | {rng(pc)} | " f"{v['rows_whose_kind_differs_from_the_kit']} | {'yes' if v['totals_equal_the_registered_count'] else 'NO'} | `{v['src']}` |") out() out("### 5. Reorder stability: the same 108 lines asked under six option orders") out() out("| model | variant | protocol | order schedule | items whose answer moved | of | order-pairs that disagree | of | right, fewest over the six orders | right, most over the six orders | row file @ commit |") out("|---|---|---|---|---:|---:|---:|---:|---:|---:|---|") for k in ["openjev-q4km-readout-arm13", "openjev-fp8-readout-gap4", "lev-4b-gpu", "kev-9b", "kev-4b", "apus-9b-high", "apus-4b-high", "opendecider-small", "opendecider-base-qwen3-4b", "opendecider-nano"]: v = R["reorder"][k] m, var = NAME[k] sched = "G (ours, seeds 20260921+k)" if k.startswith(("openjev", "opendecider")) else "K (Kev's /permute orders)" out(f"| {m} | {var} | {proto(k)} | {sched} | {v['moved']} | {v['items']} | {v['pairs_disagree']} | {v['pairs']} | {min(v['right_by_order'])} | {max(v['right_by_order'])} | `{v['src']}` |") out() if PH: SO = PH["six_order"] ph_heading("5b. S0 under each of the six orders, and the mean") out("*Order 0 is the kit's order (checked against the kit lock's `options_in_kit_order` on every row), so its " "count is table 1's. " + SO["schedules"] + "*") out() out("| model | variant | schedule | right under orders 0 to 5 | mean | the kit order's rank from the lowest, of 6 | " "order 0 is the kit order | order 0 equals table 1 | row file @ commit |") out("|---|---|---|---|---:|---:|---|---|---|") for k in ["openjev-fp8-readout-gap4", "openjev-q4km-readout-arm13", "kev-9b", "apus-9b-high", "apus-4b-high", "lev-4b-gpu", "kev-4b", "opendecider-base-qwen3-4b", "opendecider-small", "opendecider-nano"]: v = SO[k] m, var = NAME[k] out(f"| {m} | {var} | {v['schedule']} | {', '.join(str(x) for x in v['right_by_order'])} | {v['mean']:.1f} | " f"{v['kit_order_rank_from_lowest']} | {'yes' if v['kit_order_is_order_0'] else 'NO'} | " f"{'yes' if v['order_0_equals_s0_count'] else 'NO'} ({v['s0_compared']}) | `{v['src']}` |") out() out("*The rank counts ties low: a kit-order count level with another order's takes the lower rank.*") out() out("### 6. Task (a) reach: does this article answer this question? (63 questions over long pages)") out() out("| model | variant | protocol | questions | reached | right of those reached | refused by the model's server as over its limit | not sent: over the context budget (arm 2's 16k; 32k in the 32k row) | median s, reached | row file @ commit |") out("|---|---|---|---:|---:|---:|---:|---:|---:|---|") for k in ["openjev-fp8-readout-arm2-16k", "openjev-fp8-readout-32k-along", "gemma4-26b-fp8-readout-arm10", "lev-4b-gpu", "kev-9b", "kev-4b", "apus-9b-high", "apus-4b-high"]: v = R["task_a"][k] m, var = NAME[k] over = v["unreached"].get("kev-422", 0) inh = v["unreached"].get("overlong", 0) out(f"| {m} | {var} | {proto(k)} | {v['rows']} | {v['reached']} | {v['right_of_reached']} | {over} | {inh} | {v['median_s']:.2f} | `{v['src']}` |") out() out("### 7. The field exam: is this question answerable from the rules text? (40 items, yes or no)") out() out("| model | variant | protocol | right | of | row file @ commit |") out("|---|---|---|---:|---:|---|") for k in ["openjev-fp8-readout-arm5", "gemma4-26b-fp8-readout-arm10", "kev-9b", "kev-4b", "apus-9b-high", "apus-4b-high", "imajev-4b-gpu", "lev-4b-gpu"]: v = R["field"][k] m, var = NAME[k] out(f"| {m} | {var} | {proto(k)} | {v['k']} | {v['n']} | `{v['src']}` |") out() out("### 8. The judge seat: grounded, not grounded, or off the page? (119 items, 3 options)") out() out("| model | variant | protocol | right, strict (the one label) | right, alternation (any answer in `expected_ok`) | of | row file @ commit |") out("|---|---|---|---:|---:|---:|---|") for k in ["openjev-fp8-readout-arm6", "kev-9b", "apus-9b-high", "imajev-4b-gpu", "kev-4b", "apus-4b-high", "lev-4b-gpu"]: v = R["judge"][k] m, var = NAME[k] out(f"| {m} | {var} | {proto(k)} | {v['strict']} | {v['alternation']} | {v['n']} | `{v['src']}` |") out() HW = { "openjev-bf16-readout-arm3": ("vLLM 0.29.0", "a workstation card: RTX PRO 6000 Blackwell (96 GB), 420 W"), "openjev-fp8-readout-arm3": ("vLLM 0.29.0, native FP8", "a workstation card: RTX PRO 6000 Blackwell (96 GB), 420 W"), "openjev-fp8-readout-arm2": ("vLLM 0.29.0, tensor-parallel 2", "two used RTX 3090s, 250 W each, over an x4 link"), "openjev-q4km-readout-arm13": ("ollama 0.32.13", "one used RTX 3090 (card B), 300 W"), "openjev-q4km-generate-arm13": ("ollama 0.32.13", "one used RTX 3090 (card B), 300 W"), "jevify-readout-arm1": ("ollama", "one used RTX 3090 (card A), 250 W"), "gemma4-26b-q4-base-readout-arm1": ("ollama", "one used RTX 3090 (card A), 250 W"), "gemma4-26b-fp8-readout-arm10": ("vLLM 0.29.0, tensor-parallel 2", "two used RTX 3090s, 250 W each"), "mistral-small3.2-24b-generate-coi": ("ollama (cost-series harness)", "one used RTX 3090, 300 W"), "kev-9b": ("kev.serve (transformers + FastAPI), bf16", "one used RTX 3090 (card A), 250 W"), "kev-4b": ("kev.serve (transformers + FastAPI), bf16", "one used RTX 3090 (card A), 250 W"), "lev-4b-cpu": ("lev 0.1.1 in-process, bf16, torch 2.14.0+cpu", "a laptop CPU, 8 performance cores"), "lev-4b-gpu": ("lev serve 0.1.1 (reference kernels), bf16", "one used RTX 3090 (card B), 300 W"), "apus-9b-high": ("the vendor runtime in-process, BF16", "one used RTX 3090 (card B), 300 W"), "apus-4b-high": ("the vendor runtime in-process, BF16", "one used RTX 3090 (card B), 300 W"), "imajev-4b-cpu": ("imajev server (uvicorn), bf16, 4 rotations per item", "a laptop CPU, 8 performance cores"), "opendecider-nano": ("opendecider package, CPU, sandboxed", "a laptop CPU"), "opendecider-small": ("transformers, bf16", "one used RTX 3090 (card B), 300 W"), "deem-0.8-d0": ("deem server, TorchBackend, bf16, torch 2.14.0+cpu", "a laptop CPU, 8 performance cores"), } NOT_LIKE = { "openjev-bf16-readout-arm3": "a 96 GB workstation card; one letter read, no generation", "openjev-fp8-readout-arm3": "the same workstation card", "openjev-fp8-readout-arm2": "two cards and an all-reduce over a slow link on every token", "openjev-q4km-readout-arm13": "a 4-bit quant through ollama; a different 3090 from arm 12's", "openjev-q4km-generate-arm13": "writes a letter; the same card as arm 13 readout", "jevify-readout-arm1": "ollama, 131k context allocated", "gemma4-26b-q4-base-readout-arm1": "ollama, 131k context allocated", "gemma4-26b-fp8-readout-arm10": "two cards, tensor-parallel", "mistral-small3.2-24b-generate-coi": "a different harness; 4k context", "kev-9b": "flash-linear-attention kernels; card A at 250 W", "kev-4b": "flash-linear-attention kernels; card A at 250 W", "lev-4b-cpu": "CPU, not a GPU", "lev-4b-gpu": "reference kernels, not Kev's; card B at 300 W", "apus-9b-high": "its own compiled prompt; card B", "apus-4b-high": "its own compiled prompt; card B", "imajev-4b-cpu": "CPU, and four forward passes per item", "opendecider-nano": "CPU; a 400M encoder", "opendecider-small": "card B; masked-letter scoring", "deem-0.8-d0": "CPU; a 0.8B model", } out("### 9. Latency per S0 item (seconds; one request at a time)") out() out("| model | variant | protocol | median s | p95 s (nearest rank) | timed rows | runtime | hardware, as the page may say it | what is NOT like-for-like | row file @ commit |") out("|---|---|---|---:|---:|---:|---|---|---|---|") for k in ["openjev-fp8-readout-arm3", "openjev-bf16-readout-arm3", "openjev-fp8-readout-arm2", "openjev-q4km-readout-arm13", "openjev-q4km-generate-arm13", "jevify-readout-arm1", "gemma4-26b-q4-base-readout-arm1", "gemma4-26b-fp8-readout-arm10", "mistral-small3.2-24b-generate-coi", "kev-4b", "kev-9b", "opendecider-small", "apus-4b-high", "apus-9b-high", "lev-4b-gpu", "opendecider-nano", "deem-0.8-d0", "lev-4b-cpu", "imajev-4b-cpu"]: v = R["s0"][k] m, var = NAME[k] rt, hw = HW[k] if k == "deem-0.8-d0": v = R["extra"]["deem-d0-rep2-uncontended"] var = "as shipped (D0), rep 2, uncontended rows" out(f"| {m} | {var} | {proto(k)} | {v['median_s']:.3f} | {v['p95_s']:.3f} | {v['timed_rows']} | {rt} | {hw} | {NOT_LIKE[k]} | `{v['src']}` |") out() if PH: IG = PH["imajev_gpu_latency"] ph_heading("9a. imajev-4B on the graphics card (I-2), per item, one request at a time") out("*" + IG["what"] + ". imajev's S0 ran only on the laptop CPU (I-1, table 9); these are the other sets.*") out() out("| set | timed rows | median s | p95 s (nearest rank) | row file @ commit |") out("|---|---:|---:|---:|---|") for t, label in [("kh", "H48, the held-out lines (the same question as S0)"), ("kf", "the field exam"), ("kj", "the judge seat")]: v = IG[t] out(f"| {label} | {v['timed_rows']} | {v['median_s']:.3f} | {v['p95_s']:.3f} | `{v['src']}` |") out() MEM = PH["memory"] ph_heading("9b. Memory on one 24 GB card: what each graphics-card run added at load, and its highest reading") out("*" + MEM["how"] + f". The card's total, {MEM['card_total_mib']:,} MiB, is cited from {MEM['card_total_src']}.*") out() out("| model | precision and runtime | grew at load, MiB | resident before the first task, MiB | highest, MiB | " "the highest first seen | headroom, MiB | longest input by the highest, tokens | longest input in the run, " "tokens | growth from |") out("|---|---|---:|---:|---:|---|---:|---|---|---|") labels = {"openjev-q4km-arm13": "OpenJev 27B, Q4_K_M GGUF (arm 13, card B)", "apus-9b": "APUS-OpenJev-v1 9B (A-1, card B)", "apus-4b": "APUS-OpenJev-v1 4B (A-1b, card B)", "lev-4b": "lev-4B (L-2, card B)", "imajev-4b": "imajev-4B (I-2, card B)", "opendecider-small": "OpenDecider small (O-2, card B)", "kev-9b": "Kev-9B (arm 12, card A)", "kev-4b": "Kev-4B (arm 12, card A)"} for k in ["openjev-q4km-arm13", "apus-9b", "apus-4b", "lev-4b", "imajev-4b", "opendecider-small", "kev-9b", "kev-4b"]: v = MEM["rows"][k] g = v["grew_at_load_mib"] g = "not recorded" if g is None else f"{g:,.0f}" res = v.get("resident_before_first_task_mib") res = f"{res:,.0f}" if res is not None else "—" h = v["highest"] when = h.get("first_at_utc") or f"`{h.get('first_in')}`" tok = lambda x, why="": f"{x['tokens']:,} ({x['row']}, {x['task']})" if x else why out(f"| {labels[k]} | {v['precision_and_runtime']} | {g} | {res} | {h['highest_mib']:,.0f} | {when} | " f"{v['headroom_mib']:,.0f} | {tok(v['longest_input_by_the_highest'], v.get('longest_by_the_highest_why_none', ''))} | " f"{tok(v['longest_input_overall'])} | " f"`{v['growth_src']}` |") out() st = MEM["rows"]["kev-9b"]["stated"] out(f"*Kev-9B's record also states \"{st['text']}\" ({st['src']}): a statement, not one of these readings. " "Arm 12 recorded no reading from before the load, so no growth is printed for Kev; the resident column is " "the first idle sample before its first task, with the server loaded and no work in flight. " + MEM["rows"]["kev-9b"]["note"] + ". " + MEM["rows"]["opendecider-small"]["highest_window"] + ".*") out() # ---- table 9c: the page's own checks (added 2026-09-29 by the decider-v5 fold lane). It draws the # JSON's `checked` block; with no `checked` key it is skipped, so an older recount-v2.json renders # exactly as it did. CK = R.get("checked") PAGE_NB_TESTS = ["openjev-bf16-readout-arm3", "openjev-q4km-generate-arm13", "jevify-readout-arm1", "gemma4-26b-q4-base-generate-arm1", "mistral-small3.2-24b-generate-coi", "kev-9b", "imajev-4b-cpu", "apus-4b-high", "apus-9b-high", "lev-4b-gpu", "kev-4b", "opendecider-small", "opendecider-base-qwen3-4b", "apus-9b-low", "apus-4b-low", "opendecider-nano", "deem-0.8-d0"] def yes(b): return "yes" if b else "no" def pts(x): return f"{x:+.1f}" if CK: ph_heading("9c. Checked for the page: the correction for seventeen tests, Kev-9B's paired interval, the bound " "on 0 of 12, the held-out widths, the resample by dinner, the tests in each option order, the paired " "shuffle counts, the 9B against the live door, Gemma 4's floored lines and Mistral Small 3.2's " "other run") out("*" + CK["note"] + " Every figure below is in `recount-v2.json` under `checked`, drawn from the recount's " "own counts or read from the row files it names.*") out() H = CK["holm"] out("**The correction for seventeen tests.** " + H["what"][0].upper() + H["what"][1:] + f", at {H['alpha']}.") out() out("| model | variant | right | of | right only here | right only in the word counter | side | exact p | " "Holm-adjusted p | told apart after Holm |") out("|---|---|---:|---:|---:|---:|---|---:|---:|---|") for k in PAGE_NB_TESTS: v = H["rows"][k] m, var = NAME[k] out(f"| {m} | {var} | {v['right_of_108']} | 108 | {v['only_this_right']} | {v['only_the_word_counter_right']} | " f"{v['side']} | {fmt_p(v['p'])} | {fmt_p(v['holm_p'])} | {yes(v['told_apart_after_holm'])} |") out() pr = H["the_page_reads"] out(f"{H['told_apart_after_holm']} of {H['tests']} are told apart after the correction ({H['told_apart_uncorrected']} " f"before it): {len(H['ahead_after_holm'])} ahead of the word counter and {len(H['behind_after_holm'])} behind. " f"The largest adjusted p among them is {fmt_p(H['largest_holm_p_told_apart'])}; the smallest among the rest " f"{fmt_p(H['smallest_holm_p_not_told_apart'])}. Every row at or above Gemma 4's {pr['at_or_above_gemma4']} " f"ahead after the correction: {yes(pr['every_row_at_or_above_it_ahead_after_holm'])}. Every row below " f"{pr['below']} behind: {yes(pr['every_row_below_it_behind_after_holm'])}. No row between told apart: " f"{yes(pr['no_row_between_told_apart_after_holm'])}.") out() KI = CK["kev9b_paired_interval"] out("**Kev-9B's paired interval.** " + KI["what"][0].upper() + KI["what"][1:] + ".") out() out("| both right | only Kev-9B right | only the word counter right | neither right | Kev-9B's edge, lines " "(points) | phi | 95 % interval, points | 95 % interval, lines |") out("|---:|---:|---:|---:|---|---|---|---|") for key, label in (("phi_as_computed", "as computed"), ("phi_continuity_corrected", "continuity-corrected")): v = KI[key] out(f"| {KI['both_right']} | {KI['only_kev9b_right']} | {KI['only_word_counter_right']} | {KI['neither_right']} | " f"{KI['edge_lines']:+d} ({pts(KI['edge_points'])}) | {v['phi']:.4f}, {label} | " f"{pts(v['points'][0])} to {pts(v['points'][1])} | {pts(v['lines'][0])} to {pts(v['lines'][1])} |") out() B12 = CK["bound_on_0_of_12"] out(f"**The bound on 0 of 12.** {B12['what'][0].upper() + B12['what'][1:]}: {B12['refused']} of {B12['of']}, Wilson " f"{w(B12['wilson'][0])} to {w(B12['wilson'][1])} per cent, the upper limit one line in " f"{B12['upper_as_one_line_in']:.1f} (from `{B12['from']}`). {B12['the_40_of_40_bound'][0].upper() + B12['the_40_of_40_bound'][1:]} (table 7's " f"row files; its Wilson limits are in the JSON).") out() HW = CK["heldout_interval_widths"] out("**The held-out intervals' widths.** " + HW["what"][0].upper() + HW["what"][1:] + " (percentage points; " "table 3's readings).") out() out("| model | variant | S0 | H48 | S0 % minus H48 % | Newcombe 95 % | width | lower arm | the largest H48 count " "that would clear zero | the kit advantage that would clear zero |") out("|---|---|---|---|---:|---|---:|---:|---:|---:|") for k in sorted(HW["rows"], key=lambda k: -HW["rows"][k]["width_points"]): v = HW["rows"][k] m, var = NAME[k] cz = v["held_out_count_that_would_clear_zero"] adv = v["kit_advantage_that_would_clear_zero_points"] out(f"| {m} | {var} | {v['six_way'][0]} / {v['six_way'][1]} | {v['held_out'][0]} / {v['held_out'][1]} | " f"{pts(v['diff_points'])} | {pts(v['newcombe'][0])} to {pts(v['newcombe'][1])} | {v['width_points']:.1f} | " f"{v['lower_arm_points']:.1f} | {'none' if cz is None else cz} | {'none' if adv is None else f'{adv:.1f}'} |") out() NN = HW["near_the_word_counter"] out(f"Near the word counter's level ({NN['rule']}: {', '.join(NN['models'])}): widths " f"{NN['width_points'][0]:.1f} to {NN['width_points'][1]:.1f}, lower arms {NN['lower_arm_points'][0]:.1f} to " f"{NN['lower_arm_points'][1]:.1f}, and the kit advantage that would clear zero, the S0 count held, " f"{NN['kit_advantage_that_would_clear_zero_points'][0]:.1f} to {NN['kit_advantage_that_would_clear_zero_points'][1]:.1f} points.") out() RD = CK["resample_by_dinner"] out("**The resample by dinner.** " + RD["what"][0].upper() + RD["what"][1:] + f". {RD['dinners']} dinners, " f"{RD['lines_per_dinner'][0]} to {RD['lines_per_dinner'][1]} lines each; each line's dinner from `{RD['kit_lock']}`.") out() out("| model | variant | split | exact p | verdict | difference, points | SE, lines independent | SE, by dinner | " "design effect | by-dinner z p | by-dinner resample 95 %, points | resample p | verdict by dinner | the same |") out("|---|---|---|---:|---|---:|---:|---:|---:|---:|---|---:|---|---|") for k in PAGE_NB_TESTS: v = RD["paired_against_the_word_counter"][k] m, var = NAME[k] lo, hi = v["by_dinner_bootstrap_95_points"] out(f"| {m} | {var} | {v['only_this_right']} to {v['only_the_word_counter_right']} | {fmt_p(v['p'])} | {v['verdict']} | " f"{pts(v['diff_points'])} | {v['se_lines_independent_points']:.2f} | {v['se_by_dinner_points']:.2f} | " f"{v['design_effect']:.2f} | {fmt_p(v['by_dinner_z_p'])} | {pts(lo)} to {pts(hi)} | {fmt_p(v['by_dinner_bootstrap_p'])} | " f"{v['verdict_by_dinner']} | {yes(v['same_verdict'])} |") out() out(f"Verdicts changed by resampling dinners: {RD['verdicts_changed_by_dinner']} of {len(PAGE_NB_TESTS)}. Paired " f"standard errors wider by dinner (design effect over 1): {RD['paired_design_effects_over_1']} of {len(PAGE_NB_TESTS)}.") out() out("| model | variant | right | of | Wilson 95 %, % | by-dinner resample 95 %, % | SE, independent | SE, by dinner | " "design effect | wider by dinner |") out("|---|---|---:|---:|---|---|---:|---:|---:|---|") for k in ["openjev-bf16-readout-arm3", "kev-9b", "naive-bayes"]: v = RD["single_shares"][k] m, var = NAME[k] out(f"| {m} | {var} | {v['right_of_108']} | 108 | {w(v['wilson_95'][0])} to {w(v['wilson_95'][1])} | " f"{w(v['by_dinner_bootstrap_95'][0])} to {w(v['by_dinner_bootstrap_95'][1])} | {w(v['se_independent'])} | " f"{w(v['se_by_dinner'])} | {v['design_effect']:.2f} | {yes(v['wider_by_dinner'])} |") out() TO = CK["tests_in_each_option_order"] out("**The tests in each option order.** " + TO["what"][0].upper() + TO["what"][1:] + ". Each cell: right of 108 " "(right only here to right only in the word counter, exact p).") out() out("| model | variant | order 1, the kit's | order 2 | order 3 | order 4 | order 5 | order 6 | closest | row file @ commit |") out("|---|---|---|---|---|---|---|---|---|---|") for k in ["kev-9b", "apus-4b-high", "apus-9b-high", "lev-4b-gpu", "kev-4b"]: v = TO["models"][k] m, var = NAME[k] cells = " | ".join(f"{c['right']} ({c['only_this_right']} to {c['only_the_word_counter_right']}, p " f"{c['p']:.2f})" for c in v["orders"]) out(f"| {m} | {var} | {cells} | order {v['closest']['order']}, p {v['closest']['p']:.2f} | `{v['src']}` |") out() P4 = TO["the_page_four"] out(f"Among the four the page shuffled ({', '.join(NAME[x][0] + ' ' + NAME[x][1] for x in P4['models'])}), the " f"closest is {NAME[P4['closest']['model']][0]} {NAME[P4['closest']['model']][1]} at order " f"{P4['closest']['order']}, p {P4['closest']['p']:.3f}; told apart in any order: " f"{yes(P4['told_apart_in_any_order'])}.") out() PS = CK["paired_shuffle_counts"] out("**The paired shuffle counts.** " + PS["what"][0].upper() + PS["what"][1:] + ".") out() out("| first | second | moved, first | moved, second | moved only in the first | moved only in the second | " "exact p | told apart |") out("|---|---|---:|---:|---:|---:|---:|---|") for pair, v in PS["pairs"].items(): a, b = pair.split("|") out(f"| {NAME[a][0]} {NAME[a][1]} | {NAME[b][0]} {NAME[b][1]} | {PS['moved'][a]} | {PS['moved'][b]} | " f"{v['only_first_moved']} | {v['only_second_moved']} | {v['p']:.3f} | {yes(v['told_apart'])} |") out() DL = CK["door_9b_against_the_live_door"] out("**The 9B against the live door.** " + DL["what"][0].upper() + DL["what"][1:] + ".") out() out("| planted lines | the 9B refused | the live door refused | only the 9B refused (named, unnamed) | only the live " "door refused (named, unnamed) | exact p | the 9B's rows record the live verdict on every line | each row's " "kind equals the kit's |") out("|---:|---:|---:|---|---|---:|---|---|") out(f"| {DL['planted_lines']} | {DL['nine_b_refused']} | {DL['live_door_refused']} | {DL['only_the_9b_refused']} " f"({DL['only_the_9b_refused_named']}, {DL['only_the_9b_refused_unnamed']}) | {DL['only_the_live_door_refused']} " f"({DL['only_the_live_door_refused_named']}, {DL['only_the_live_door_refused_unnamed']}) | {DL['p']:.3f} | " f"{yes(DL['the_rows_gemma_choice_equals_the_0913_verdict_on_every_line'])} | {yes(DL['each_rows_kind_equals_the_kit'])} |") out() out("Row files: " + ", ".join(f"`{s}`" for s in DL["src"]) + ".") out() FL = CK["gemma4_floored_lines"] out("**Gemma 4's floored lines.** " + FL["what"][0].upper() + FL["what"][1:] + ".") out() out("| model | variant | rows | floored rows | right | row file @ commit |") out("|---|---|---:|---:|---:|---|") for k in ["gemma4-26b-q4-base-readout-arm1", "gemma4-26b-fp8-readout-arm10"]: v = FL[k] m, var = NAME[k] out(f"| {m} | {var} | {v['rows']} | {v['floored_rows']} | {v['right']} | `{v['src']}` |") out() MO = CK["mistral_other_run"] out("**Mistral Small 3.2's other run.** " + MO["what"][0].upper() + MO["what"][1:] + ".") out() out("| run | right, strict | of | unparsed | right, lenient | the first and last stamp, UTC | the same prompts in " "the same order as the card-A run | row file @ commit |") out("|---|---:|---:|---:|---:|---|---|---|") ca = R["s0"]["mistral-small3.2-24b-generate-coi"] out(f"| the workstation card | {MO['k']} | {MO['n']} | {MO['unparsed']} | {MO['lenient']} | {MO['window_utc'][0]} to " f"{MO['window_utc'][1]} | {yes(MO['same_prompts_in_the_same_order_as_the_card_a_run'])} | `{MO['src']}` |") out(f"| card A (table 1) | {MO['card_a_run_right']} | {ca['n']} | {ca.get('unparsed', 0)} | {ca.get('lenient', '—')} | " f"{MO['card_a_run_window_utc'][0]} to {MO['card_a_run_window_utc'][1]} | — | `{ca['src']}` |") out()