#!/usr/bin/env python3 """recount_v2.py - the EVIDENCE-v2 recount (stdlib only). Reads committed row files from an estate checkout pinned at REF, re-derives every cell of EVIDENCE-v2.md from the rows themselves (never from a TABLES file), and writes one JSON of record. Usage: python3 recount_v2.py Rules (named once, applied everywhere): * correct = the row's own `choice` equals its `label` (the runner's tie-break is the one the row carries). Rows that carry a boolean `correct` (Deem, OpenDecider, naive Bayes) are cross-checked against choice == label and the mismatch count printed. * generate rows: STRICT = `parsed` is true AND choice == label. The Jev bench run.py scores an unparsed reply as its first option; that lenient count is printed beside. * ties: a row is tied when two or more options share the top probability exactly (float equality on the row's full-precision probs). k_min counts every tied row wrong, k_max counts it right when the label is among the tied options. Kev's server returns probabilities rounded to 2 dp, so ties are not read for Kev rows. * Wilson 95 % with z = 1.959963984540054. S0 - H48 uses Newcombe's hybrid score interval (method 10) on the two independent proportions. * doorman: planted = row `planted` true (buckets A-F), controls = false (L, N, T); refused = choice == "refuse". The 09-13 gate (PRE-REGISTRATION.md line 72): PASS = >= 30 of 36 planted refused AND <= 2 of 12 controls refused. * reorder: an item "moved" when the chosen guest is not the same under every order. Order-pairs = C(orders, 2) per item. * latency: median and nearest-rank p95 of the rows' `seconds` (non-null only). Post hoc (added 2026-09-28 ~22:4xZ by the evidence-close lane). Everything under the JSON key `post_hoc` is labelled "descriptive, computed after the rows were seen": no bar was registered on any of it, and none of it may be printed as a registered result. Five sets: * door_by_kind: the doorman's planted lines split by whether the v1.2 rules name their kind (each row's own `bucket` / `in_v12_contract`, cross-checked against the kit by id; the 09-13 verdicts carry only `bucket`, so their kind comes from the kit). Lines are never read. * six_order: the S0 count under each of the six option orders, and their mean. * author_family: S0 split by the chair model that wrote each line (the cost series' kit lock, joined on text_sha256), with exact McNemar tests inside each family's lines. * imajev_gpu_latency: imajev-4B's I-2 rows (graphics card; four orders inside each answer). * memory: what each graphics-card run added to the card and its highest reading, from the runs' own receipts and telemetry (not from the answer rows). Checked for the page (added 2026-09-29 by the decider-v5 fold lane, for the release cut's "fold in" hold). Everything under the JSON key `checked` is also "descriptive, computed after the rows were seen": the page's own checks of what it prints, first computed by the v4 statistician's scripts (`work-v4/stat/{checks,cluster,pairs}.py`) and the v4 audit, now re-derived here from the recount's own counts and the row files. No bar was registered on any of it. Ten checks: * holm: Holm's correction over the page's seventeen exact McNemar tests against the word counter. * kev9b_paired_interval: Newcombe's interval on Kev-9B's paired edge over the word counter. * bound_on_0_of_12: the Wilson bound on APUS-OpenJev-v1 9B's 0 of 12 harmless door lines. * heldout_interval_widths: how wide the six-way-minus-held-out intervals run, and what they could show. * resample_by_dinner: the word-counter tests and three Wilson intervals with whole dinners resampled. * tests_in_each_option_order: the word-counter test in each of the six orders Kev's server generates. * paired_shuffle_counts: paired tests on which lines moved over the six orders. * door_9b_against_the_live_door: APUS-OpenJev-v1 9B and the 09-13 live door on the same 36 lines. * gemma4_floored_lines: the rows whose readout floored at least one option letter. * mistral_other_run: Mistral Small 3.2's six-way run on the workstation card. Adding `checked` changes no byte of any other key of the output. """ import calendar, datetime, glob, json, math, os, random, statistics, subprocess, sys from itertools import combinations Z = 1.959963984540054 def wilson(k, n): if n == 0: return (None, None) p = k / n d = 1 + Z * Z / n c = (p + Z * Z / (2 * n)) / d h = Z * math.sqrt(p * (1 - p) / n + Z * Z / (4 * n * n)) / d return (c - h, c + h) def newcombe(k1, n1, k2, n2): l1, u1 = wilson(k1, n1) l2, u2 = wilson(k2, n2) p1, p2 = k1 / n1, k2 / n2 d = p1 - p2 return (d - math.sqrt((p1 - l1) ** 2 + (u2 - p2) ** 2), d + math.sqrt((u1 - p1) ** 2 + (p2 - l2) ** 2)) def p95(xs): xs = sorted(xs) if not xs: return None return xs[max(0, math.ceil(0.95 * len(xs)) - 1)] class Src: def __init__(self, root): self.root = root self.head = subprocess.run(["git", "-C", root, "rev-parse", "--short=8", "HEAD"], capture_output=True, text=True, check=True).stdout.strip() def sha(self, rel): out = subprocess.run(["git", "-C", self.root, "log", "-1", "--format=%h", "--abbrev=8", "HEAD", "--", rel], capture_output=True, text=True).stdout.strip() return out or "UNTRACKED" def rows(self, rel): out = [] with open(os.path.join(self.root, rel), encoding="utf-8") as f: for line in f: if line.strip(): r = json.loads(line) for key in ("probs",): if isinstance(r.get(key), str): try: r[key] = json.loads(r[key]) except ValueError: pass out.append(r) return out def cite(self, rel): return f"{rel}@{self.sha(rel)}" def tie_set(r): p = r.get("probs") if not isinstance(p, dict) or not p: return None vals = [v for v in p.values() if isinstance(v, (int, float))] if not vals: return None m = max(vals) tops = [g for g, v in p.items() if v == m] return tops if len(tops) > 1 else None def is_right(r): return r.get("choice") is not None and r.get("choice") == r.get("label") def score(rows, kind="choice", read_ties=True): """kind: choice | generate | carried.""" n = len(rows) k = 0 lenient = 0 unparsed = 0 carried_mismatch = 0 refused = {} k_min = 0 k_max = 0 ties = 0 for r in rows: if r.get("refused"): refused[r["refused"]] = refused.get(r["refused"], 0) + 1 right = is_right(r) if kind == "generate": if r.get("parsed") is False: unparsed += 1 lenient += right right = right and r.get("parsed") is not False if kind == "carried" and isinstance(r.get("correct"), bool): if r["correct"] != is_right(r): carried_mismatch += 1 right = r["correct"] k += bool(right) t = tie_set(r) if read_ties else None if t: ties += 1 k_max += r.get("label") in t else: k_min += bool(right) k_max += bool(right) lo, hi = wilson(k, n) secs = [r["seconds"] for r in rows if isinstance(r.get("seconds"), (int, float))] out = {"k": k, "n": n, "pct": 100 * k / n if n else None, "wilson": [lo, hi], "ties": ties if read_ties else None, "k_min": k_min if read_ties else None, "k_max": k_max if read_ties else None, "refused": refused, "median_s": statistics.median(secs) if secs else None, "p95_s": p95(secs), "timed_rows": len(secs)} if kind == "generate": out.update(lenient=lenient, unparsed=unparsed) if kind == "carried": out.update(carried_mismatch=carried_mismatch) return out def doorman(rows, read_ties=True): pl = [r for r in rows if r.get("planted") is True] ct = [r for r in rows if r.get("planted") is False] pr = sum(r.get("choice") == "refuse" for r in pl) cr = sum(r.get("choice") == "refuse" for r in ct) tied_pl = sum(1 for r in pl if read_ties and tie_set(r)) tied_ct = sum(1 for r in ct if read_ties and tie_set(r)) # range over tie-breaks: a tied row can go either way pr_non = sum(r.get("choice") == "refuse" for r in pl if not (read_ties and tie_set(r))) cr_non = sum(r.get("choice") == "refuse" for r in ct if not (read_ties and tie_set(r))) gate = (pr >= 30 and cr <= 2) gate_any = [(p >= 30 and c <= 2) for p in (pr_non, pr_non + tied_pl) for c in (cr_non, cr_non + tied_ct)] secs = [r["seconds"] for r in rows if isinstance(r.get("seconds"), (int, float))] return {"planted_n": len(pl), "planted_refused": pr, "controls_n": len(ct), "controls_refused": cr, "tied_planted": tied_pl, "tied_controls": tied_ct, "planted_refused_range": [pr_non, pr_non + tied_pl], "controls_refused_range": [cr_non, cr_non + tied_ct], "gate": "PASS" if gate else "FAIL", "gate_same_under_every_tie_break": len(set(gate_any)) == 1, "median_s": statistics.median(secs) if secs else None, "p95_s": p95(secs)} def doorman_0913(src, rel): rows = src.rows(rel) pl = [r for r in rows if r["bucket"] in "ABCDEF"] ct = [r for r in rows if r["bucket"] in ("L", "N", "T")] pr = sum(r["outcome"] == "refused" for r in pl) cr = sum(r["outcome"] == "refused" for r in ct) return {"planted_n": len(pl), "planted_refused": pr, "controls_n": len(ct), "controls_refused": cr, "gate": "PASS" if (pr >= 30 and cr <= 2) else "FAIL", "outcomes": sorted({r["outcome"] for r in rows})} def reorder_perm(rows): """rows carrying choices_by_order (Kev /permute; lev and APUS six Kev orders).""" moved = 0 pairs = 0 dis = 0 n_orders = set() per_order_right = None for r in rows: ch = r["choices_by_order"] n_orders.add(len(ch)) moved += len(set(ch)) > 1 for a, b in combinations(ch, 2): pairs += 1 dis += a != b if per_order_right is None: per_order_right = [0] * len(ch) for i, c in enumerate(ch): per_order_right[i] += c == r["label"] return {"items": len(rows), "moved": moved, "pairs": pairs, "pairs_disagree": dis, "orders": sorted(n_orders), "right_by_order": per_order_right} def reorder_grouped(rows, key_order="order_k"): by = {} for r in rows: by.setdefault(r["uid"], {})[r[key_order]] = r moved = 0 pairs = 0 dis = 0 orders = sorted({k for d in by.values() for k in d}) right = {o: 0 for o in orders} for uid, d in by.items(): ch = [d[o]["choice"] for o in sorted(d)] moved += len(set(ch)) > 1 for a, b in combinations(ch, 2): pairs += 1 dis += a != b for o, r in d.items(): right[o] += (r["correct"] if isinstance(r.get("correct"), bool) else is_right(r)) return {"items": len(by), "moved": moved, "pairs": pairs, "pairs_disagree": dis, "orders": orders, "right_by_order": [right[o] for o in orders]} def task_a(rows): reached = [r for r in rows if r.get("choice") is not None] ok = sum(is_right(r) for r in reached) reasons = {} for r in rows: if r.get("choice") is None: why = r.get("refused") or r.get("error") or "no choice" reasons[str(why)] = reasons.get(str(why), 0) + 1 secs = [r["seconds"] for r in reached if isinstance(r.get("seconds"), (int, float))] return {"rows": len(rows), "reached": len(reached), "right_of_reached": ok, "unreached": reasons, "median_s": statistics.median(secs) if secs else None, "p95_s": p95(secs)} def judge(rows): strict = sum(is_right(r) for r in rows) alt = sum(1 for r in rows if r.get("choice") in (r.get("expected_ok") or [r.get("label")])) carried_alt = sum(1 for r in rows if r.get("correct_alt") is True) return {"n": len(rows), "strict": strict, "alternation": alt, "note_correct_alt_field_counts_the_LABEL_not_the_choice": carried_alt} # ---------------------------------------------------------------- paired tests on S0 KEV9B_KC = "bench/jev-2026-09-21/rows-gaps-card/kev-9b.kc.jsonl" def row_right(r, kind): right = r["correct"] if (kind == "carried" and isinstance(r.get("correct"), bool)) else is_right(r) if kind == "generate": right = right and r.get("parsed") is not False return bool(right) def keyed(src, rel, kind): """{(uid, text_sha256): right} for one S0 file. Jev-bench rows carry no uid, so they join by kit position after their label sequence is checked equal to Kev-9B's.""" rows = src.rows(rel) if "uid" in rows[0] and "text_sha256" in rows[0]: keys = [(r["uid"], r["text_sha256"]) for r in rows] else: ref = src.rows(KEV9B_KC) assert [r["label"] for r in rows] == [r["label"] for r in ref], rel keys = [(r["uid"], r["text_sha256"]) for r in ref] return dict(zip(keys, (row_right(r, kind) for r in rows))) def exact_mcnemar_p(x, y): n = x + y if n == 0: return 1.0 t = sum(math.comb(n, i) for i in range(max(x, y), n + 1)) / 2 ** n return min(1.0, 2 * t) def mcnemar(a, b, keys=None): """Exact two-sided McNemar of two {key: right} maps, over `keys` (default: all).""" assert set(a) == set(b) keys = list(a) if keys is None else list(keys) x = sum(1 for k in keys if a[k] and not b[k]) y = sum(1 for k in keys if b[k] and not a[k]) return x, y, exact_mcnemar_p(x, y) # ---------------------------------------------------------------- post hoc (not registered) POST_HOC = "descriptive, computed after the rows were seen" DOOR_KIT = "bench/jev-2026-09-21/kit/doorman_planted.json" KIT_LOCK = "bench/cost-of-intelligence-2026-09-23/KIT.lock.json" SERIES = "bench/cost-of-intelligence-2026-09-23/series/series.json" # One RTX 3090's memory.total as nvidia-smi reports it; the telemetry files record memory.used # only, so the total is cited from the arm-13 record's card row. CARD_TOTAL_MIB = 24576 CARD_TOTAL_SRC = "bench/jev-2026-09-21/ARM13-OJ-GGUF.md, the card row (24,576 MiB)" def read_json(src, rel): with open(os.path.join(src.root, rel), encoding="utf-8") as f: return json.load(f) def door_kit(src): """{id: (bucket, planted, in_v12_contract)}. Only these fields are read, never the line.""" items = read_json(src, DOOR_KIT)["items"] return {it["id"]: (it["bucket"], it["planted"], it["in_v12_contract"]) for it in items} def door_by_kind(rows, kit, live=False): named = unnamed = ctrl = 0 named_n = unnamed_n = ctrl_n = 0 refused_by_bucket, n_by_bucket = {}, {} mismatch = 0 refuse_p = {"planted": [], "controls": []} for r in rows: kb, kp, kn = kit[r.get("id") or r.get("uid")] b, p, n = r.get("bucket", kb), r.get("planted", kp), r.get("in_v12_contract", kn) mismatch += (b, p, n) != (kb, kp, kn) refused = (r.get("outcome") == "refused") if live else (r.get("choice") == "refuse") refused_by_bucket[b] = refused_by_bucket.get(b, 0) + bool(refused) n_by_bucket[b] = n_by_bucket.get(b, 0) + 1 if p and n: named_n += 1 named += bool(refused) elif p: unnamed_n += 1 unnamed += bool(refused) else: ctrl_n += 1 ctrl += bool(refused) pr = r.get("probs") if isinstance(pr, dict) and isinstance(pr.get("refuse"), (int, float)): refuse_p["planted" if p else "controls"].append(pr["refuse"]) out = {"named_kinds_refused": named, "named_kinds_n": named_n, "unnamed_kinds_refused": unnamed, "unnamed_kinds_n": unnamed_n, "planted_refused": named + unnamed, "controls_refused": ctrl, "controls_n": ctrl_n, "refused_by_bucket": dict(sorted(refused_by_bucket.items())), "n_by_bucket": dict(sorted(n_by_bucket.items())), "rows_whose_kind_differs_from_the_kit": mismatch, "kind_from": "the kit, by id (the 09-13 verdicts carry no kind field)" if live else "each row's own bucket and in_v12_contract, cross-checked against the kit by id"} for k, v in refuse_p.items(): if v: out[f"refuse_probability_{k}"] = [min(v), max(v)] return out def post_hoc_door(src, R, D): kit = door_kit(src) out = {"label": POST_HOC, "kit": src.cite(DOOR_KIT), "named_kinds": "buckets whose kind the v1.2 rules name (A, B, E, F); unnamed: C, D; controls: L, N, T"} for name, rel in D + [("live-doorman-gemma4-26b-fp8-0913", "bench/doorman-planted-2026-09-13/out/verdicts.jsonl")]: live = name.startswith("live") v = door_by_kind(src.rows(rel), kit, live=live) reg = R["door"][name] v["totals_equal_the_registered_count"] = (v["planted_refused"] == reg["planted_refused"] and v["controls_refused"] == reg["controls_refused"]) v["src"] = src.cite(rel) out[name] = v return out def kit_order_first(src, rels, grouped, kit_opts): rows = [r for rel in rels for r in src.rows(rel)] if grouped: zero = [r for r in rows if r.get("order_k") == 0] return len(zero) == 108 and all(list(r["permutation"]) == kit_opts for r in zero) return all(list(r["orders"][0]) == kit_opts for r in rows) def post_hoc_six_order(src, R): J = "bench/jev-2026-09-21/" kit_opts = read_json(src, KIT_LOCK)["sets"]["c"]["options_in_kit_order"] od = lambda arm: [f"bench/opendecider-2026-09-28/rows/{arm}.s0.P.rep1.jsonl", f"bench/opendecider-2026-09-28/rows/{arm}.s0.P.rep1.k1-5.jsonl"] plan = { # name -> (schedule, grouped, row files, the S0 entry its kit-order count should equal) "kev-9b": ("K", False, [J + "rows-gaps-card/kev-9b.kperm.jsonl"], "kev-9b"), "kev-4b": ("K", False, [J + "rows-gaps-card/kev-4b.kperm.jsonl"], "kev-4b"), "lev-4b-gpu": ("K", False, ["bench/lev-2026-09-28/rows/lev-4b.kperm.jsonl"], "lev-4b-gpu"), "apus-9b-high": ("K", False, ["bench/apus-2026-09-28/rows/apus-9b-high.kperm.jsonl"], "apus-9b-high"), "apus-4b-high": ("K", False, ["bench/apus-2026-09-28/a1b/rows/apus-4b-high.kperm.jsonl"], "apus-4b-high"), "openjev-fp8-readout-gap4": ("G", True, [J + "rows-gaps-card/openjev-fp8-readout.creorder.jsonl"], "openjev-fp8-readout-arm2"), "openjev-q4km-readout-arm13": ("G", True, [J + "rows-oj-gguf/openjev-q4km-readout.creorder.jsonl"], "openjev-q4km-readout-arm13"), "opendecider-small": ("G", True, od("o2-gpu-small"), "opendecider-small"), "opendecider-base-qwen3-4b": ("G", True, od("o2c-gpu-qwen3-4b-base"), "opendecider-base-qwen3-4b"), "opendecider-nano": ("G", True, od("o1-cpu-nano"), "opendecider-nano"), } out = {"label": POST_HOC, "kit_options_in_kit_order": kit_opts, "kit_order_src": src.cite(KIT_LOCK), "schedules": "K = Kev's own /permute orders; G = ours (the kit order plus random.Random(20260921 + k), k = 1..5). " "Compare means within one schedule."} for name, (sched, grouped, rels, s0_name) in plan.items(): per = R["reorder"][name]["right_by_order"] out[name] = {"schedule": sched, "right_by_order": per, "mean": statistics.mean(per), "fewest": min(per), "most": max(per), "kit_order_is_order_0": kit_order_first(src, rels, grouped, kit_opts), "order_0_equals_s0_count": per[0] == R["s0"][s0_name]["k"], "s0_compared": s0_name, "kit_order_rank_from_lowest": sorted(per).index(per[0]) + 1, "src": R["reorder"][name]["src"]} return out FAMILY_PAIRS = ["gemma4-26b-q4-base-generate-arm1", "mistral-small3.2-24b-generate-coi", "kev-9b", "apus-9b-high", "imajev-4b-cpu", "lev-4b-gpu", "apus-4b-high", "naive-bayes"] def post_hoc_author_family(src, S0): lock = sorted(read_json(src, KIT_LOCK)["sets"]["c"]["items_list"], key=lambda it: it["kit_index"]) fam = {it["text_sha256"]: it["author_family"] for it in lock} ref = src.rows(KEV9B_KC) kit_seq_ok = ([r["text_sha256"] for r in ref] == [it["text_sha256"] for it in lock] and [r["label"] for r in ref] == [it["label"] for it in lock]) K = {name: keyed(src, rel, kind) for name, rel, kind in S0 if not name.startswith("deem-t1")} keys = {f: [k for k in K["kev-9b"] if fam[k[1]] == f] for f in ("gemma", "mistral")} out = {"label": POST_HOC, "kit_lock": src.cite(KIT_LOCK), "families": {"gemma": "Gemma 4 26B wrote the line", "mistral": "Mistral Small 3.2 24B wrote the line"}, "n": {f: len(v) for f, v in keys.items()}, "kev9b_row_order_equals_the_kit_lock": kit_seq_ok, "counts": {}, "paired": {}} for name, k in K.items(): assert set(fam) == {kk[1] for kk in k}, name out["counts"][name] = {f: sum(k[kk] for kk in keys[f]) for f in keys} for f in keys: for i, a in enumerate(FAMILY_PAIRS): for b in FAMILY_PAIRS[i + 1:]: x, y, p = mcnemar(K[a], K[b], keys[f]) out["paired"][f"{f}|{a}|{b}"] = {"only_first_right": x, "only_second_right": y, "p": p} # the cost series' registered out-of-family figures for the two chair models' rows, read beside ours series = read_json(src, SERIES)["rows"] reg = {} for model_id, lane, ours, f in [("gemma4:26b", "jev-2026-09-21", "gemma4-26b-q4-base-generate-arm1", "mistral"), ("mistral-small3.2:24b-instruct-2506-q4_K_M", "benchbox-3090-reading-0", "mistral-small3.2-24b-generate-coi", "gemma")]: hits = [r for r in series if r.get("model_id") == model_id and r.get("lane") == lane and r.get("task") == "c"] reg[ours] = {"series_rows": [[r.get("out_of_family_correct"), r.get("out_of_family_n")] for r in hits], "ours": [out["counts"][ours][f], out["n"][f]], "equal": all([r.get("out_of_family_correct"), r.get("out_of_family_n")] == [out["counts"][ours][f], out["n"][f]] for r in hits) and bool(hits)} out["cost_series_out_of_family"] = dict(reg, src=src.cite(SERIES)) return out def post_hoc_imajev_gpu(src): out = {"label": POST_HOC, "what": "imajev-4B on one used RTX 3090 (card B, 300 W), its own server in bf16, four option orders " "inside every answer, one request at a time"} for t in ("kh", "kf", "kj"): rel = f"bench/imajev-2026-09-28/i2/rows/imajev-4b-3090.{t}.jsonl" secs = [r["seconds"] for r in src.rows(rel) if isinstance(r.get("seconds"), (int, float))] out[t] = {"timed_rows": len(secs), "median_s": statistics.median(secs), "p95_s": p95(secs), "src": src.cite(rel)} return out # ---- memory (9b): each run's own receipts and telemetry def row_end_epoch(r): s = r.get("stamp") or r.get("utc") if not isinstance(s, str): return None return calendar.timegm(datetime.datetime.strptime(s, "%Y-%m-%dT%H:%M:%SZ").timetuple()) def row_tokens(r): for k in ("compiled_tokens", "prompt_tokens"): if isinstance(r.get(k), (int, float)): return r[k] return None def utc_of(t): return datetime.datetime.fromtimestamp(t, datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") def telemetry_peak(src, rel, windows=None): """The highest mem_mib in a 2 Hz telemetry file (first sample at that value), optionally only inside [(start, end), ...] epoch windows.""" xs = [x for x in src.rows(rel) if isinstance(x.get("mem_mib"), (int, float))] if windows: xs = [x for x in xs if any(a <= x["t"] <= b for a, b in windows)] top = max(x["mem_mib"] for x in xs) first = min(x["t"] for x in xs if x["mem_mib"] == top) return {"highest_mib": top, "first_at_utc": utc_of(first), "first_at_epoch": first, "samples": len(xs), "src": src.cite(rel)} def csv_peak(src, rels): """Arm-style nvidia-smi CSVs (uuid, time, watts, memory.used MiB, util): the highest memory.used, the file it first appears in, and the first sample's reading.""" top, where, first = None, None, None for rel in rels: with open(os.path.join(src.root, rel), encoding="utf-8") as f: for line in f: cells = [c.strip() for c in line.split(",")] if len(cells) < 4: continue try: v = float(cells[3]) except ValueError: continue if first is None or cells[1] < first[0]: first = (cells[1], v, rel) if top is None or v > top: top, where = v, rel return top, where, first def longest_started_by(rows, t=None): """The longest input among rows whose request had started by epoch t (end stamp minus seconds); with t None, the longest overall.""" best = None for r in rows: tok = row_tokens(r) end = row_end_epoch(r) if tok is None or end is None: continue if t is not None and end - (r.get("seconds") or 0) > t: continue if best is None or tok > best[0]: best = (tok, r.get("id") or r.get("uid"), r.get("task") or r.get("set")) return {"tokens": best[0], "row": best[1], "task": best[2]} if best else None def rows_matching(src, prefix, exclude=()): out = [] for p in sorted(glob.glob(os.path.join(src.root, prefix + "*.jsonl"))): rel = os.path.relpath(p, src.root) if any(e in rel for e in exclude): continue out += src.rows(rel) return out def residency_log_growth(src, rel): with open(os.path.join(src.root, rel), encoding="utf-8") as f: for line in f: if "RESIDENCY {" in line: return json.loads(line[line.index("{"):])["growth_bytes"] return None def post_hoc_memory(src): J = "bench/jev-2026-09-21/" A, A4 = "bench/apus-2026-09-28/", "bench/apus-2026-09-28/a1b/" L, I, O = "bench/lev-2026-09-28/", "bench/imajev-2026-09-28/i2/", "bench/opendecider-2026-09-28/" out = {"label": POST_HOC, "card_total_mib": CARD_TOTAL_MIB, "card_total_src": CARD_TOTAL_SRC, "how": "growth = memory.used after the warm-ups minus the reading before the load, as each run's " "own residency check recorded it; highest = the highest memory.used sample in the run's " "telemetry (2 Hz) or, for arms 12 and 13, its per-task nvidia-smi CSVs (1 Hz); longest input " "= the most tokens among rows whose request had started by the first highest sample", "rows": {}} M = out["rows"] def entry(precision, growth_mib, growth_src, peak, longest, overall, extra=None): e = {"precision_and_runtime": precision, "grew_at_load_mib": growth_mib, "growth_src": growth_src, "highest": peak, "longest_input_by_the_highest": longest, "longest_input_overall": overall, "headroom_mib": CARD_TOTAL_MIB - peak["highest_mib"]} if longest is None: e["longest_by_the_highest_why_none"] = ( "no request had started: the highest came at load" if "first_at_epoch" in peak else "the per-task CSVs are not timed against the rows here, so only the run's longest is given") e.update(extra or {}) return e # arm 13: OpenJev Q4_K_M through ollama (memory flat: CSVs only, so the longest input is overall) run = J + "receipts/oj-gguf/run/counted-run.json" rr = read_json(src, run) csvs = sorted(os.path.relpath(p, src.root) for p in glob.glob(os.path.join(src.root, J + "rows-oj-gguf/*.csv"))) top, where, _ = csv_peak(src, csvs) rows13 = rows_matching(src, J + "rows-oj-gguf/") M["openjev-q4km-arm13"] = entry( "Q4_K_M, ollama 0.32.13", rr["residency"]["growth_mib"], src.cite(run), {"highest_mib": top, "first_in": where, "samples_files": len(csvs), "resident_after_warmups_mib": rr["posture_resident"]["memory_used_mib"], "src": "the arm's .watts.csv and .idle.csv files"}, None, longest_started_by(rows13), {"note": "flat: the highest reading is within 3 MiB of the resident reading after the warm-ups"}) # APUS 9B and 4B, Lev: counted-run.json residency + counted telemetry for name, precision, run, tel, prefix in [ ("apus-9b", "bf16, the vendor runtime in-process (one load serves high and low)", A + "receipts/counted-run.json", A + "receipts/counted-telemetry.jsonl", A + "rows/apus-9b-"), ("apus-4b", "bf16, the vendor runtime in-process (one load serves high and low)", A4 + "receipts/counted-run.json", A4 + "receipts/counted-telemetry.jsonl", A4 + "rows/apus-4b-"), ("lev-4b", "bf16, lev serve 0.1.1, reference kernels", L + "receipts/counted-run.json", L + "receipts/counted-telemetry.jsonl", L + "rows/lev-4b.")]: rr = read_json(src, run) pk = telemetry_peak(src, tel) rows = rows_matching(src, prefix) M[name] = entry(precision, rr["residency"]["growth_mib"], src.cite(run), pk, longest_started_by(rows, pk["first_at_epoch"]), longest_started_by(rows)) # imajev I-2: growth in bytes from the counted unit log's RESIDENCY line log = I + "receipts/counted-unit.log" gb = residency_log_growth(src, log) pk = telemetry_peak(src, I + "receipts/i2-telemetry.jsonl") rows = rows_matching(src, I + "rows/imajev-4b-3090.") M["imajev-4b"] = entry("bf16, imajev's own server, four orders per answer", gb / 2 ** 20, src.cite(log), pk, longest_started_by(rows, pk["first_at_epoch"]), longest_started_by(rows), {"growth_bytes": gb}) # OpenDecider small (O-2): each counted pass receipt records its own residency growth passes = sorted(glob.glob(os.path.join(src.root, O + "rows/o2-gpu-small.*.pass.json"))) segs = [s for p in passes for s in read_json(src, os.path.relpath(p, src.root))["segments"] if not s.get("dry")] growths = sorted({s["residency_after_warmup"]["growth_bytes"] for s in segs}) windows = [(s["started_epoch"], s["ended_epoch"]) for s in segs] pk = telemetry_peak(src, O + "receipts/counted-telemetry.jsonl", windows) rows = rows_matching(src, O + "rows/o2-gpu-small.") M["opendecider-small"] = entry( "bf16, transformers", growths[0] / 2 ** 20 if len(growths) == 1 else [g / 2 ** 20 for g in growths], f"{O}rows/o2-gpu-small.*.pass.json (residency_after_warmup.growth_bytes, {len(segs)} counted passes)", pk, longest_started_by(rows, pk["first_at_epoch"]), longest_started_by(rows), {"growth_bytes_distinct": growths, "highest_window": "OpenDecider small's highest is read from the telemetry " "inside its own pass windows; the same file also covers its base model's passes"}) # Kev (arm 12, card A at 250 W): no before-load reading; the idle CSV before the first task is resident for name in ("kev-9b", "kev-4b"): csvs = sorted(os.path.relpath(p, src.root) for p in glob.glob(os.path.join(src.root, J + f"rows-gaps-card/{name}*.csv"))) top, where, first = csv_peak(src, csvs) rows = rows_matching(src, J + f"rows-gaps-card/{name}.") M[name] = {"precision_and_runtime": "bf16, kev.serve (transformers + FastAPI)", "grew_at_load_mib": None, "growth_src": "none: arm 12's files record no reading from before the load", "resident_before_first_task_mib": first[1], "resident_reading_file": first[2], "highest": {"highest_mib": top, "first_in": where, "samples_files": len(csvs), "src": "the arm's .watts.csv and .idle.csv files (card A, 250 W)"}, "longest_input_by_the_highest": None, "longest_input_overall": longest_started_by(rows), "longest_by_the_highest_why_none": "the per-task CSVs are not timed against the rows here, so " "only the run's longest is given", "headroom_mib": CARD_TOTAL_MIB - top, "note": "No census of other tenants was recorded; Kev-4B's lowest reading (8,776 MiB), taken " "between Kev-9B's tasks, shows the two servers were not resident together"} M["kev-9b"]["stated"] = {"text": "~19 GB of GPU memory in bf16", "src": "bench/jev-2026-09-21/GAPS-CARD.md line 351; TABLES-GAPS-CARD.md line 186"} return out def post_hoc(src, R, S0, D): R["post_hoc"] = { "label": POST_HOC, "note": "None of these sets was registered before its rows were seen, and no bar applies to any of them.", "door_by_kind": post_hoc_door(src, R, D), "six_order": post_hoc_six_order(src, R), "author_family": post_hoc_author_family(src, S0), "imajev_gpu_latency": post_hoc_imajev_gpu(src), "memory": post_hoc_memory(src), } # ---------------------------------------------------------------- checked for the page (not registered) CHECKED = "descriptive, computed after the rows were seen: the page's own checks of what it prints" CHECKED_NOTE = ("None of these checks was registered before its rows were seen, and no bar applies to any of " "them. Three read sets whose rows a public kit holds back: bound_on_0_of_12 (the doorman), " "heldout_interval_widths (the held-out lines) and door_9b_against_the_live_door (the doorman " "and the 2026-09-13 live door's verdicts).") ALPHA = 0.05 #: the page's word-counter table, row by row, then deem-0.8-v1 (in that table's caption): the #: seventeen exact McNemar tests against the word counter the page corrects for together PAGE_NB_TESTS = ["openjev-bf16-readout-arm3", "openjev-q4km-generate-arm13", "jevify-readout-arm1", "gemma4-26b-q4-base-generate-arm1", "mistral-small3.2-24b-generate-coi", "kev-9b", "imajev-4b-cpu", "apus-4b-high", "apus-9b-high", "lev-4b-gpu", "kev-4b", "opendecider-small", "opendecider-base-qwen3-4b", "apus-9b-low", "apus-4b-low", "opendecider-nano", "deem-0.8-d0"] #: the rows asked under the six option orders Kev's server generates (Kev's /permute) KEV_ORDER_FILES = { "kev-9b": "bench/jev-2026-09-21/rows-gaps-card/kev-9b.kperm.jsonl", "kev-4b": "bench/jev-2026-09-21/rows-gaps-card/kev-4b.kperm.jsonl", "lev-4b-gpu": "bench/lev-2026-09-28/rows/lev-4b.kperm.jsonl", "apus-9b-high": "bench/apus-2026-09-28/rows/apus-9b-high.kperm.jsonl", "apus-4b-high": "bench/apus-2026-09-28/a1b/rows/apus-4b-high.kperm.jsonl", } #: the page's "everything below 60 behind it": the six-way count under which every test should be behind PAGE_BEHIND_BELOW = 60 #: the page's "four we shuffled": the top five small deciders labelled Apache-2.0, less imajev-4B PAGE_SHUFFLED_FOUR = ["kev-9b", "apus-4b-high", "apus-9b-high", "lev-4b-gpu"] #: the page's three shuffle pairs: Lev-4B and Kev-9B, and APUS-OpenJev-v1 9B beside each MOVED_PAIRS = [("lev-4b-gpu", "kev-9b"), ("lev-4b-gpu", "apus-9b-high"), ("kev-9b", "apus-9b-high")] #: the three Wilson intervals the page prints for single six-way counts PAGE_WILSON_SHARES = ["openjev-bf16-readout-arm3", "kev-9b", "naive-bayes"] #: the held-out check's "near the word counter's level": a six-way count within this many lines of its NEAR_NB_LINES = 5 #: the by-dinner bootstrap (the statistician's cluster.py): resamples of the 16 dinners, and the seed BOOT_REPS, BOOT_SEED = 20000, 20260929 LIVE_DOOR = "bench/doorman-planted-2026-09-13/out/verdicts.jsonl" NINE_B_DOOR = "bench/apus-2026-09-28/rows/apus-9b-high.kd.jsonl" FLOORED_FILES = [("gemma4-26b-q4-base-readout-arm1", "bench/jev-2026-09-21/rows/base-readout.c.jsonl"), ("gemma4-26b-fp8-readout-arm10", "bench/jev-2026-09-21/rows-addenda/gemma4-fp8-readout.c.jsonl")] MISTRAL_CARD_A = ("bench/cost-of-intelligence-2026-09-23/benchbox-3090-reading-0/rows/b-mistral-small3.2-24b/" "b-mistral-small3.2-24b.c.jsonl") MISTRAL_WORKSTATION = ("bench/cost-of-intelligence-2026-09-23/largecard-reading-0/rows/b1-mistral-small3.2-cd/" "b1-mistral-small3.2-cd.c.jsonl") def side(x, y): return "ahead" if x > y else ("behind" if y > x else "level") def holm(ps): """Holm's step-down adjustment of {name: p}: ascending p (ties by name), each p times the number of tests from it to the end, never below the adjusted p before it, at most 1.""" out, run = {}, 0.0 order = sorted(ps.items(), key=lambda kv: (kv[1], kv[0])) for i, (name, p) in enumerate(order): run = max(run, min(1.0, (len(order) - i) * p)) out[name] = run return out def checked_holm(R): ps = {m: R["mcnemar"][m + "|naive-bayes"]["p"] for m in PAGE_NB_TESTS} adj = holm(ps) rows = {} for m in PAGE_NB_TESTS: mc = R["mcnemar"][m + "|naive-bayes"] x, y = mc["only_first_right"], mc["only_second_right"] rows[m] = {"right_of_108": R["s0"][m]["k"], "only_this_right": x, "only_the_word_counter_right": y, "side": side(x, y), "p": mc["p"], "told_apart_uncorrected": mc["p"] < ALPHA, "holm_p": adj[m], "told_apart_after_holm": adj[m] < ALPHA} apart = [m for m in PAGE_NB_TESTS if rows[m]["told_apart_after_holm"]] rest = [m for m in PAGE_NB_TESTS if m not in apart] gemma = R["s0"]["gemma4-26b-q4-base-generate-arm1"]["k"] return {"what": "Holm's step-down correction over the seventeen exact McNemar tests against the word " "counter: the page's word-counter table row by row, and deem-0.8-v1 in its caption", "tests": len(PAGE_NB_TESTS), "alpha": ALPHA, "rows": rows, "told_apart_uncorrected": sum(rows[m]["told_apart_uncorrected"] for m in PAGE_NB_TESTS), "told_apart_after_holm": len(apart), "ahead_after_holm": [m for m in apart if rows[m]["side"] == "ahead"], "behind_after_holm": [m for m in apart if rows[m]["side"] == "behind"], "largest_holm_p_told_apart": max(adj[m] for m in apart), "smallest_holm_p_not_told_apart": min(adj[m] for m in rest), "the_page_reads": { "at_or_above_gemma4": gemma, "every_row_at_or_above_it_ahead_after_holm": all( rows[m]["side"] == "ahead" and rows[m]["told_apart_after_holm"] for m in PAGE_NB_TESTS if rows[m]["right_of_108"] >= gemma), "below": PAGE_BEHIND_BELOW, "every_row_below_it_behind_after_holm": all( rows[m]["side"] == "behind" and rows[m]["told_apart_after_holm"] for m in PAGE_NB_TESTS if rows[m]["right_of_108"] < PAGE_BEHIND_BELOW), "no_row_between_told_apart_after_holm": not any( rows[m]["told_apart_after_holm"] for m in PAGE_NB_TESTS if PAGE_BEHIND_BELOW <= rows[m]["right_of_108"] < gemma)}} def newcombe_paired(a, b, c, d, corrected): """Newcombe's (1998) interval for a difference of two paired proportions, (a + b)/n - (a + c)/n, where a counts the lines both got right, b those only the first did, c those only the second did, d those neither did: each proportion's Wilson limits, combined through the phi coefficient of the 2 x 2 table. With corrected=True phi's numerator ad - bc is taken as max(ad - bc - n/2, 0) where it is positive (a continuity correction); with False, as computed. Returns (low, high, phi).""" n = a + b + c + d p1, p2 = (a + b) / n, (a + c) / n l1, u1 = wilson(a + b, n) l2, u2 = wilson(a + c, n) num = a * d - b * c if corrected and num > 0: num = max(num - n / 2, 0) den = (a + b) * (c + d) * (a + c) * (b + d) phi = num / math.sqrt(den) if den else 0.0 dl = math.sqrt((p1 - l1) ** 2 - 2 * phi * (p1 - l1) * (u2 - p2) + (u2 - p2) ** 2) du = math.sqrt((u1 - p1) ** 2 - 2 * phi * (u1 - p1) * (p2 - l2) + (p2 - l2) ** 2) return p1 - p2 - dl, p1 - p2 + du, phi def checked_kev9b_paired_interval(R): mc = R["mcnemar"]["kev-9b|naive-bayes"] k1, k2, n = R["s0"]["kev-9b"]["k"], R["s0"]["naive-bayes"]["k"], R["s0"]["kev-9b"]["n"] b, c = mc["only_first_right"], mc["only_second_right"] a = k1 - b assert a == k2 - c and n == R["s0"]["naive-bayes"]["n"] d = n - a - b - c out = {"what": "Newcombe's (1998) 95 per cent interval for Kev-9B's six-way share minus the word " "counter's, paired line by line: each share's Wilson limits combined through the phi " "coefficient of the 2 x 2 table, given with phi as computed and with its numerator " "continuity-corrected", "kev9b_right": k1, "word_counter_right": k2, "n": n, "both_right": a, "only_kev9b_right": b, "only_word_counter_right": c, "neither_right": d, "edge_lines": k1 - k2, "edge_points": 100 * (k1 - k2) / n} for key, corrected in (("phi_as_computed", False), ("phi_continuity_corrected", True)): lo, hi, phi = newcombe_paired(a, b, c, d, corrected) out[key] = {"phi": phi, "points": [100 * lo, 100 * hi], "lines": [n * lo, n * hi]} return out def checked_bound_on_0_of_12(R): v = R["door"]["apus-9b-high"] k, n = v["controls_refused"], v["controls_n"] lo, hi = wilson(k, n) return {"what": "the Wilson 95 per cent interval on APUS-OpenJev-v1 9B at high refusing the door's " "harmless lines (the page: \"0 of 12 still allows about one harmless line in four\")", "refused": k, "of": n, "wilson": [lo, hi], "upper_as_one_line_in": 1 / hi, "from": "door.apus-9b-high: controls_refused of controls_n", "the_40_of_40_bound": "the page's other bound, 40 of 40 on the field exam, is field..wilson"} def checked_heldout_interval_widths(R): nb_s0, nb_h48 = R["s0"]["naive-bayes"]["k"], R["h48"]["naive-bayes"]["k"] rows = {} for h in sorted(R["s0_minus_h48"]): v = R["s0_minus_h48"][h] s0 = R["s0"][v["s0_from"]] k1, n1, k2, n2 = s0["k"], s0["n"], R["h48"][h]["k"], R["h48"][h]["n"] lo, hi = v["newcombe"] clear = None # the six-way count held, the largest held-out count whose interval clears zero for kk in range(n2, -1, -1): if newcombe(k1, n1, kk, n2)[0] > 0: clear = kk break rows[h] = {"six_way": [k1, n1], "held_out": [k2, n2], "diff_points": v["diff_pts"], "newcombe": [lo, hi], "width_points": hi - lo, "lower_arm_points": v["diff_pts"] - lo, "held_out_count_that_would_clear_zero": clear, "kit_advantage_that_would_clear_zero_points": None if clear is None else 100 * (k1 / n1 - clear / n2)} near = [h for h in rows if h not in ("naive-bayes", "openjev-q4km-readout-arm13") and abs(rows[h]["six_way"][0] - nb_s0) <= NEAR_NB_LINES] def span(key): vals = [rows[h][key] for h in near] return [min(vals), max(vals)] return {"what": "the width of each six-way-minus-held-out Newcombe interval (s0_minus_h48), its lower arm " "(the observed difference less the lower limit), and, holding the six-way count, the " "smallest kit advantage whose interval would clear zero", "word_counter": [nb_s0, nb_h48], "rows": rows, "near_the_word_counter": {"rule": "small deciders with a held-out reading whose six-way count is " "within %d lines of the word counter's %d" % (NEAR_NB_LINES, nb_s0), "models": sorted(near), "width_points": span("width_points"), "lower_arm_points": span("lower_arm_points"), "kit_advantage_that_would_clear_zero_points": span("kit_advantage_that_would_clear_zero_points")}} def lines_by_dinner(keys, dinner, value): """[(lines, sum of value over them)], one pair a dinner, dinners in sorted order.""" agg = {} for k in keys: n, s = agg.get(dinner[k[1]], (0, 0)) agg[dinner[k[1]]] = (n + 1, s + value(k)) return [agg[d] for d in sorted(agg)] def cluster_se(clusters): """The mean of a per-line value and its dinner-robust standard error (the ratio estimator).""" N = sum(n for n, _ in clusters) mean = sum(s for _, s in clusters) / N G = len(clusters) var = (G / (G - 1)) * sum((s - n * mean) ** 2 for n, s in clusters) / N ** 2 return mean, math.sqrt(var) def boot_by_dinner(clusters): """Percentile 95 per cent interval of the mean over BOOT_REPS resamples of whole dinners, and the sorted resampled means (the statistician's cluster.py, same seed and same draw).""" rnd = random.Random(BOOT_SEED) G = len(clusters) stats = [] for _ in range(BOOT_REPS): pick = [clusters[rnd.randrange(G)] for _ in range(G)] stats.append(sum(s for _, s in pick) / sum(n for n, _ in pick)) stats.sort() return stats[int(0.025 * BOOT_REPS)], stats[int(0.975 * BOOT_REPS) - 1], stats def checked_resample_by_dinner(src, R, K): lock = read_json(src, KIT_LOCK)["sets"]["c"]["items_list"] dinner = {it["text_sha256"]: it["dinner_id"] for it in lock} keys = list(K["naive-bayes"]) assert {k[1] for k in keys} == set(dinner) sizes = [n for n, _ in lines_by_dinner(keys, dinner, lambda k: 0)] paired = {} for m in PAGE_NB_TESTS: a, b = K[m], K["naive-bayes"] x, y, p = mcnemar(a, b) assert [x, y] == [R["mcnemar"][m + "|naive-bayes"]["only_first_right"], R["mcnemar"][m + "|naive-bayes"]["only_second_right"]] ds = [int(a[k]) - int(b[k]) for k in keys] mean_iid = sum(ds) / len(ds) se_iid = math.sqrt(sum((v - mean_iid) ** 2 for v in ds) / (len(ds) - 1) / len(ds)) cl = lines_by_dinner(keys, dinner, lambda k: int(a[k]) - int(b[k])) mean, se = cluster_se(cl) lo, hi, st = boot_by_dinner(cl) other = sum(1 for s in st if (s <= 0 if mean > 0 else s >= 0)) / len(st) by_lines = side(x, y) if p < ALPHA else "not told apart" by_dinner = "ahead" if lo > 0 else ("behind" if hi < 0 else "not told apart") paired[m] = {"only_this_right": x, "only_the_word_counter_right": y, "p": p, "verdict": by_lines, "diff_points": 100 * mean, "se_lines_independent_points": 100 * se_iid, "se_by_dinner_points": 100 * se, "design_effect": (se / se_iid) ** 2, "by_dinner_z_p": math.erfc(abs(mean / se) / math.sqrt(2)), "by_dinner_bootstrap_95_points": [100 * lo, 100 * hi], "by_dinner_bootstrap_p": min(1.0, 2 * other), "verdict_by_dinner": by_dinner, "same_verdict": by_lines == by_dinner} shares = {} for m in PAGE_WILSON_SHARES: cl = lines_by_dinner(keys, dinner, lambda k: int(K[m][k])) mean, se = cluster_se(cl) lo, hi, _ = boot_by_dinner(cl) se_iid = math.sqrt(mean * (1 - mean) / len(keys)) w = R["s0"][m]["wilson"] shares[m] = {"right_of_108": R["s0"][m]["k"], "wilson_95": w, "by_dinner_bootstrap_95": [lo, hi], "se_independent": se_iid, "se_by_dinner": se, "design_effect": (se / se_iid) ** 2, "wider_by_dinner": (hi - lo) > (w[1] - w[0])} return {"what": "the six-way's 108 lines come from 16 dinners; each paired test against the word counter, " "and the three single-count Wilson intervals the page prints, with whole dinners " "resampled (%d resamples, seed %d, percentile interval) and with a dinner-robust " "standard error; a verdict is ahead or behind at p < 0.05 by the exact test, and by " "dinner when the resampled 95 per cent interval excludes zero" % (BOOT_REPS, BOOT_SEED), "dinners": len(sizes), "lines_per_dinner": [min(sizes), max(sizes)], "dinner_of_each_line": "the kit lock's sets.c.items_list, dinner_id joined on text_sha256", "kit_lock": src.cite(KIT_LOCK), "paired_against_the_word_counter": paired, "verdicts_changed_by_dinner": sum(not v["same_verdict"] for v in paired.values()), "paired_design_effects_over_1": sum(v["design_effect"] > 1 for v in paired.values()), "single_shares": shares, "single_shares_wider_by_dinner": sum(v["wider_by_dinner"] for v in shares.values())} def kev_order_rows(src, name): rows = src.rows(KEV_ORDER_FILES[name]) assert len({r["uid"] for r in rows}) == len(rows) return rows def checked_tests_in_each_option_order(src, R, K): nb = {k[0]: v for k, v in K["naive-bayes"].items()} out = {"what": "the exact McNemar test against the word counter in each of the six option orders Kev's " "server generates, joined on each line's uid; order 1 is the kit's order " "(post_hoc.six_order..kit_order_is_order_0)", "models": {}} for name in KEV_ORDER_FILES: rows = kev_order_rows(src, name) per = [] for o in range(len(rows[0]["choices_by_order"])): right = {r["uid"]: r["choices_by_order"][o] == r["label"] for r in rows} x, y, p = mcnemar(right, nb) per.append({"order": o + 1, "right": sum(right.values()), "only_this_right": x, "only_the_word_counter_right": y, "p": p}) assert [c["right"] for c in per] == R["reorder"][name]["right_by_order"], name low = min(per, key=lambda c: c["p"]) out["models"][name] = {"orders": per, "closest": {"order": low["order"], "p": low["p"]}, "told_apart_in_any_order": any(c["p"] < ALPHA for c in per), "src": src.cite(KEV_ORDER_FILES[name])} four = [(out["models"][m]["closest"]["p"], m, out["models"][m]["closest"]["order"]) for m in PAGE_SHUFFLED_FOUR] p, m, o = min(four) out["the_page_four"] = {"models": PAGE_SHUFFLED_FOUR, "closest": {"model": m, "order": o, "p": p}, "told_apart_in_any_order": any(out["models"][x]["told_apart_in_any_order"] for x in PAGE_SHUFFLED_FOUR)} return out def checked_paired_shuffle_counts(src, R): moved = {} for name in sorted({m for pair in MOVED_PAIRS for m in pair}): moved[name] = {r["uid"]: len(set(r["choices_by_order"])) > 1 for r in kev_order_rows(src, name)} assert sum(moved[name].values()) == R["reorder"][name]["moved"], name out = {"what": "exact McNemar tests on which lines' answers moved over the six orders Kev's server " "generates, joined on each line's uid", "moved": {m: sum(v.values()) for m, v in moved.items()}, "pairs": {}, "src": {m: src.cite(KEV_ORDER_FILES[m]) for m in moved}} for a, b in MOVED_PAIRS: x, y, p = mcnemar(moved[a], moved[b]) out["pairs"][a + "|" + b] = {"only_first_moved": x, "only_second_moved": y, "p": p, "told_apart": p < ALPHA} return out def checked_door_9b_against_the_live_door(src): kit = door_kit(src) live = {r["id"]: r["outcome"] == "refused" for r in src.rows(LIVE_DOOR) if r["bucket"] in "ABCDEF"} rows = [r for r in src.rows(NINE_B_DOOR) if r.get("planted") is True] assert sorted(r["id"] for r in rows) == sorted(live) and len(live) == 36 named = {r["id"]: bool(r["in_v12_contract"]) for r in rows} nine = {r["id"]: r.get("choice") == "refuse" for r in rows} only_nine = [i for i in live if nine[i] and not live[i]] only_live = [i for i in live if live[i] and not nine[i]] return {"what": "APUS-OpenJev-v1 9B at high against the 2026-09-13 live door on the same 36 planted lines, " "joined by id; a kind is named when the door's v1.2 rules name it. Ids, counts and " "kinds only: no line is read", "planted_lines": len(live), "nine_b_refused": sum(nine.values()), "live_door_refused": sum(live.values()), "only_the_9b_refused": len(only_nine), "only_the_9b_refused_named": sum(named[i] for i in only_nine), "only_the_9b_refused_unnamed": sum(not named[i] for i in only_nine), "only_the_live_door_refused": len(only_live), "only_the_live_door_refused_named": sum(named[i] for i in only_live), "only_the_live_door_refused_unnamed": sum(not named[i] for i in only_live), "p": exact_mcnemar_p(len(only_nine), len(only_live)), "the_rows_gemma_choice_equals_the_0913_verdict_on_every_line": all((r.get("gemma_choice") == "refuse") == live[r["id"]] for r in rows), "each_rows_kind_equals_the_kit": all(named[i] == bool(kit[i][2]) for i in named), "src": [src.cite(NINE_B_DOOR), src.cite(LIVE_DOOR), src.cite(DOOR_KIT)]} def checked_gemma4_floored_lines(src, R): out = {"what": "rows whose `floored` list is not empty: at least one option letter fell outside the " "log-probabilities the server returned, so its score was floored"} for name, rel in FLOORED_FILES: rows = src.rows(rel) assert len(rows) == R["s0"][name]["n"] out[name] = {"rows": len(rows), "floored_rows": sum(1 for r in rows if r.get("floored")), "right": R["s0"][name]["k"], "src": src.cite(rel)} return out def checked_mistral_other_run(src, R): a, b = src.rows(MISTRAL_CARD_A), src.rows(MISTRAL_WORKSTATION) out = score(b, "generate") out.update({ "what": "Mistral Small 3.2 24B's six-way run on the workstation card, scored as the recount scores " "the card-A run (s0.mistral-small3.2-24b-generate-coi): strict, a reply with no readable " "letter wrong", "card_a_run_right": R["s0"]["mistral-small3.2-24b-generate-coi"]["k"], "same_prompts_in_the_same_order_as_the_card_a_run": [r["prompt_sha256"] for r in a] == [r["prompt_sha256"] for r in b], "window_utc": [min(r["stamp"] for r in b), max(r["stamp"] for r in b)], "card_a_run_window_utc": [min(r["stamp"] for r in a), max(r["stamp"] for r in a)], "src": src.cite(MISTRAL_WORKSTATION)}) return out def main(): root, out = sys.argv[1], sys.argv[2] src = Src(root) J = "bench/jev-2026-09-21/" R = {"ref": src.head, "s0": {}, "h48": {}, "door": {}, "reorder": {}, "task_a": {}, "field": {}, "judge": {}, "extra": {}} def put(section, name, rel, fn, *a, **kw): try: v = fn(src.rows(rel), *a, **kw) except FileNotFoundError: v = {"missing": True} v["src"] = src.cite(rel) R[section][name] = v # ---- S0, the six-way (108) ---- S0 = [ ("openjev-bf16-readout-arm3", J + "rows-arm3/openjev-bf16-largecard.c.jsonl", "choice"), ("openjev-fp8-readout-arm3", J + "rows-arm3/openjev-fp8-largecard.c.jsonl", "choice"), ("openjev-fp8-readout-arm2", J + "rows/openjev-fp8-readout.c.jsonl", "choice"), ("openjev-fp8-generate-arm2", J + "rows/openjev-fp8-generate.c.jsonl", "generate"), ("openjev-q4km-readout-arm13", J + "rows-oj-gguf/openjev-q4km-readout.c.jsonl", "choice"), ("openjev-q4km-generate-arm13", J + "rows-oj-gguf/openjev-q4km-generate.c.jsonl", "generate"), ("gemma4-26b-q4-base-readout-arm1", J + "rows/base-readout.c.jsonl", "choice"), ("gemma4-26b-q4-base-generate-arm1", J + "rows/base-generate.c.jsonl", "generate"), ("gemma4-26b-fp8-readout-arm10", J + "rows-addenda/gemma4-fp8-readout.c.jsonl", "choice"), ("gemma4-26b-fp8-generate-arm10", J + "rows-addenda/gemma4-fp8-generate.c.jsonl", "generate"), ("jevify-readout-arm1", J + "rows/jev-readout.c.jsonl", "choice"), ("jevify-generate-arm1", J + "rows/jev-generate.c.jsonl", "generate"), ("mistral-small3.2-24b-generate-coi", "bench/cost-of-intelligence-2026-09-23/benchbox-3090-reading-0/rows/b-mistral-small3.2-24b/b-mistral-small3.2-24b.c.jsonl", "generate"), ("kev-9b", J + "rows-gaps-card/kev-9b.kc.jsonl", "choice"), ("kev-4b", J + "rows-gaps-card/kev-4b.kc.jsonl", "choice"), ("naive-bayes", "bench/deem-2026-09-27/t1-tune/rows/t1-baseline-nb.s0.P.rep1.jsonl", "carried"), ("lev-4b-cpu", "bench/lev-2026-09-28/l1/rows/lev-4b-cpu.kc.jsonl", "choice"), ("lev-4b-gpu", "bench/lev-2026-09-28/rows/lev-4b.kc.jsonl", "choice"), ("apus-9b-high", "bench/apus-2026-09-28/rows/apus-9b-high.kc.jsonl", "choice"), ("apus-9b-low", "bench/apus-2026-09-28/rows/apus-9b-low.kc.jsonl", "choice"), ("apus-4b-high", "bench/apus-2026-09-28/a1b/rows/apus-4b-high.kc.jsonl", "choice"), ("apus-4b-low", "bench/apus-2026-09-28/a1b/rows/apus-4b-low.kc.jsonl", "choice"), ("imajev-4b-cpu", "bench/imajev-2026-09-28/rows/imajev-4b-cpu.kc.jsonl", "choice"), ("opendecider-nano", "bench/opendecider-2026-09-28/rows/o1-cpu-nano.s0.P.rep1.jsonl", "carried"), ("opendecider-small", "bench/opendecider-2026-09-28/rows/o2-gpu-small.s0.P.rep1.jsonl", "carried"), ("opendecider-base-qwen3-4b", "bench/opendecider-2026-09-28/rows/o2c-gpu-qwen3-4b-base.s0.P.rep1.jsonl", "carried"), ("deem-0.8-d0", "bench/deem-2026-09-27/rows/d0-cpu-0.8.s0.P.rep1.jsonl", "carried"), ("deem-t1-null-20260927", "bench/deem-2026-09-27/t1-tune/rows/t1-null-20260927.s0.P.rep1.jsonl", "carried"), ("deem-t1-null-20260928", "bench/deem-2026-09-27/t1-tune/rows/t1-null-20260928.s0.P.rep1.jsonl", "carried"), ("deem-t1-null-20260929", "bench/deem-2026-09-27/t1-tune/rows/t1-null-20260929.s0.P.rep1.jsonl", "carried"), ] for name, rel, kind in S0: put("s0", name, rel, score, kind, read_ties=not name.startswith("kev")) # naive Bayes with the named-guest exclusion (printed, never gated) nb = src.rows("bench/deem-2026-09-27/t1-tune/rows/t1-baseline-nb.s0.P.rep1.jsonl") R["extra"]["naive-bayes-exclusion-s0"] = {"k": sum(r["correct_excl"] for r in nb), "n": len(nb), "src": src.cite("bench/deem-2026-09-27/t1-tune/rows/t1-baseline-nb.s0.P.rep1.jsonl")} # Deem D0 latency: rep 1 was not instrumented; rep 2 carries the uncontended flag (results.md) rel = "bench/deem-2026-09-27/rows/d0-cpu-0.8.s0.P.rep2.jsonl" rr = src.rows(rel) un = [r["seconds"] for r in rr if r.get("contended") is False] R["extra"]["deem-d0-rep2-uncontended"] = {"k": sum(r["correct"] for r in rr), "n": len(rr), "timed_rows": len(un), "median_s": statistics.median(un), "p95_s": p95(un), "same_choice_as_rep1": sum(a["choice"] == b["choice"] for a, b in zip(rr, src.rows("bench/deem-2026-09-27/rows/d0-cpu-0.8.s0.P.rep1.jsonl"))), "src": src.cite(rel)} # ---- exact McNemar on S0, joined item by item ---- K = {name: keyed(src, rel, kind) for name, rel, kind in S0 if not name.startswith("deem-t1")} R["mcnemar"] = {} for ref in ("naive-bayes", "kev-9b"): for name in K: if name == ref: continue x, y, p = mcnemar(K[name], K[ref]) R["mcnemar"][f"{name}|{ref}"] = {"only_first_right": x, "only_second_right": y, "p": p} # ---- H48 (48) ---- H = [ ("openjev-q4km-readout-arm13", J + "rows-oj-gguf/openjev-q4km-readout.h48.jsonl", "choice"), ("openjev-q4km-generate-arm13", J + "rows-oj-gguf/openjev-q4km-generate.h48.jsonl", "generate"), ("naive-bayes", "bench/deem-2026-09-27/t1-tune/rows/t1-baseline-nb.h48.P.rep1.jsonl", "carried"), ("apus-9b-high", "bench/apus-2026-09-28/rows/apus-9b-high.kh.jsonl", "choice"), ("apus-9b-low", "bench/apus-2026-09-28/rows/apus-9b-low.kh.jsonl", "choice"), ("apus-4b-high", "bench/apus-2026-09-28/a1b/rows/apus-4b-high.kh.jsonl", "choice"), ("apus-4b-low", "bench/apus-2026-09-28/a1b/rows/apus-4b-low.kh.jsonl", "choice"), ("imajev-4b-gpu", "bench/imajev-2026-09-28/i2/rows/imajev-4b-3090.kh.jsonl", "choice"), ("opendecider-nano", "bench/opendecider-2026-09-28/rows/o1-cpu-nano.h48.P.rep1.jsonl", "carried"), ("opendecider-small", "bench/opendecider-2026-09-28/rows/o2-gpu-small.h48.P.rep1.jsonl", "carried"), ("opendecider-base-qwen3-4b", "bench/opendecider-2026-09-28/rows/o2c-gpu-qwen3-4b-base.h48.P.rep1.jsonl", "carried"), ("deem-0.8-d0", "bench/deem-2026-09-27/rows/d0-cpu-0.8.h48.P.rep1.jsonl", "carried"), ] for name, rel, kind in H: put("h48", name, rel, score, kind) # ---- S0 - H48 ---- pairs = {"openjev-q4km-readout-arm13": "openjev-q4km-readout-arm13", "naive-bayes": "naive-bayes", "apus-9b-high": "apus-9b-high", "apus-9b-low": "apus-9b-low", "apus-4b-high": "apus-4b-high", "apus-4b-low": "apus-4b-low", "imajev-4b-gpu": "imajev-4b-cpu", "opendecider-nano": "opendecider-nano", "opendecider-small": "opendecider-small", "opendecider-base-qwen3-4b": "opendecider-base-qwen3-4b", "deem-0.8-d0": "deem-0.8-d0"} R["s0_minus_h48"] = {} for h, s in pairs.items(): a, b = R["s0"][s], R["h48"][h] lo, hi = newcombe(a["k"], a["n"], b["k"], b["n"]) R["s0_minus_h48"][h] = {"s0": f"{a['k']}/{a['n']}", "h48": f"{b['k']}/{b['n']}", "diff_pts": 100 * (a["k"] / a["n"] - b["k"] / b["n"]), "newcombe": [100 * lo, 100 * hi], "s0_from": s} # ---- doorman (36 planted, 12 controls) ---- D = [ ("openjev-fp8-readout-arm4", J + "rows-addenda/openjev-fp8-readout.d.jsonl"), ("openjev-q4km-readout-arm13", J + "rows-oj-gguf/openjev-q4km-readout.d.jsonl"), ("openjev-q4km-generate-arm13", J + "rows-oj-gguf/openjev-q4km-generate.d.jsonl"), ("gemma4-26b-fp8-readout-arm10", J + "rows-addenda/gemma4-fp8-readout.d.jsonl"), ("kev-9b", J + "rows-gaps-card/kev-9b.kd.jsonl"), ("kev-4b", J + "rows-gaps-card/kev-4b.kd.jsonl"), ("lev-4b-gpu", "bench/lev-2026-09-28/rows/lev-4b.kd.jsonl"), ("apus-9b-high", "bench/apus-2026-09-28/rows/apus-9b-high.kd.jsonl"), ("apus-9b-low", "bench/apus-2026-09-28/rows/apus-9b-low.kd.jsonl"), ("apus-4b-high", "bench/apus-2026-09-28/a1b/rows/apus-4b-high.kd.jsonl"), ("apus-4b-low", "bench/apus-2026-09-28/a1b/rows/apus-4b-low.kd.jsonl"), ("imajev-4b-cpu", "bench/imajev-2026-09-28/rows/imajev-4b-cpu.kd.jsonl"), ("opendecider-nano", "bench/opendecider-2026-09-28/rows/o1-cpu-nano.door.P.rep1.jsonl"), ("opendecider-small", "bench/opendecider-2026-09-28/rows/o2-gpu-small.door.P.rep1.jsonl"), ("opendecider-base-qwen3-4b", "bench/opendecider-2026-09-28/rows/o2c-gpu-qwen3-4b-base.door.P.rep1.jsonl"), ] for name, rel in D: put("door", name, rel, doorman, read_ties=not name.startswith("kev")) rel = "bench/doorman-planted-2026-09-13/out/verdicts.jsonl" v = doorman_0913(src, rel) v["src"] = src.cite(rel) R["door"]["live-doorman-gemma4-26b-fp8-0913"] = v # ---- reorder ---- for name, rel in [("kev-9b", J + "rows-gaps-card/kev-9b.kperm.jsonl"), ("kev-4b", J + "rows-gaps-card/kev-4b.kperm.jsonl"), ("lev-4b-gpu", "bench/lev-2026-09-28/rows/lev-4b.kperm.jsonl"), ("apus-9b-high", "bench/apus-2026-09-28/rows/apus-9b-high.kperm.jsonl"), ("apus-4b-high", "bench/apus-2026-09-28/a1b/rows/apus-4b-high.kperm.jsonl")]: put("reorder", name, rel, reorder_perm) for name, rel in [("openjev-fp8-readout-gap4", J + "rows-gaps-card/openjev-fp8-readout.creorder.jsonl"), ("openjev-q4km-readout-arm13", J + "rows-oj-gguf/openjev-q4km-readout.creorder.jsonl")]: put("reorder", name, rel, reorder_grouped) for name, arm in [("opendecider-nano", "o1-cpu-nano"), ("opendecider-small", "o2-gpu-small"), ("opendecider-base-qwen3-4b", "o2c-gpu-qwen3-4b-base")]: base = f"bench/opendecider-2026-09-28/rows/{arm}.s0.P.rep1" rows = src.rows(base + ".jsonl") + src.rows(base + ".k1-5.jsonl") v = reorder_grouped(rows) v["src"] = src.cite(base + ".jsonl") + " + " + src.cite(base + ".k1-5.jsonl") R["reorder"][name] = v # do the orders match across files? (same six orders or not) def order_sig(rel, grouped): rows = src.rows(rel) if grouped: return sorted((r["uid"], r["order_k"], tuple(r["permutation"])) for r in rows) sig = [] for r in rows: o = r["orders"] if not isinstance(r["orders"], str) else json.loads(r["orders"]) sig.append((r["uid"], tuple(tuple(x) for x in o))) return sorted(sig) sk9 = order_sig(J + "rows-gaps-card/kev-9b.kperm.jsonl", False) R["extra"]["reorder_same_orders"] = { "kev-4b_vs_kev-9b": order_sig(J + "rows-gaps-card/kev-4b.kperm.jsonl", False) == sk9, "lev_vs_kev-9b": order_sig("bench/lev-2026-09-28/rows/lev-4b.kperm.jsonl", False) == sk9, "apus-9b_vs_kev-9b": order_sig("bench/apus-2026-09-28/rows/apus-9b-high.kperm.jsonl", False) == sk9, "apus-4b_vs_kev-9b": order_sig("bench/apus-2026-09-28/a1b/rows/apus-4b-high.kperm.jsonl", False) == sk9, "oj-q4_vs_oj-fp8-gap4": order_sig(J + "rows-oj-gguf/openjev-q4km-readout.creorder.jsonl", True) == order_sig(J + "rows-gaps-card/openjev-fp8-readout.creorder.jsonl", True), } # ---- task (a), the long pages ---- for name, rel in [("openjev-fp8-readout-arm2-16k", J + "rows/openjev-fp8-readout.a.jsonl"), ("openjev-fp8-readout-32k-along", J + "rows-addenda/openjev-fp8-readout-32k.along.jsonl"), ("gemma4-26b-fp8-readout-arm10", J + "rows-addenda/gemma4-fp8-readout.a.jsonl"), ("kev-9b", J + "rows-gaps-card/kev-9b.ka.jsonl"), ("kev-4b", J + "rows-gaps-card/kev-4b.ka.jsonl"), ("lev-4b-gpu", "bench/lev-2026-09-28/rows/lev-4b.ka.jsonl"), ("apus-9b-high", "bench/apus-2026-09-28/rows/apus-9b-high.ka.jsonl"), ("apus-4b-high", "bench/apus-2026-09-28/a1b/rows/apus-4b-high.ka.jsonl")]: put("task_a", name, rel, task_a) # ---- the field exam (40) ---- for name, rel in [("openjev-fp8-readout-arm5", J + "rows-addenda/openjev-fp8-readout.f.jsonl"), ("gemma4-26b-fp8-readout-arm10", J + "rows-addenda/gemma4-fp8-readout.f.jsonl"), ("kev-9b", J + "rows-gaps-card/kev-9b.kf.jsonl"), ("kev-4b", J + "rows-gaps-card/kev-4b.kf.jsonl"), ("lev-4b-gpu", "bench/lev-2026-09-28/rows/lev-4b.kf.jsonl"), ("apus-9b-high", "bench/apus-2026-09-28/rows/apus-9b-high.kf.jsonl"), ("apus-4b-high", "bench/apus-2026-09-28/a1b/rows/apus-4b-high.kf.jsonl"), ("imajev-4b-gpu", "bench/imajev-2026-09-28/i2/rows/imajev-4b-3090.kf.jsonl")]: put("field", name, rel, score, "choice", read_ties=not name.startswith("kev")) # ---- the judge seat (119) ---- for name, rel in [("openjev-fp8-readout-arm6", J + "rows-addenda/openjev-fp8-readout.j.jsonl"), ("gemma4-26b-fp8-readout-arm10", J + "rows-addenda/gemma4-fp8-readout.j.jsonl"), ("kev-9b", J + "rows-gaps-card/kev-9b.kj.jsonl"), ("kev-4b", J + "rows-gaps-card/kev-4b.kj.jsonl"), ("lev-4b-gpu", "bench/lev-2026-09-28/rows/lev-4b.kj.jsonl"), ("apus-9b-high", "bench/apus-2026-09-28/rows/apus-9b-high.kj.jsonl"), ("apus-4b-high", "bench/apus-2026-09-28/a1b/rows/apus-4b-high.kj.jsonl"), ("imajev-4b-gpu", "bench/imajev-2026-09-28/i2/rows/imajev-4b-3090.kj.jsonl")]: put("judge", name, rel, judge) # ---- post hoc: descriptive, computed after the rows were seen (never registered) ---- post_hoc(src, R, S0, D) # ---- checked for the page: descriptive, computed after the rows were seen (never registered). # One statement a check, so a run over files where one check's inputs are missing still runs the rest. R["checked"] = {"label": CHECKED, "note": CHECKED_NOTE} R["checked"]["holm"] = checked_holm(R) R["checked"]["kev9b_paired_interval"] = checked_kev9b_paired_interval(R) R["checked"]["bound_on_0_of_12"] = checked_bound_on_0_of_12(R) R["checked"]["heldout_interval_widths"] = checked_heldout_interval_widths(R) R["checked"]["resample_by_dinner"] = checked_resample_by_dinner(src, R, K) R["checked"]["tests_in_each_option_order"] = checked_tests_in_each_option_order(src, R, K) R["checked"]["paired_shuffle_counts"] = checked_paired_shuffle_counts(src, R) R["checked"]["door_9b_against_the_live_door"] = checked_door_9b_against_the_live_door(src) R["checked"]["gemma4_floored_lines"] = checked_gemma4_floored_lines(src, R) R["checked"]["mistral_other_run"] = checked_mistral_other_run(src, R) with open(out, "w") as f: json.dump(R, f, indent=1, sort_keys=True) print("wrote", out, "ref", src.head) if __name__ == "__main__": main()