"""Cluster-robust checks on the six-way: the 108 lines come from 16 dinners. Reuses recount_v2's row scoring (row_right) and its S0 file list, but reads rows directly (no git) from the lead-ledger worktree. Every count is compared to recount-v2.json before it is used. """ import importlib.util import json import math import random import sys from collections import defaultdict ROOT = "/workshop/estate-worktrees/lead-ledger" DM = "/workshop/bench-archive/plans-2026-09-28/decision-models" spec = importlib.util.spec_from_file_location("rc", f"{DM}/recount_v2.py") rc = importlib.util.module_from_spec(spec) spec.loader.exec_module(rc) R = json.load(open(f"{DM}/recount-v2.json")) def rows(rel): out = [] for line in open(f"{ROOT}/{rel}", encoding="utf-8"): if line.strip(): r = json.loads(line) if isinstance(r.get("probs"), str): try: r["probs"] = json.loads(r["probs"]) except ValueError: pass out.append(r) return out J = "bench/jev-2026-09-21/" KEV9B = J + "rows-gaps-card/kev-9b.kc.jsonl" FILES = { "openjev-bf16-readout-arm3": (J + "rows-arm3/openjev-bf16-largecard.c.jsonl", "choice"), "gemma4-26b-q4-base-generate-arm1": (J + "rows/base-generate.c.jsonl", "generate"), "jevify-readout-arm1": (J + "rows/jev-readout.c.jsonl", "choice"), "mistral-small3.2-24b-generate-coi": ("bench/cost-of-intelligence-2026-09-23/benchbox-3090-reading-0/rows/b-mistral-small3.2-24b/b-mistral-small3.2-24b.c.jsonl", "generate"), "kev-9b": (KEV9B, "choice"), "kev-4b": (J + "rows-gaps-card/kev-4b.kc.jsonl", "choice"), "imajev-4b-cpu": ("bench/imajev-2026-09-28/rows/imajev-4b-cpu.kc.jsonl", "choice"), "apus-4b-high": ("bench/apus-2026-09-28/a1b/rows/apus-4b-high.kc.jsonl", "choice"), "apus-9b-high": ("bench/apus-2026-09-28/rows/apus-9b-high.kc.jsonl", "choice"), "lev-4b-gpu": ("bench/lev-2026-09-28/rows/lev-4b.kc.jsonl", "choice"), "opendecider-small": ("bench/opendecider-2026-09-28/rows/o2-gpu-small.s0.P.rep1.jsonl", "carried"), "opendecider-base-qwen3-4b": ("bench/opendecider-2026-09-28/rows/o2c-gpu-qwen3-4b-base.s0.P.rep1.jsonl", "carried"), "naive-bayes": ("bench/deem-2026-09-27/t1-tune/rows/t1-baseline-nb.s0.P.rep1.jsonl", "carried"), } ref = rows(KEV9B) REF_KEYS = [(r["uid"], r["text_sha256"]) for r in ref] def keyed(rel, kind): rr = rows(rel) if "uid" in rr[0] and "text_sha256" in rr[0]: keys = [(r["uid"], r["text_sha256"]) for r in rr] else: assert [r["label"] for r in rr] == [r["label"] for r in ref], rel keys = REF_KEYS return dict(zip(keys, (rc.row_right(r, kind) for r in rr))) K = {m: keyed(rel, kind) for m, (rel, kind) in FILES.items()} for m, k in K.items(): got = sum(k.values()) want = R["s0"][m]["k"] assert got == want, (m, got, want) print("per-model counts equal recount-v2.json s0 k for all", len(K), "models") lock = json.load(open(f"{ROOT}/bench/cost-of-intelligence-2026-09-23/KIT.lock.json"))["sets"]["c"]["items_list"] dinner_of = {it["text_sha256"]: it["dinner_id"] for it in lock} dinners = sorted(set(dinner_of.values())) print("dinners:", len(dinners)) def per_dinner(a, b): """For each dinner: (n, sum of (a_right - b_right)).""" agg = defaultdict(lambda: [0, 0]) for key in a: dnr = dinner_of[key[1]] agg[dnr][0] += 1 agg[dnr][1] += int(a[key]) - int(b[key]) return [tuple(agg[d]) for d in dinners] def per_dinner_single(a): agg = defaultdict(lambda: [0, 0]) for key in a: dnr = dinner_of[key[1]] agg[dnr][0] += 1 agg[dnr][1] += int(a[key]) return [tuple(agg[d]) for d in dinners] def cluster_z(clusters): """Ratio-estimator cluster-robust SE for a mean of per-line values.""" N = sum(n for n, _ in clusters) mean = sum(s for _, s in clusters) / N G = len(clusters) resid = [(s - n * mean) for n, s in clusters] var = (G / (G - 1)) * sum(r * r for r in resid) / N ** 2 return mean, math.sqrt(var) def boot_ci(clusters, reps=20000, seed=20260929): rnd = random.Random(seed) G = len(clusters) stats = [] for _ in range(reps): pick = [clusters[rnd.randrange(G)] for _ in range(G)] N = sum(n for n, _ in pick) stats.append(sum(s for _, s in pick) / N) stats.sort() return stats[int(0.025 * reps)], stats[int(0.975 * reps) - 1], stats def iid_se_paired(a, b): n = len(a) ds = [int(a[k]) - int(b[k]) for k in a] m = sum(ds) / n return m, math.sqrt(sum((x - m) ** 2 for x in ds) / (n - 1) / n) print("\n== paired differences v the naive Bayes: iid v by-dinner ==") for m in ["openjev-bf16-readout-arm3", "jevify-readout-arm1", "gemma4-26b-q4-base-generate-arm1", "mistral-small3.2-24b-generate-coi", "kev-9b", "imajev-4b-cpu", "apus-4b-high", "apus-9b-high", "lev-4b-gpu", "kev-4b", "opendecider-small", "opendecider-base-qwen3-4b"]: a, b = K[m], K["naive-bayes"] m_iid, se_iid = iid_se_paired(a, b) cl = per_dinner(a, b) m_cl, se_cl = cluster_z(cl) lo, hi, st = boot_ci(cl) z_cl = m_cl / se_cl if se_cl else float("inf") # t with G-1 = 15 df for the cluster statistic p_norm = math.erfc(abs(z_cl) / math.sqrt(2)) # bootstrap two-sided p: share of resampled means on the other side of zero, doubled other = sum(1 for s in st if (s <= 0 if m_cl > 0 else s >= 0)) / len(st) deff = (se_cl / se_iid) ** 2 if se_iid else float("nan") x, y, p_mc = rc.mcnemar(a, b) print(f" {m:34s} diff {100*m_cl:+5.1f} pts | McNemar {x}-{y} p {p_mc:.2g} | iid SE {100*se_iid:.2f} " f"cluster SE {100*se_cl:.2f} (design effect {deff:.2f}) | cluster z p {p_norm:.2g} | " f"boot 95% {100*lo:+.1f} to {100*hi:+.1f}, boot p {min(1,2*other):.3g}") print("\n== Gemma 4 v Kev-9B, and Kev-4B v Kev-9B (the other pooled claims) ==") for a_, b_ in [("gemma4-26b-q4-base-generate-arm1", "kev-9b"), ("kev-4b", "kev-9b"), ("mistral-small3.2-24b-generate-coi", "kev-9b")]: a, b = K[a_], K[b_] cl = per_dinner(a, b) m_cl, se_cl = cluster_z(cl) lo, hi, st = boot_ci(cl) x, y, p_mc = rc.mcnemar(a, b) print(f" {a_} v {b_}: {x}-{y} McNemar p {p_mc:.2g}; cluster z p {math.erfc(abs(m_cl/se_cl)/math.sqrt(2)):.2g}; " f"boot 95% {100*lo:+.1f} to {100*hi:+.1f}") print("\n== single-model share: Wilson v by-dinner bootstrap ==") for m in ["openjev-bf16-readout-arm3", "kev-9b", "naive-bayes", "apus-9b-high"]: cl = per_dinner_single(K[m]) mean, se = cluster_z(cl) lo, hi, _ = boot_ci(cl) w = R["s0"][m]["wilson"] iid_se = math.sqrt(mean * (1 - mean) / 108) print(f" {m:28s} {100*mean:.1f}% Wilson {100*w[0]:.1f}-{100*w[1]:.1f} | cluster SE {100*se:.2f} v iid {100*iid_se:.2f} " f"(design effect {(se/iid_se)**2:.2f}) | dinner bootstrap {100*lo:.1f}-{100*hi:.1f}") print("\n== per-dinner accuracy spread (Kev-9B, NB) ==") for m in ["kev-9b", "naive-bayes", "apus-9b-high"]: cl = per_dinner_single(K[m]) print(f" {m:14s}", " ".join(f"{s}/{n}" for n, s in cl))