#!/usr/bin/env python3 """parity.py -- RENDER PARITY and RAW-PATH PARITY (SPEC Δ6 item 2; PLAN-v2 §2.4; CRITIQUE-method M-1 fix 1). Two captures, two rules, one file (`rows/parity.json`, written through the pen): chat-render parity (R-P1, REPORTED, never refused): each engine's chat render of a prompt as token IDs -- vLLM POST /tokenize with messages; Ollama /api/chat and /v1/chat/completions with _debug_render_only, then its runner's /tokenize; llama-server /apply-template, then /tokenize. Identical IDs, or the difference printed: the first differing index, 8 tokens either side decoded with their pieces, and both counts. M-1 predicts Ollama's render DIFFERS (no empty thought block for gemma4: model/renderers/renderer.go L102-105 at v0.34.4). The headline goes on the raw path, and a chat row is labelled "as a user calls it". raw-path parity (R-P2, a GATE): EVERY string an arm sends on the raw path, on every engine the arm uses: the ID list's count and sha256. Identical on every engine, with BOS first exactly once (no double BOS), or that prompt is REFUSED for that engine pair and carries no scored row (raw_parity_broken). R-7x, per stream: the engine's own prompt count (vLLM usage.prompt_tokens, Ollama prompt_eval_count, llama-server timings.prompt_n + cache_n) must equal the parity count of the string it was sent, or the stream fails token_count_mismatch. A chat row compares against the engine's own chat-render count from R-P1. """ import hashlib import json import seatlib as L WINDOW = 8 def ids_sha(ids): return hashlib.sha256(json.dumps(list(ids), separators=(",", ":")).encode("utf-8")).hexdigest() def bos_id(engine): """The BOS ID as this engine adds it: the one token of an empty string tokenized with specials added.""" ids = engine.tokenize_raw("", add_special=True) return ids[0] if ids else None def bos_once(ids, bos): return bool(ids) and bos is not None and ids[0] == bos and (len(ids) < 2 or ids[1] != bos) def raw_ids(engine, strings, bos=None): """{key: {count, sha256, head, bos_once}} for every raw string `strings` = {key: S} on one engine.""" bos = bos_id(engine) if bos is None else bos out = {} for key, text in strings.items(): ids = engine.tokenize_raw(text, add_special=True) out[key] = {"count": len(ids), "sha256": ids_sha(ids), "head": ids[:6], "bos_once": bos_once(ids, bos)} return out, bos def raw_compare(per_engine): """R-P2 across engines: per_engine = {engine: {key: raw_ids record}}. Returns {key: {verdict ok|broken, by_engine, sentence}}; a key missing on an engine is broken (never assumed equal).""" engines = sorted(per_engine) keys = sorted({k for recs in per_engine.values() for k in recs}) out = {} for k in keys: by = {e: per_engine[e].get(k) for e in engines} problems = [] missing = [e for e, r in by.items() if r is None] if missing: problems.append(f"not tokenized on {', '.join(missing)}") present = {e: r for e, r in by.items() if r is not None} double = [e for e, r in present.items() if not r["bos_once"]] if double: problems.append(f"BOS is not first exactly once on {', '.join(double)}") shas = {r["sha256"] for r in present.values()} if len(shas) > 1: items = sorted(present.items()) a, b = items[0], next((x for x in items[1:] if x[1]["sha256"] != items[0][1]["sha256"]), items[1]) problems.append(f"{a[0]} {a[1]['count']} IDs sha {a[1]['sha256'][:12]}, {b[0]} {b[1]['count']} sha " f"{b[1]['sha256'][:12]}") out[k] = {"verdict": "broken" if problems else "ok", "by_engine": by, "count": next(iter(present.values()))["count"] if present and len(shas) == 1 else None, "problems": problems} return out def refusals(compare, prompt_of): """The R-P2 refusals by prompt: {prompt: sentence} for every prompt with a broken string (the prompt is refused for the engine pair; `prompt_of(key)` names a string's prompt).""" out = {} for k, r in compare.items(): if r["verdict"] == "broken": pid = prompt_of(k) out.setdefault(pid, f"REFUSED: raw parity broken for {pid} ({'; '.join(r['problems'])})") return out def window(ids, i, pieces_fn, all_pieces=None): """8 IDs either side of index i with their pieces: from `all_pieces` (each ID's piece, read on the engine while it was up) when given, else from `pieces_fn(segment)`.""" lo, hi = max(0, i - WINDOW), min(len(ids), i + WINDOW + 1) seg = ids[lo:hi] if all_pieces is not None and len(all_pieces) == len(ids): return {"from": lo, "ids": seg, "pieces": list(all_pieces[lo:hi]), "pieces_from": "the engine's detokenizer"} try: pieces = pieces_fn(seg) except Exception as e: # noqa: BLE001 -- a window that cannot be decoded is printed as IDs pieces = [f""] * len(seg) return {"from": lo, "ids": seg, "pieces": pieces} def chat_compare(renders, pieces_fn=None, base=None, pieces=None): """R-P1 over {render name: IDs or None}: identical, or the difference printed -- the first differing index, 8 IDs either side with their pieces, and every count -- each render against `base` (the canonical template's: vLLM's render, when read; else the first by name). NEVER a refusal. `pieces` = {render name: [each ID's piece]}, read on each engine while it was up (the fold's fairness N-3); without it a window prints the IDs.""" named = {k: v for k, v in renders.items() if v is not None} counts = {k: (len(v) if v is not None else None) for k, v in renders.items()} if len(named) < 2: return {"verdict": "unread", "counts": counts, "why": "fewer than two renders were read", "refused": False, "rule": L.RULES["render_parity"]} base_name = base if base in named else sorted(named)[0] ref = named[base_name] diffs = [] for other in sorted(n for n in named if n != base_name): ids = named[other] if ids == ref: continue i = next((j for j, (a, b) in enumerate(zip(ref, ids)) if a != b), min(len(ref), len(ids))) pf = pieces_fn or (lambda seg: [str(x) for x in seg]) pc = pieces or {} diffs.append({"a": base_name, "b": other, "first_index": i, "a_window": window(ref, i, pf, pc.get(base_name)), "b_window": window(ids, i, pf, pc.get(other)), "a_count": len(ref), "b_count": len(ids)}) return {"verdict": "differs" if diffs else "identical", "base": base_name, "counts": counts, "diffs": diffs, "shas": {k: ids_sha(v) for k, v in named.items()}, "refused": False, "rule": L.RULES["render_parity"]} def r7x(stream_count, parity_count): """None when the stream's own prompt count is the parity count, else the failure sentence.""" if parity_count is None: return "no parity count was recorded for this string" if stream_count != parity_count: return f"prompt_tokens {stream_count} ≠ parity count {parity_count}" return None def frame_proof(chat_ids, composed_ids, prompt_id): """R-F1: vLLM's chat render IDs == the composed string's IDs, ID for ID; else the refusal with the first index.""" if chat_ids == composed_ids: return {"prompt": prompt_id, "verdict": "ok", "count": len(chat_ids)} i = next((j for j, (a, b) in enumerate(zip(chat_ids, composed_ids)) if a != b), min(len(chat_ids), len(composed_ids))) return {"prompt": prompt_id, "verdict": "refused", "first_index": i, "chat_count": len(chat_ids), "composed_count": len(composed_ids), "sentence": f"REFUSED: the frozen frame is not the template for {prompt_id}: chat IDs and composed IDs " f"differ at index {i}"}