#!/usr/bin/env python3 """nonces.py -- THE SEEDED NONCE per (arm, prompt, level, run, request): identical across engines (N-3). What this file is, for a reader who opened it cold (SPEC Δ6, the bench-client lane §3.11; PLAN-v2 §2.6). The sweep puts a short, deterministic ticket number at the head of every user text, so no engine's prompt cache can serve one request from another's (vLLM hashes 16-token blocks; llama.cpp matches the longest common prefix), and greedy outputs stay comparable across engines because the same key gives the same string everywhere. The anchor inside arm C reuses the first block's keys, so the A-B-A strings are identical. nonce(arm, prompt_id, level, run, request) = f"Ticket {d:08d}.\\n\\n", d = int.from_bytes(sha256(f"{NONCE_SEED}|{arm}|{prompt_id}|{level}|{run}|{request}").digest()[:8]) % 10**8 The seed is registered in prereg.py and never typed at a call site. The ENGINE is never in the key. R-N (a D gate): the design bets that Gemma's tokenizer splits digits one per token, so every nonce-bearing string of a prompt tokenizes to ONE count on each engine. D proves it rather than assuming it: a prompt whose strings do not share one count is REFUSED, never re-drawn silently (the lead rules the fix, an amendment). The nonce's own tokens must also fit NONCE_ROOM_TOKENS, the room the long-prompt ceiling leaves it (seatlib.long_prompt_ceiling). The nonce-leak FLAG (never a void): a sweep stream whose cached_tokens exceeds the nonce's END -- the tokens two strings share when they carry the same nonce and differ right after it (BOS + frame head + the whole nonce, read on the engine by arms.nonce_end) -- had a cache reach past the nonce into the prompt; both counts are printed. (The frame head alone flagged every llama.cpp stream: its LCP match reuses the nonce's constant `Ticket` too; fairness N-2.) """ import hashlib import prereg #: The room the long-prompt ceiling leaves for a nonce, in tokens (a registered upper bound: `Ticket`, eight digits, #: `.` and the blank line read ~11-12 on a digit-per-token tokenizer; D reads the real count and refuses past it). NONCE_ROOM_TOKENS = 16 NONCE_FORMAT = "Ticket {d:08d}.\n\n" def nonce(arm, prompt_id, level, run, request, seed=None): """The nonce text for one request; the engine is never part of the key.""" s = prereg.NONCE_SEED if seed is None else seed key = f"{s}|{arm}|{prompt_id}|{level}|{run}|{request}".encode("utf-8") d = int.from_bytes(hashlib.sha256(key).digest()[:8], "big") % 10 ** 8 return NONCE_FORMAT.format(d=d) def key(arm, prompt_id, level, run, request): return {"arm": arm, "prompt": prompt_id, "level": int(level), "run": int(run), "request": int(request)} def plan_nonces(cells): """Every nonce an arm will send: `cells` is [(arm, prompt_id, level, runs)] where runs counts warm-ups and scored runs together (run 0 is the warm-up); each level fires `level` requests per run. Returns [(key dict, nonce text)] in firing order -- what D checks in full (R-N, R-P2).""" out = [] for arm, pid, level, runs in cells: for run in range(int(runs)): for req in range(int(level)): out.append((key(arm, pid, level, run, req), nonce(arm, pid, level, run, req))) return out def collisions(planned): """[(nonce, [keys])] for any nonce text two different keys share within one plan (a collision check).""" seen = {} for k, text in planned: seen.setdefault(text, []).append(k) return [(t, ks) for t, ks in seen.items() if len(ks) > 1] class NonceRefused(Exception): """R-N: a prompt's nonce-bearing strings do not share one count on an engine, or overflow the room.""" def __init__(self, sentence, readings): self.sentence = sentence if sentence.startswith("REFUSED") else f"REFUSED: {sentence}" self.readings = readings super().__init__(self.sentence) def r_n_check(counts, room=NONCE_ROOM_TOKENS): """R-N over `counts` = {engine: {prompt_id: {"strings": [token count of each nonce-bearing string], "nonce_tokens": [token count of each bare nonce]}}}. Returns {engine: {prompt: {counts, verdict}}}; a prompt whose strings read more than one count on an engine, or whose nonce reads more tokens than the room, is refused (its entry reads REFUSED with the sentence; the caller refuses that prompt).""" out = {} for engine, per_prompt in counts.items(): for pid, c in per_prompt.items(): distinct = sorted(set(c.get("strings") or [])) nonce_counts = sorted(set(c.get("nonce_tokens") or [])) rec = {"string_counts": distinct, "nonce_token_counts": nonce_counts, "n": len(c.get("strings") or []), "room": room} if len(distinct) != 1: rec["verdict"] = "REFUSED" rec["sentence"] = (f"REFUSED: {pid}'s nonce strings tokenize to {len(distinct)} different counts " f"on {engine}") elif nonce_counts and max(nonce_counts) > room: rec["verdict"] = "REFUSED" rec["sentence"] = (f"REFUSED: {pid}'s nonce reads {max(nonce_counts)} tokens on {engine}, over the " f"{room}-token room the long-prompt ceiling leaves it") else: rec["verdict"] = "ok" out.setdefault(engine, {})[pid] = rec return out def nonce_leak(cached_tokens, nonce_end): """The nonce-leak FLAG for one stream, or None: cached_tokens past the nonce's end (never a void).""" if cached_tokens is None or nonce_end is None: return None if cached_tokens > nonce_end: return {"kind": "nonce_leak", "cached": cached_tokens, "nonce_end": nonce_end, "line": f"cached {cached_tokens} > nonce end {nonce_end}"} return None