#!/usr/bin/env python3 """gates.py -- THE ENGINE PATH'S GATES: hidden tokens, all layers, finish reasons, the void roll-up (PLAN-v2 §2.8). What this file is, for a reader who opened it cold (SPEC Δ6, the bench-client lane §3.13; items 3 and 5). Every gate runs AFTER a run, never inside the timed window, and says what it read and why: THE HIDDEN-TOKEN GATE (G-H; CRITIQUE-method M-1 fix 3). `hidden_surplus` counts the generated tokens that carry no visible text. On a `stop` finish one trailing end-of-generation token is allowed; on `length` none is. * method `ids` (exact) -- the stream carried its generated IDs (vLLM `token_ids` under return_token_ids, llama-server `tokens`): each ID's piece, specials kept, is read from the engine's own detokenizer; an ID whose piece is a special marker, empty, or missing from the visible text is surplus. When the pieces joined equal the visible text, the surplus is 0. A BYTE-FALLBACK run -- consecutive IDs whose single pieces are partial UTF-8 (they decode alone to U+FFFD, or read `<0xNN>`) -- is detokenized JOINTLY and is visible when that joint text is (the bench-client fold's fairness I-1: one em dash as three byte tokens is not three hidden tokens). * method `retokenize` -- Ollama's /api/generate returns no IDs: surplus = eval_count - the allowed EOG - the runner's /tokenize count of the visible text (add_special false, parse_special false). Any surplus VOIDS that level on that engine (fail kind hidden_reasoning), printed with its method. Re-tokenizing can over-count a legitimate output whose split was not the canonical one, so arm D calibrates it on every engine x prompt (one extra request each) and `calibration` blocks arm C when the retokenize method reads more than the exact one -- the lane never sets a slack (SPEC §10 D-5). AND every retokenize surplus in a level is REPLAYED once (client.replay_hidden: the same string at level 1 with logprobs, after the level's runs and outside every timed window) and the replay's exact count is printed beside it -- confirmed, not-confirmed or unread. The VOID stands either way (PLAN-v2 §2.4's registered rule); an unconfirmed one reads VOID(hidden_reasoning:unconfirmed), the evidence the lead needs to rule an amendment (the fold's fairness I-1). THE ALL-LAYERS VOID (G-L; M-3). The runner log up to a level's first run must hold a latest `offloaded N/M layers to GPU` line (Ollama's own regex, llm/llama_server.go L2846 at v0.34.4) with N = M. Absent is offload_unread, N < M is partial_offload, and a SECOND such line inside the level's span is a reload -- each VOIDS the level. vLLM reads n/a (it loads whole or refuses at boot: R-3). THE REST OF §2.8 AS CODE: a failed request (any fail kind: transport, an HTTP error, a truncated stream, R-7x's token_count_mismatch, Ollama's repeat abort) voids its run (law 4, `trial_void`) AND the level on that engine, as PLAN-v2 §2.8 registers it ("VOID a level on an engine if any request fails"; the bench-client fold's I-4 -- the lane's first cut voided only the run); a finish reason that is not `length` is a FLAG, never dropped, and the completion-token total rides beside every aggregate (M-5); a nonce leak is a FLAG. A VOID level keeps every run printed, but it carries no scored figure: client.unscore_cell moves its medians to `unscored_medians`, every receipt prints VOID() for it, and every judge reads it UNREAD (prereg.unread_if_void) -- the trap D-20260924-017 names (a voided 177.7 tok/s once read as a result). """ import re import seatlib as L OFFLOADED = re.compile(r"offloaded\s+(\d+)/(\d+)\s+layers to GPU") #: A special marker's piece: ``, `<|turn>`, ``, `<|channel>`, ``, `<|think|>` ... SPECIAL_PIECE = re.compile(r"^<\|?[A-Za-z_][A-Za-z0-9_]*\|?>$") #: Gemma 4's end-of-generation pieces (tokenizer_config.json at @1d2c2d7f: eos_token ``, eot_token ``). EOG_PIECES = ("", "") ALL_LAYERS_RULE = L.RULES["all_layers"] HIDDEN_RULE = L.RULES["hidden_surplus"] def is_special(piece): return bool(SPECIAL_PIECE.match(piece or "")) def eog_allowed(finish_reason): """How many trailing end-of-generation tokens a stream may carry: one on a `stop` finish, none on `length`.""" return 1 if finish_reason == "stop" else 0 #: A piece that is a partial UTF-8 byte of a byte-fallback run: it decodes alone to U+FFFD, or reads `<0xNN>`. BYTE_PIECE = re.compile(r"^(?:\ufffd+|<0x[0-9A-Fa-f]{2}>)$") def is_byte_piece(piece): return bool(BYTE_PIECE.match(piece or "")) def hidden_ids(ids, pieces, visible, finish_reason, detok=None): """The exact method: (surplus, detail). `pieces[i]` is ids[i]'s piece with specials kept; `detok(ids)` decodes a byte-fallback run jointly (None: such a run cannot be read and counts as missing).""" ids, pieces = list(ids or []), list(pieces or []) detail = {"ids": len(ids), "eog_allowed": False, "hidden": [], "byte_runs": 0} if eog_allowed(finish_reason) and pieces and pieces[-1] in EOG_PIECES: ids, pieces = ids[:-1], pieces[:-1] detail["eog_allowed"] = True if "".join(pieces) == (visible or ""): return 0, dict(detail, exact_join=True) cur, text = 0, visible or "" k = 0 while k < len(ids): i, p = ids[k], pieces[k] if is_byte_piece(p): j = k while j < len(ids) and is_byte_piece(pieces[j]): j += 1 joint = detok(ids[k:j]) if detok else None at = text.find(joint, cur) if joint and "\ufffd" not in joint else -1 if at >= 0: cur = at + len(joint) detail["byte_runs"] += 1 else: for i2, p2 in zip(ids[k:j], pieces[k:j]): detail["hidden"].append({"id": i2, "piece": p2, "why": "a byte run whose joint text is not visible"}) k = j continue k += 1 if p and not is_special(p): at = text.find(p, cur) if at < 0 and p.strip(): at = text.find(p.strip(), cur) if at >= 0: cur = at + len(p.strip()) continue elif at >= 0: cur = at + len(p) continue why = "special" if is_special(p) else ("empty" if not p else "missing from the visible text") detail["hidden"].append({"id": i, "piece": p, "why": why}) return len(detail["hidden"]), dict(detail, exact_join=False) def hidden_retokenize(eval_count, visible_count, finish_reason): """The retokenize method: (surplus, detail) = eval_count - the allowed EOG - the visible text's count (never < 0).""" if eval_count is None or visible_count is None: return None, {"why": "eval_count or the visible count is unread"} eog = eog_allowed(finish_reason) raw = int(eval_count) - eog - int(visible_count) return max(0, raw), {"eval_count": eval_count, "eog_allowed": eog, "visible_count": visible_count, "raw": raw} def hidden_tokens(rec, engine): """Set rec's hidden_surplus / hidden_method / hidden_detail from the engine's own tokens; returns the surplus.""" if rec.get("status") not in (None, "ok") and rec.get("completion_tokens") is None: rec["hidden_method"], rec["hidden_surplus"] = "n/a", None return None visible = rec.get("completion_text") or "" if rec.get("generated_ids"): surplus, detail = hidden_ids(rec["generated_ids"], engine.pieces(rec["generated_ids"]), visible, rec.get("finish_reason"), detok=engine.detokenize) method = "ids" else: surplus, detail = hidden_retokenize(rec.get("completion_tokens"), engine.visible_count(visible) if visible else 0, rec.get("finish_reason")) method = "retokenize" rec["hidden_surplus"], rec["hidden_method"], rec["hidden_detail"] = surplus, method, detail return surplus def calibrate(exact_surplus, retokenize_surplus, engine_key, prompt_id): """D's calibration line for one engine x prompt: None when the retokenize method agrees with the exact one (or reads less), else the blocking line (SPEC §10 D-5).""" if exact_surplus is None or retokenize_surplus is None: return {"engine": engine_key, "prompt": prompt_id, "verdict": "UNREAD", "exact": exact_surplus, "retokenize": retokenize_surplus, "line": None} k = retokenize_surplus - exact_surplus line = f"HIDDEN-GATE CALIBRATION: retokenize over-counts by {k} on {engine_key},{prompt_id}" if k > 0 else None return {"engine": engine_key, "prompt": prompt_id, "verdict": "blocks-C" if k > 0 else "ok", "exact": exact_surplus, "retokenize": retokenize_surplus, "over_count": max(0, k), "line": line} def all_layers(log_to_first_run, level_span=None, engine_key="ollama"): """The all-layers void: {verdict ok|void|n/a, fail_kind, line, sentence, rule}.""" if engine_key == "vllm": return {"verdict": "n/a", "fail_kind": None, "line": None, "sentence": "n/a (vLLM loads whole or refuses at boot: R-3)", "rule": ALL_LAYERS_RULE} found = list(OFFLOADED.finditer(log_to_first_run or "")) if not found: return {"verdict": "void", "fail_kind": "offload_unread", "line": None, "sentence": "VOID: the runner log does not say how many layers it offloaded", "rule": ALL_LAYERS_RULE} last = found[-1] n, m = int(last.group(1)), int(last.group(2)) rec = {"line": last.group(0), "offloaded": n, "of": m, "rule": ALL_LAYERS_RULE} if n < m: return dict(rec, verdict="void", fail_kind="partial_offload", sentence=f"VOID: the runner log reads {n}/{m} layers") if level_span and OFFLOADED.search(level_span): again = OFFLOADED.findall(level_span) return dict(rec, verdict="void", fail_kind="partial_offload" if any(int(a) < int(b) for a, b in again) else "reload", sentence="VOID: the runner log reads the model reloaded during the level", reloads=len(again)) return dict(rec, verdict="ok", fail_kind=None, sentence=f"all {m} layers offloaded") def finish_flags(streams): """(flagged count, completion-token total, the flags) over a run's ok streams (M-5: FLAG, never dropped).""" ok = [s for s in streams if s.get("status") == "ok"] flags = [{"stream": s.get("stream"), "finish_reason": s.get("finish_reason"), "completion_tokens": s.get("completion_tokens")} for s in ok if s.get("finish_reason") != "length"] return len(flags), sum(s.get("completion_tokens") or 0 for s in ok), flags def replay_summary(runs): """{verdict: count} over the level's replayed retokenize surpluses (client.replay_hidden), or {} when none.""" out = {} for r in runs: for s in r.get("streams") or []: v = (s.get("hidden_confirm") or {}).get("verdict") if v: out[v] = out.get(v, 0) + 1 return out def replay_confirmed(replay): """True when a replay confirmed a hidden token, False when every replay read one and none did, None unreplayed.""" if not replay: return None if replay.get("confirmed") or replay.get("partly"): return True return False if not replay.get("unread") else None def replay_words(replay): return ", ".join(f"{replay[k]} {k}" for k in ("confirmed", "partly", "not-confirmed", "unread") if replay.get(k)) def void_rollup(runs, layers=None, engine_key=None, prompt_id=None, level=None): """The level's voids: every failed request by its fail kind (warm-up included), every hidden-token surplus (hidden_reasoning) and the all-layers verdict; [] when clean.""" voids = [] worst = 0 methods = set() failed = {} for r in runs: for s in r.get("streams") or []: if s.get("status") not in (None, "ok"): kind = s.get("fail_kind") or "failed" failed.setdefault(kind, {"streams": 0, "runs": []}) failed[kind]["streams"] += 1 if r.get("n") not in failed[kind]["runs"]: failed[kind]["runs"].append(r.get("n")) if (s.get("hidden_surplus") or 0) > 0: worst = max(worst, s["hidden_surplus"]) methods.add(s.get("hidden_method")) warm = {r.get("n") for r in runs if r.get("warmup")} for kind in sorted(failed): f = failed[kind] runs_txt = ", ".join("warm-up" if n in warm else str(n) for n in f["runs"]) voids.append({"fail_kind": kind, "streams": f["streams"], "runs": f["runs"], "sentence": f"VOID: {engine_key} {prompt_id} L{level}: {f['streams']} failed request(s) " f"({kind}) in run(s) {runs_txt} (PLAN-v2 §2.8: any failed request voids the level)"}) if worst: replay = replay_summary(runs) voids.append({"fail_kind": "hidden_reasoning", "max_surplus": worst, "methods": sorted(methods), "replay": replay, "confirmed": replay_confirmed(replay), "sentence": f"VOID: {engine_key} {prompt_id} L{level}: {worst} hidden token(s) " f"({', '.join(sorted(m for m in methods if m))})" + (f"; the logprob replay: {replay_words(replay)}" if replay else "")}) if layers and layers.get("verdict") == "void": voids.append({"fail_kind": layers.get("fail_kind"), "sentence": layers.get("sentence"), "line": layers.get("line")}) return voids