#!/usr/bin/env python3 """framing.py -- ONE RENDERED STRING (SPEC Δ6; PLAN-v2 §2.4; CRITIQUE-method M-1). What this file is, for a reader who opened it cold. The ollama-vs-vLLM sweep never lets an engine's chat renderer build the prompt it times: it sends ONE string, the Gemma 4 user turn as Google's canonical template renders it, to each engine's raw completion endpoint (vLLM /v1/completions, Ollama /api/generate with raw, llama-server /completion). The frame -- the text before and after the user's words -- is FROZEN in prompts/frame/gemma4-canonical.json by tools/render_frame.py from the pinned chat_template.jinja (never typed: no one hand-writes a Gemma token here), and arm D proves it on the day: vLLM's own chat render of the same user text must be the composed string ID for ID (R-F1). THE BOS RULE (PREREG Amendment 3, D-20260928-048 ruling 1; prereg.RAW_BOS_ROAD registers it). The template's render begins with ``. `compose` removes it -- the composed string S is the one engine-agnostic string every row and every nonce is keyed on -- and each engine's raw road puts exactly ONE BOS back its own way: * llama-server inserts one for a string prompt when the model's add_bos_token is set (its README L491 at b11081), and Ollama's runner strips a leading `` from a raw prompt itself because llama.cpp forces add_bos on for Gemma 4 (llm/llama_server.go L246-279 at v0.34.4): both are sent S; * vLLM is NOT: the pinned checkpoint's tokenizer.json adds no special token (its post-processor is TemplateProcessing `single: [A]` with `special_tokens: {}`, and tokenizer_config.json leaves add_bos_token unset), so `add_special_tokens: true` -- the request's default, entrypoints/openai/completion/protocol.py L115 at v0.30.0 -- adds NOTHING, and arm D on the bench box read every raw string one ID short on vLLM (R-F1 refused, R-P2 0/364, D-20260928-041). vLLM's road therefore keeps the frame's literal `` (`with_frame_bos(S)`: the template's own render, byte for byte) and sends `add_special_tokens: false` -- arm S's road, whose IDs matched the checkpoint's reference 536 of 536 (the bench's PREREG §S.4). Raw parity (R-P2) checks that BOS comes first exactly once on every engine, on the IDs each road evaluates. THE TRIM RULE. The template strips a user turn's text (`message['content'] | trim`, chat_template.jinja L328 at the pinned sha); the frame file records it (`user_text_rule: trim`) and `compose` applies it, so the composed string is what the template itself would render for that user text. """ import hashlib import json import os HERE = os.path.dirname(os.path.abspath(__file__)) FRAME_FILE = os.path.join(HERE, "prompts", "frame", "gemma4-canonical.json") BOS_TEXT = "" def _load(path=FRAME_FILE): with open(path, "rb") as fh: raw = fh.read() return json.loads(raw.decode("utf-8")), hashlib.sha256(raw).hexdigest() FRAME, FRAME_SHA = _load() #: The frozen file's `rule` (rendered 2026-09-28T10:27:29Z; its sha keys every ledger cell, so the file is never #: re-written for words) ends "(each engine adds exactly one BOS itself)": true of Ollama's runner and llama-server, NOT #: of vLLM on this checkpoint. Every record of the frame carries this beside the rule. FROZEN_RULE_NOTE = ("PREREG Amendment 3 (D-20260928-048 ruling 1): the frozen rule's closing words hold for Ollama's " "runner and llama-server only; this checkpoint's tokenizer adds no special token, so vLLM's raw " "road keeps the frame's literal and sends add_special_tokens false (prereg.RAW_BOS_ROAD)") def strip_bos(text): """`text` without ONE leading `` (the BOS rule); unchanged when it carries none.""" return text[len(BOS_TEXT):] if text.startswith(BOS_TEXT) else text def with_frame_bos(text, frame=None): """`text` with the frame head's own literal `` at its head, exactly once (a leading one is not doubled): for a composed string, the template's render byte for byte -- what vLLM's raw road sends, with add_special_tokens false (the BOS rule). Refused (ValueError) for a frame whose head does not open with ``: the BOS is the template's, never one this module invents.""" f = frame or FRAME if not f["head"].startswith(BOS_TEXT): raise ValueError(f"the frame's head does not open with {BOS_TEXT}: there is no literal BOS to keep") return BOS_TEXT + strip_bos(text) def user_text(nonce_text, prompt_text, frame=None): """The user turn's text as the template renders it: the nonce at the head, then the prompt, by the frame's rule.""" f = frame or FRAME text = (nonce_text or "") + prompt_text rule = f.get("user_text_rule", "verbatim") if rule == "trim": return text.strip() if rule == "verbatim": return text raise ValueError(f"the frame file names user_text_rule {rule!r}; this module knows trim and verbatim") def compose(nonce_text, prompt_text, frame=None): """head + the user text + tail, with head's leading removed (the BOS rule): the one string every engine's raw road starts from (vLLM's puts the literal back: `with_frame_bos`).""" f = frame or FRAME return strip_bos(f["head"]) + user_text(nonce_text, prompt_text, f) + f["tail"] def head_without_bos(frame=None): """The frame's text before the user's words, BOS removed: its token count is the nonce-leak line (nonces.py).""" return strip_bos((frame or FRAME)["head"]) def record(frame=None, frame_sha=None): """What every row carries about the frame: its file's sha256, the template's sha256, the rule as frozen, and the note that amends the rule's BOS words (FROZEN_RULE_NOTE).""" f = frame or FRAME return {"file": os.path.relpath(FRAME_FILE, HERE), "sha256": frame_sha or FRAME_SHA, "template_sha256": f.get("template_sha256"), "revision": f.get("revision"), "user_text_rule": f.get("user_text_rule"), "rule": f.get("rule"), "rule_note": FROZEN_RULE_NOTE}