#!/usr/bin/env python3 """engine_ollama.py -- Ollama 0.32.13 / 0.34.4 as one engine of three (SPEC Δ6, the bench-client lane §3.8). LAUNCH. `/bin/ollama serve` with the posture's environment; is withheld-2026-10-07, unpacked by tools/ollama_fetch.sh from the pinned tarball (stack/ollama.lock.json). The system /usr/local/bin/ollama is never used (SPEC §10 D-7); readback() records the binary's sha256 either way. The serve log (stdout and stderr, the runner's lines relayed into it) is kept raw in private/. LOAD. Ollama loads on the first request, so load() sends one UNSCORED raw request with the day's first prompt and keeps its load_duration and the wall to the first answer (the start-up harvest: n = 1, "page cache warm", PLAN-v2 §2.7 M/K). Then THE RUNNER is found: the child of `ollama serve` whose program is `llama-server` (0.34.4 finds it with findLlamaCppBinary, llm/llama_server.go L343, and starts it with exec.Command, L426, in the same process group: LlamaServerSysProcAttr is empty, llm/llm_linux.go). Its argv is read from /proc AND from its own `starting llama-server cmd=` line (L436); the two must agree. Its environ is kept with whitelisted keys only. THE READ-BACK (S-1): the `server config env=map[...]` line (server/routes.go L2005 at 0.34.4, L1933 at 0.32.13), and the runner's argv: `-c -np ` (llm/llama_server.go L378-379), `-b -ub ` (L593-596: 512 on a 16-slot posture, which pins num_batch in every request -- the bench's PREREG Amendment 1; the scheduler's own pick on the default posture, servelines' THE BATCH), `--log-verbosity 4` (L578), `--flash-attn auto` (L617-624), `--no-jinja --chat-template chatml` for a Go-rendered model (L775-786); and all layers offloaded (gates.all_layers, Ollama's own regex L2846). THE WIRE. raw POST /api/generate {model: gemma4:12b-it-qat, prompt: S, raw: true, stream: true, options: {temperature: 0, seed: 0, num_predict: 256, **the posture's request options}} -- num_batch 512 on a 16-slot posture (PREREG Amendment 1: on EVERY request, the load included, so the runner never re-plans), nothing on the default posture; no num_ctx and no keep_alive (both set at the server; F-4, M-8); raw mode builds no parser (server/routes.go L489), and the runner strips a leading itself (llm/llama_server.go L246-256; tokenizerAddsBOS is true for gemma4, L260-279). The NDJSON (routes.go L693-760; the runner's rule llm/llama_server.go L1760-1850): one line per runner chunk that carries content (a token that decodes to no text sends NO line), then the done line: prompt_eval_count (= cache_n + prompt_n, L1603-1607), prompt_eval_cached_count (0.33.3 onward; ABSENT on 0.32.13), eval_count, eval_duration, load_duration, prompt_eval_duration, done_reason. first token = the first done:false line; last token = the last done:false line before done:true. TT's roads: (b)/(c) add options.num_ctx; (d) is POST /v1/completions with the same string (OpenAI SSE, no server clock: M-7). The calibration request adds logprobs: true (api/types.go L124) -- never on a scored run. chat POST /v1/chat/completions {model, messages, max_tokens: 256, temperature: 0, seed: 0, stream: true, stream_options: {include_usage: true}, reasoning_effort: "none"} -- Ollama's thinking switch (openai/openai.go L125, L539-545); its /v1 has no chat_template_kwargs (M-1). render POST /api/chat {model, messages, think: false, _debug_render_only: true, options: the posture's request options} -> _debug_info.rendered_template (the route schedules the runner BEFORE it renders, routes.go L2726 then L2800, so it carries the options too) (api/types.go L170, L541-553; server/routes.go L2800-2806), and the /v1/chat/completions twin with _debug_render_only (openai.go L128, L139) -- both kept -- then the RUNNER's POST /tokenize {content: strip_bos(rendered), add_special: true, parse_special: true} (llama.cpp server-context.cpp L5084-5122). raw IDs the runner's /tokenize on strip_bos(S): what /completion evaluates for a string prompt (README L491). visible the runner's /tokenize with add_special false and parse_special false (the retokenize gate). repeat abort a line or error reading `prediction aborted, token repeat limit reached` (llm/llama_server.go L1806-1808): fail_kind repeat_abort (N-10), never transport. """ import json import os import re import time import engine as E import framing import gates import seat as S import seatlib as L import servelines as SL TAG = SL.OLLAMA_TAG REPEAT_ABORT = "prediction aborted, token repeat limit reached" #: Ollama 0.32.13 aborts after MORE THAN 30 identical (trimmed) chunks and returns ctx.Err() -- nil -- so the stream #: just ends with no done line and no error (llm/llama_server.go L1686-1696 at v0.32.13; 0.34.4 waits for 100 and #: says so, L1798-1808). The chunk that trips it is never sent: the client sees 31 identical lines, 30 repeats. SILENT_REPEAT_LIMIT = {"0.32.13": 30} _STARTING = re.compile(r'msg="starting llama-server" cmd="?([^"\n]+)"?') _CTX_LINE = re.compile(r"llama_context:\s+(\w+)\s+=\s+(\S+)") _ISWA = re.compile(r"llama_kv_cache_iswa: creating\s+(non-SWA|SWA) KV cache, size = (\d+) cells") _BUFFER = re.compile(r"(\w+)\s+(model|KV|compute|output|RS) buffer size\s*=\s*([\d.]+) MiB") _GRAPHS = re.compile(r"CUDA graph[^\n]*", re.I) def sha256_path(path): try: return L.sha256_file(path) except OSError: return None class OllamaEngine(E.Engine): key = "ollama" replays_hidden = True frame_rule = ("/api/generate raw NDJSON: the first done:false line is the first token; the last done:false line " "before done:true is the last; counts from the done:true line") def __init__(self, posture_key, ctx, run_dir, log=print): super().__init__(posture_key, ctx, run_dir, log) self.runner = None # {pid, argv, env, port, argv_log} self.version = None self.load_rec = None def ready(self, timeout_s): def probe(): st, body = self.call("GET", "/api/version", timeout=2.0) if st == 200 and isinstance(body, dict): self.version = body.get("version") return True return False return self.wait_ready(timeout_s, probe) # ---- the runner ------------------------------------------------------------------------- # def find_runner(self, timeout_s=30.0): """The llama-server child of `ollama serve` (argv, environ, port), read from /proc; None when absent.""" deadline = time.monotonic() + timeout_s while time.monotonic() < deadline: for pid in S.descendants(self.proc.pid): argv = E.proc_argv(pid) if E.engine_of_argv(argv) == "llama-server": prog = E.program_argv(argv) port = SL.argv_value(prog, "withheld-2026-10-07") m = None for m in _STARTING.finditer(self.log_since(0)): pass self.runner = {"pid": pid, "argv": prog, "env": E.proc_environ(pid) or {}, "port": int(port) if port else None, "argv_log": m.group(1).split() if m else None} self.proc.known_pids.add(pid) return self.runner time.sleep(0.1) return None def runner_port(self): """The runner's loopback port; re-read from /proc when the runner the adapter knew has gone (a re-plan).""" if self.runner and self.runner.get("pid") and not S._alive(self.runner["pid"]): self.runner = None if not self.runner or not self.runner.get("port"): if not self.find_runner(): raise L.Refused("R-3", f"{self.posture['key']}: no llama-server runner is a child of ollama serve") return self.runner["port"] def load(self, first_prompt_text, nonce_text): """One UNSCORED raw request that loads the model: its load_duration and wall to the first answer kept (n = 1).""" body = self.raw_body(framing.compose(nonce_text, first_prompt_text), max_tokens=8) t0 = time.monotonic() st, text = self.call("POST", "/api/generate", body, timeout=600.0) wall = time.monotonic() - t0 done = None if isinstance(text, str): for line in text.splitlines(): try: obj = json.loads(line) except ValueError: continue if obj.get("done"): done = obj elif isinstance(text, dict): done = text self.find_runner() self.load_rec = {"http_status": st, "wall_to_first_answer_s": round(wall, 3), "load_duration_ms": None if not done else round(done.get("load_duration", 0) / 1e6, 3), "done_reason": None if not done else done.get("done_reason"), "n": 1, "label": "page cache warm (PLAN-v2 §2.7 M/K): start to first answer, n = 1"} if st != 200 or done is None: raise L.Refused("R-3", f"{self.posture['key']}: the loading request answered HTTP {st}: {str(text)[:200]}", {"load": self.load_rec}) return self.load_rec # ---- the read-back ----------------------------------------------------------------------- # def readback(self): text = self.log_since(0) sc = SL.parse_server_config(text) r = self.runner or self.find_runner() or {} argv = r.get("argv") agree = None if argv is not None and r.get("argv_log") is not None: agree = [os.path.basename(argv[0])] + argv[1:] == [os.path.basename(r["argv_log"][0])] + r["argv_log"][1:] layers = gates.all_layers(text) internals = {"llama_context": dict(_CTX_LINE.findall(text)) or "UNREAD", "swa_cache": [{"cache": a, "cells": int(b)} for a, b in _ISWA.findall(text)] or "UNREAD", "buffers": [{"device": a, "kind": b, "mib": float(c)} for a, b, c in _BUFFER.findall(text)] or "UNREAD", "cuda_graphs": [m.group(0)[:200] for m in _GRAPHS.finditer(text)][:5] or "UNREAD", "offloaded": layers.get("line") or "UNREAD", "projector_in_argv": SL.argv_value(argv, "--mmproj") is not None if argv else None, "flash_attn_argv": SL.argv_value(argv, "--flash-attn") if argv else None} props = None if r.get("port"): st, props = self.call("GET", "/props", port=r["port"], timeout=5.0) props = props if st == 200 else None rb = {"engine_version": self.version, "uuid": self.ctx.get("uuid"), "server_config": sc, "runner_argv": argv, "runner_argv_from_log": r.get("argv_log"), "runner_argv_agree": agree, "runner_env": SL.whitelist_env(r.get("env")), "runner_port": r.get("port"), "all_layers": layers["verdict"] == "ok", "all_layers_detail": layers, "internals": internals, "binary_sha256": sha256_path(self.ctx["binary"]), "runner_binary_sha256": sha256_path(argv[0]) if argv else None, "runner_props": props, "load": self.load_rec, "env_serving": self.env_serving(), "inherited_env_dropped": self.env_dropped, "unread": [k for k, v in (("server_config", sc), ("runner_argv", argv)) if v is None]} if rb["env_serving"] is None: rb["unread"].append("ollama serve's /proc environ") if agree is False: rb["unread"].append("the runner's /proc argv and its `starting llama-server cmd=` line disagree") return rb def sampler(self): """N-9: the runner's /slots params after a D request, beside the request's explicit temperature 0.""" st, slots = self.call("GET", "/slots", port=self.runner_port(), timeout=5.0) return {"engine": self.key, "slots": slots if st == 200 else None, "request": {"temperature": 0, "seed": 0}, "tag_params": "the tag's params blob {temperature 1, top_k 64, top_p 0.95} (MODEL-OLLAMA-12B-QAT.lock.json)", "rule": "the runner's own /slots params after a request; temperature 0 reduces every sampler to argmax"} # ---- the wire ------------------------------------------------------------------------------ # def raw_path(self): return "/api/generate" def chat_path(self): return "/v1/chat/completions" def raw_body(self, rendered, *, max_tokens, num_ctx=None, logprobs=False): opts = {"temperature": 0, "seed": 0, "num_predict": int(max_tokens), **SL.request_options(self.posture)} if num_ctx is not None: opts["num_ctx"] = int(num_ctx) # TT roads (b) and (c) only; the sweep's shape carries none body = {"model": TAG, "prompt": rendered, "raw": True, "stream": True, "options": opts} if logprobs: body["logprobs"] = True # the D calibration only: never a scored run (N-12) return body def replay_exact_surplus(self, rec): """The replay's exact count (client.replay_hidden): logprob_surplus with the EOG a `stop` finish allows.""" return logprob_surplus(rec, gates.eog_allowed(rec.get("finish_reason"))) def v1_completions_body(self, rendered, *, max_tokens): """TT road (d): the same string on Ollama's /v1/completions (no server clock, M-7). Ollama's /v1 builds its own options map (openai/openai.go L813-864 at v0.34.4) and carries no num_batch; the runner the posture loaded with an explicit 512 is not re-planned by it (servelines' THE BATCH).""" return {"model": TAG, "prompt": rendered, "max_tokens": int(max_tokens), "temperature": 0, "seed": 0, "stream": True, "stream_options": {"include_usage": True}} def chat_row_body(self, user_text, *, max_tokens): return {"model": TAG, "messages": [{"role": "user", "content": user_text}], "max_tokens": int(max_tokens), "temperature": 0, "seed": 0, "stream": True, "stream_options": {"include_usage": True}, "reasoning_effort": "none"} def decode_stream(self, resp, t0, rec): if rec.get("path") == "chat": return E.decode_chat_sse(resp, t0, rec) if rec.get("path") == "v1-completions": return decode_completions_sse(resp, t0, rec) text, logprobs = [], [] t_last = None last_piece, repeats = None, 0 for t, line in E.ndjson_events(resp): rec["frames"] += 1 try: obj = json.loads(line) except ValueError: rec["bad_frames"] += 1 continue if obj.get("error"): rec["error"] = str(obj["error"])[:300] if REPEAT_ABORT in rec["error"]: rec["repeat_abort"] = True continue if not obj.get("done"): rec["token_frames"] += 1 if rec["t_first_token_s"] is None: rec["t_first_token_s"] = t - t0 t_last = t - t0 piece = obj.get("response") if "response" in obj else (obj.get("message") or {}).get("content", "") text.append(piece or "") trimmed = (piece or "").strip() repeats = repeats + 1 if trimmed == last_piece else 0 last_piece = trimmed if obj.get("logprobs"): logprobs.extend(obj["logprobs"]) continue rec["done_line"] = {k: obj.get(k) for k in ("done_reason", "total_duration", "load_duration", "prompt_eval_count", "prompt_eval_cached_count", "prompt_eval_duration", "eval_count", "eval_duration")} rec["finish_reason"] = obj.get("done_reason") rec["t_finish_s"] = t_last rec["prompt_tokens"] = obj.get("prompt_eval_count") rec["completion_tokens"] = obj.get("eval_count") if "prompt_eval_cached_count" in obj: rec["cached_tokens"] = obj["prompt_eval_cached_count"] else: rec["cached_why"] = (f"Ollama {self.posture['version']} reports no prompt_eval_cached_count (it arrived " f"in 0.33.3)") rec["load_duration_ms"] = round((obj.get("load_duration") or 0) / 1e6, 3) rec["prompt_eval_duration_ms"] = round((obj.get("prompt_eval_duration") or 0) / 1e6, 3) rec["completion_text"] = "".join(text) if logprobs: rec["logprobs"] = logprobs rec["t_end_s"] = time.monotonic() - t0 rec["trailing_repeats"] = repeats limit = SILENT_REPEAT_LIMIT.get(self.posture["version"]) if rec.get("finish_reason") is None and not rec.get("error") and limit is not None and repeats >= limit: rec["repeat_abort"] = True rec["error"] = (f"inferred: Ollama {self.posture['version']} ends a stream silently after more than {limit} " f"identical chunks (llm/llama_server.go L1686-1696); this one ended with no done line on " f"{repeats + 1} identical trailing lines") if rec.get("finish_reason") is None and not rec.get("error"): rec["error"] = "the stream ended without a done line" return rec # ---- tokens (the runner's) ---------------------------------------------------------------- # def _tok(self, content, add_special, parse_special=True): st, obj = self.call("POST", "/tokenize", {"content": content, "add_special": bool(add_special), "parse_special": bool(parse_special)}, port=self.runner_port()) if st != 200 or not isinstance(obj, dict) or "tokens" not in obj: raise L.Refused("R-P2", f"Ollama's runner POST /tokenize answered {st}: {str(obj)[:200]}") return [t["id"] if isinstance(t, dict) else t for t in obj["tokens"]] def tokenize_raw(self, text, *, add_special): return self._tok(framing.strip_bos(text), add_special) def render_chat_ids(self, user_text): msgs = [{"role": "user", "content": user_text}] out = {} chat = {"model": TAG, "messages": msgs, "think": False, "_debug_render_only": True} if SL.request_options(self.posture): chat["options"] = SL.request_options(self.posture) st, obj = self.call("POST", "/api/chat", chat) r1 = ((obj or {}).get("_debug_info") or {}).get("rendered_template") if st == 200 and isinstance(obj, dict) else None st2, obj2 = self.call("POST", "/v1/chat/completions", {"model": TAG, "messages": msgs, "reasoning_effort": "none", "_debug_render_only": True}) r2 = ((obj2 or {}).get("_debug_info") or {}).get("rendered_template") if st2 == 200 and isinstance(obj2, dict) else None for name, r in (("ollama /api/chat think:false", r1), ("ollama /v1/chat/completions reasoning_effort:none", r2)): out[name] = ({"ids": self._tok(framing.strip_bos(r), True), "rendered_tail": r[-80:]} if r is not None else {"ids": None, "why": "the debug render answered nothing"}) return {"renders": out, "renders_identical": (r1 == r2) if r1 is not None and r2 is not None else None, "how": "Ollama _debug_render_only, then its runner's /tokenize (add_special, parse_special)"} def visible_count(self, text): return len(self._tok(text, False, parse_special=False)) def detokenize(self, ids): st, obj = self.call("POST", "/detokenize", {"tokens": list(ids)}, port=self.runner_port()) if st != 200 or not isinstance(obj, dict): raise L.Refused("hidden", f"Ollama's runner POST /detokenize answered {st}: {str(obj)[:200]}") return obj.get("content", "") def server_clock(self, rec): d = rec.get("done_line") or {} n, dur = d.get("eval_count"), d.get("eval_duration") rate = round(n / (dur / 1e9), 3) if n and dur else None why = None if rate is None: why = ("this road returns no server clock (Ollama /v1: M-7)" if rec.get("path") != "raw" else "the done line carried no eval_duration") return {"decode_tok_s_server": rate, "load_duration_ms": rec.get("load_duration_ms"), "prompt_eval_duration_ms": rec.get("prompt_eval_duration_ms"), "eval_duration_ms": round(dur / 1e6, 3) if dur else None, "why": why, "rule": L.RULES["server_clock_ollama"]} def decode_completions_sse(resp, t0, rec): """An OpenAI /v1/completions stream with no token IDs (Ollama's TT road (d)): each choice frame is a token frame; the frame with the finish_reason is the last; counts from the usage chunk.""" text = [] for t, payload in E.sse_events(resp): rec["frames"] += 1 try: obj = json.loads(payload) except ValueError: rec["bad_frames"] += 1 continue if obj.get("error"): rec["error"] = str(obj["error"])[:300] continue if obj.get("usage"): rec["usage"] = obj["usage"] rec["prompt_tokens"] = obj["usage"].get("prompt_tokens") rec["completion_tokens"] = obj["usage"].get("completion_tokens") for ch in (obj.get("choices") or [])[:1]: if ch.get("finish_reason"): rec["finish_reason"] = ch["finish_reason"] rec["t_finish_s"] = rec.get("_t_last") continue rec["token_frames"] += 1 if rec["t_first_token_s"] is None: rec["t_first_token_s"] = t - t0 rec["_t_last"] = t - t0 text.append(ch.get("text") or "") rec.pop("_t_last", None) rec["completion_text"] = "".join(text) rec["t_end_s"] = time.monotonic() - t0 if rec.get("finish_reason") is None and not rec.get("error"): rec["error"] = "the stream ended without a finish_reason" return rec def logprob_surplus(rec, eog_allowed): """The calibration's EXACT count on Ollama (logprobs: per-token text and bytes, api/types.go L497-515): the generated tokens whose text is empty. Ollama sends a line only for content, so eval_count less the lines' tokens is the count of tokens that produced no line -- less the one EOG a `stop` finish may end on.""" n = rec.get("completion_tokens") toks = rec.get("logprobs") if n is None or toks is None: return None empty_in_lines = sum(1 for t in toks if not (t.get("token") or "")) no_line = max(0, n - len(toks)) return max(0, empty_in_lines + no_line - (1 if eog_allowed else 0))