#!/usr/bin/env python3 """engine_llamaserver.py -- llama-server from Ollama 0.34.4's bundle, alone (SPEC Δ6, the bench-client lane §3.9). Release 1 uses it in arm D and TT's road (e) only; as a scored arm it is the follow-up's. LAUNCH (M-11). arms.py reads the running Ollama 0.34.4 runner's argv and environ from /proc, stops Ollama (one engine resident), and launches `/lib/ollama/llama-server` with that argv, `withheld-2026-10-07` replaced and `--device CUDA0` added, and the copied environ plus GGML_VK_VISIBLE_DEVICES=-1 (servelines.argv_env). * THE DEVICE GATE (R-DEV): its log's device list -- one `llama_model_load_from_file_impl: using device () () - MiB free` line per device the model uses (src/llama.cpp L306 at b11081) -- names exactly one CUDA device and no Vulkan device, or the boot is REFUSED and stopped. * All layers must be offloaded (gates.all_layers). * NOTE, read from Ollama 0.34.4's source: for a Go-rendered model Ollama passes `--no-jinja --chat-template chatml` (llm/llama_server.go L775-786), so this llama-server's /apply-template and /v1 chat render ChatML, not Gemma: a finding R-P1 prints (the parity path's raw /completion is unaffected). THE WIRE (b11081 tools/server/README.md and its source). raw POST /completion {prompt: S, n_predict: 256, temperature: 0, seed: 0, stream: true, cache_prompt: true, return_tokens: true} -- cache_prompt defaults true (README L587) and Ollama sends true (llama_server.go L1409, L1657), so it is passed explicitly and the two match by construction. SSE: one chunk per token with `tokens` (README L660) and `stop: false`; the final chunk `stop: true` carries timings{prompt_n, cache_n, predicted_n, prompt_ms, predicted_ms}, tokens_cached, stop_type, generation_settings (README L660-675). first token = the first chunk with a token; last token = the last stop:false chunk. chat POST /v1/chat/completions with chat_template_kwargs.enable_thinking false. render POST /apply-template {messages, chat_template_kwargs: {enable_thinking: false}} -- b11081 honours the kwargs there (server-context.cpp L5060-5070 parses the body with oaicompat_chat_params_parse, server-common.cpp L1329-1344) -- then POST /tokenize. detokenize POST /detokenize {tokens} -> {content}: tokens_to_str with common_token_to_piece(..., special = true) (server-context.cpp L5125-5136, server-common.cpp L1582-1598, common/common.h L1062-1070): specials kept. cache defaults (A-13, D): `llama-server --help` (each flag's default), GET /props, and the startup lines. """ import json import re import subprocess import time import engine as E import framing import gates import pen import seatlib as L import servelines as SL _DEVICE = re.compile(r"using device (\S+) \(") _HELP_DEFAULT = re.compile(r"^\s*(-[\w-]+(?:,\s*--?[\w-]+)*)\s.*?\(default: ([^)]*)\)", re.M) CACHE_FLAGS = ("--cache-ram", "--ctx-checkpoints", "--swa-full", "--slot-prompt-similarity", "--cache-reuse", "--slot-save-path", "--kv-unified", "--parallel", "--gpu-layers", "--fit") def device_list(log_text): """{cuda, vulkan, names, rule} from the latest model load's `using device` lines (None counts when unread).""" names = _DEVICE.findall(log_text or "") if not names: return {"cuda": None, "vulkan": None, "names": [], "rule": "no `using device` line was printed: UNREAD"} distinct = list(dict.fromkeys(names)) return {"cuda": sum(1 for n in distinct if n.upper().startswith("CUDA")), "vulkan": sum(1 for n in distinct if n.lower().startswith("vulkan")), "names": distinct, "rule": "src/llama.cpp L306 at b11081: one `using device` line per device the model is offloaded to"} def help_defaults(text): """{flag spelling: its printed default} for the cache flags (A-13), parsed from `--help`.""" out = {} for m in _HELP_DEFAULT.finditer(text or ""): flags = [f.strip() for f in m.group(1).split(",")] for want in CACHE_FLAGS: if want in flags: out[want] = m.group(2).strip() return out class LlamaServerEngine(E.Engine): key = "llama-server" frame_rule = ("/completion SSE: the first chunk with a token is the first token; the last stop:false chunk is the " "last; counts from the stop:true chunk's timings") def ready(self, timeout_s): def probe(): st, _ = self.call("GET", "/health", timeout=2.0) return st == 200 t = self.wait_ready(timeout_s, probe) dev = device_list(self.log_since(0)) if dev["cuda"] != 1 or dev["vulkan"] != 0: raise L.Refused("R-DEV", f"llama-server lists {', '.join(dev['names']) or 'no device (unread)'}; one CUDA " f"device and no Vulkan device is the rule (M-11)", {"devices": dev}) return t def help_text(self): try: r = subprocess.run([self.argv[0], "--help"], capture_output=True, text=True, timeout=30, env=self.env) return r.stdout + r.stderr except (OSError, subprocess.TimeoutExpired) as e: return f"UNREAD: {type(e).__name__}: {e}" def readback(self): text = self.log_since(0) dev = device_list(text) layers = gates.all_layers(text) helptext = self.help_text() pen.write_private_text(self.log_path.replace(".serve.log", ".help.txt"), helptext) st, props = self.call("GET", "/props", timeout=5.0) return {"engine_version": "b11081 (Ollama 0.34.4's bundle)", "devices": dev, "all_layers": layers["verdict"] == "ok", "all_layers_detail": layers, "argv": self.argv, "env": SL.whitelist_env(self.env), "cache_defaults": {"help": help_defaults(helptext), "props": props if st == 200 else None, "startup_lines": [ln for ln in text.splitlines() if "cache" in ln.lower() or "checkpoint" in ln.lower()][:20], "rule": "A-13: each flag's default from --help, the server's own /props, and its " "startup lines on prompt cache and checkpoints"}, "chat_template_note": ("Ollama's runner argv carries `--no-jinja --chat-template chatml` for a Go-rendered " "model (llm/llama_server.go L775-786): this server's chat renders ChatML" if "--no-jinja" in self.argv else None), "unread": [] if dev["cuda"] is not None else ["devices"]} def sampler(self): return {"engine": self.key, "rule": "generation_settings from the final /completion chunk (per stream)"} def raw_path(self): return "/completion" def chat_path(self): return "/v1/chat/completions" def raw_body(self, rendered, *, max_tokens, **road): if road: raise ValueError(f"llama-server's raw road takes no options (asked {sorted(road)})") return {"prompt": rendered, "n_predict": int(max_tokens), "temperature": 0, "seed": 0, "stream": True, "cache_prompt": True, "return_tokens": True} def chat_row_body(self, user_text, *, max_tokens): return {"messages": [{"role": "user", "content": user_text}], "max_tokens": int(max_tokens), "temperature": 0, "seed": 0, "stream": True, "stream_options": {"include_usage": True}, "chat_template_kwargs": {"enable_thinking": False}} def decode_stream(self, resp, t0, rec): if rec.get("path") == "chat": return E.decode_chat_sse(resp, t0, rec) text, ids = [], [] t_last = None for t, payload in E.sse_events(resp): rec["frames"] += 1 try: obj = json.loads(payload) except ValueError: rec["bad_frames"] += 1 continue if obj.get("error"): rec["error"] = str(obj["error"])[:300] continue if not obj.get("stop"): if obj.get("tokens") or obj.get("content"): rec["token_frames"] += 1 if rec["t_first_token_s"] is None: rec["t_first_token_s"] = t - t0 t_last = t - t0 ids.extend(obj.get("tokens") or []) text.append(obj.get("content") or "") continue tm = obj.get("timings") or {} rec["timings"] = tm rec["stop_type"] = obj.get("stop_type") rec["finish_reason"] = {"limit": "length", "eos": "stop", "word": "stop"}.get(obj.get("stop_type"), obj.get("stop_type")) rec["t_finish_s"] = t_last rec["completion_tokens"] = tm.get("predicted_n") if tm.get("prompt_n") is not None: rec["prompt_tokens"] = tm.get("prompt_n", 0) + (tm.get("cache_n") or 0) rec["cached_tokens"] = tm.get("cache_n") rec["generation_settings"] = obj.get("generation_settings") rec["tokens_cached"] = obj.get("tokens_cached") rec["completion_text"] = "".join(text) rec["generated_ids"] = ids if ids else None rec["t_end_s"] = time.monotonic() - t0 if rec.get("finish_reason") is None and not rec.get("error"): rec["error"] = "the stream ended without a stop chunk" return rec def _tok(self, content, add_special, parse_special=True): st, obj = self.call("POST", "/tokenize", {"content": content, "add_special": bool(add_special), "parse_special": bool(parse_special)}) if st != 200 or not isinstance(obj, dict) or "tokens" not in obj: raise L.Refused("R-P2", f"llama-server POST /tokenize answered {st}: {str(obj)[:200]}") return [t["id"] if isinstance(t, dict) else t for t in obj["tokens"]] def tokenize_raw(self, text, *, add_special): return self._tok(text, add_special) def render_chat_ids(self, user_text): st, obj = self.call("POST", "/apply-template", {"messages": [{"role": "user", "content": user_text}], "chat_template_kwargs": {"enable_thinking": False}}) if st != 200 or not isinstance(obj, dict): return {"renders": {"llama-server /apply-template": {"ids": None, "why": f"HTTP {st}"}}} prompt = obj.get("prompt", "") return {"renders": {"llama-server /apply-template": {"ids": self._tok(framing.strip_bos(prompt), True), "rendered_tail": prompt[-80:]}}, "how": "llama-server /apply-template with chat_template_kwargs.enable_thinking false, then /tokenize"} def visible_count(self, text): return len(self._tok(text, False, parse_special=False)) def detokenize(self, ids): st, obj = self.call("POST", "/detokenize", {"tokens": list(ids)}) if st != 200 or not isinstance(obj, dict): raise L.Refused("hidden", f"llama-server POST /detokenize answered {st}: {str(obj)[:200]}") return obj.get("content", "") def server_clock(self, rec): tm = rec.get("timings") or {} n, ms = tm.get("predicted_n"), tm.get("predicted_ms") rate = round(n / (ms / 1000.0), 3) if n and ms else None return {"decode_tok_s_server": rate, "prompt_ms": tm.get("prompt_ms"), "predicted_ms": ms, "why": None if rate is not None else "the final chunk carried no timings", "rule": L.RULES["server_clock_llama-server"]}