{
  "mode": "doorman",
  "model_tag": "mistral-small3.2:24b",
  "box_class": "gpu-5090-laptop-24g",
  "base_url": "http://127.0.0.1:11470",
  "num_ctx": 65536,
  "cards": [
    "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff"
  ],
  "card_labels": {
    "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": "soldered (mobile; no socket \u2014 link width is traced under load, never assumed) at 00000000:01:00.0 - vendor unread - GPU-edff232c"
  },
  "num_predict": 8,
  "prompt_source": "a neutral invented passage carried in bench_doorman.py; no corpus, product, visitor or document text",
  "prompt_chars": 716,
  "prompt_sha256": "0050ecdd69acc7c8929f939e16b2cf446641af6c278aad0f670384c265e92368",
  "prompt_sha256_note": "the digest of the in-file neutral passage this arm sends -- NOT the series' frozen prompt, which is a different measurement",
  "card_identity": {
    "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
      "index": null,
      "bus_id": "00000000:01:00.0",
      "pci_sub_device_id": null,
      "vendor": null,
      "board_per_operator": null,
      "vbios": null,
      "serial": null,
      "pcie_link_width_max": null,
      "pcie_link_width_negotiated": null,
      "memory_total_mib": null,
      "memory_bus_width_bits": null,
      "clocks_max_sm_mhz": null,
      "clocks_max_memory_mhz": null,
      "power_default_limit_w": null,
      "power_min_limit_w": null,
      "power_max_limit_w": null,
      "power_limit_at_bench_start_w": null,
      "note": "the single NVIDIA GeForce RTX 5090 Laptop GPU 24 GB SOLDERED to this machine's board at 00000000:01:00.0. There is no seat, no partner and no card to swap. \u26a0 THE TRAP ON THIS BOX IS NOT A STALE BOOT UNIT, IT IS A RUNNING DAEMON: nvidia-powerd.service (NVIDIA Dynamic Boost) floats this board's power limit between its default and its maximum against the CPU's draw, continuously and without asking, and it has been running since 2026-09-08. The lesson every leg of this ladder carries \u2014 a cap read once at the start of a night is not a cap \u2014 is true here in its strongest form: such a reading is a sample of a moving signal. Every arm reads the limit back in the same call as its own data, every stage reads it again at its close, and a stage that finds it moved STOPS rather than relabelling its file. The second trap is a CLOCK LOCK: ai-perf.service runs `nvidia-smi -lgc 1200,2550` at boot on this box and the cards that produced the rows this bench compares against ran unlocked, so a floored clock changes what a power cap means and every record carries clocks.sm min and max so the confound is visible rather than implied."
    }
  },
  "instance_env": {
    "kind": "instance-env-receipt",
    "box_class": "gpu-5090-laptop-24g",
    "target_uuid": "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff",
    "written_utc": "2026-09-21T19:12:18Z",
    "unit": "bench-5090laptop-one",
    "shape": "one",
    "port": 11470,
    "base_url": "http://127.0.0.1:11470",
    "cuda_visible_devices": "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff",
    "OLLAMA_FLASH_ATTENTION": "1",
    "OLLAMA_KV_CACHE_TYPE": "q8_0",
    "OLLAMA_CONTEXT_LENGTH": 32768,
    "OLLAMA_<parallel-requests>": 4,
    "OLLAMA_MAX_LOADED_MODELS": 1,
    "OLLAMA_KEEP_ALIVE": "0 (the server default; every bench request sends its own keep_alive, which wins, and every arm unloads on the way out)",
    "models_dir": "/usr/share/ollama/.ollama/models",
    "models_dir_writable_by_this_user": false,
    "api_version": {
      "version": "0.32.13"
    },
    "ollama_binary": "/workshop/bench-laptop-5090-2026-09-21/ollama-0.32.13/bin/ollama",
    "ollama_client_version": "0.32.13",
    "ollama_version_pin": "0.32.13",
    "disclosed_tenants": "nomic-embed-text:latest on the SYSTEM ollama (:<the runtime's default port>, OLLAMA_KEEP_ALIVE=-1, ~323 MB VRAM) \u2014 disclosed, never unloaded",
    "power_and_clocks_at_start_csv": "[N/A], 95.00 W, 95.00 W, 175.00 W, 1590 MHz, 9001 MHz, 3090 MHz",
    "power_and_clocks_at_start_fields": "power.limit,enforced.power.limit,power.default_limit,power.max_limit,clocks.sm,clocks.mem,clocks.max.sm",
    "nvidia_powerd_at_start": "inactive",
    "clock_lock_unit_at_start": "active",
    "clock_lock_unit_name": "ai-perf.service",
    "clock_lock_range_declared": "1200,2550",
    "clock_lock_unit_state_is_not_evidence": "ai-perf.service is Type=oneshot and reads 'active (exited)' for the whole uptime whatever happens to the card afterwards. The lock's real state is read from the CARD (clocks.sm against the lock's floor) and is recorded in this leg's clock-lock receipt.",
    "power_limits_at_start_w": "0, [N/A], 95.00 W;",
    "persistence_mode_at_start": "0, Enabled;",
    "pcie_link_width_at_start": "0, 8;"
  },
  "comparability_note": "These latencies are NOT comparable to the generation arms' TTFT: a different prompt, a different length and eight tokens out instead of 256. They are comparable to each other across caps and across concurrency, which is what this arm is for.",
  "memory_temp_support": {
    "checked_utc": "2026-09-21T19:15:15Z",
    "paths": {
      "nvidia-smi --query-gpu=temperature.memory": {
        "raw": "0, GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff, N/A",
        "readable": false
      },
      "nvidia-smi -q -d TEMPERATURE": {
        "lines": [
          "GPU Target Temperature                         : 87 C",
          "Memory Current Temp                            : N/A",
          "Memory Max Operating T.Limit Temp              : N/A"
        ],
        "readable": false
      },
      "NVML NVML_FI_DEV_MEMORY_TEMP": {
        "available": true,
        "field_id": 82,
        "cards": {
          "0": {
            "call_rc": 0,
            "field_rc": 3,
            "supported": false,
            "not_supported": true,
            "value_c": null
          }
        }
      }
    },
    "readable": false,
    "verdict": "the memory die's temperature is NOT readable on these cards through any of the three paths asked, so the memory-temperature stop condition could not arm and NO memory temperature is reported anywhere in this bench. The core temperature stop and the driver's own thermal-slowdown reasons are the thermal instrument instead."
  },
  "ups_before_arm": {
    "ups": "pr1500@localhost",
    "read_utc": "2026-09-21T19:15:16Z",
    "available": false,
    "error": "no upsc on this box"
  },
  "started_utc": "2026-09-21T19:15:16Z",
  "levels": [
    {
      "concurrency": 1,
      "calls_requested": 10,
      "contention_gate": {
        "attempts": [
          {
            "window_s": 10.0,
            "samples": 20,
            "cards": {
              "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
                "utilization_pct": {
                  "min": 0.0,
                  "median": 0.0,
                  "max": 55.0,
                  "mean": 8.2,
                  "n": 20
                },
                "power_w": {
                  "min": 15.91,
                  "median": 15.97,
                  "max": 69.21,
                  "mean": 21.12,
                  "n": 20
                },
                "card_seen": true
              }
            },
            "max_mean_util_pct": 8.2,
            "all_cards_seen": true,
            "attempt": 1,
            "verdict": "busy: busiest card mean 8.2% > 5.0% -- waiting"
          },
          {
            "window_s": 10.0,
            "samples": 20,
            "cards": {
              "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
                "utilization_pct": {
                  "min": 0.0,
                  "median": 0.0,
                  "max": 0.0,
                  "mean": 0.0,
                  "n": 20
                },
                "power_w": {
                  "min": 6.65,
                  "median": 15.92,
                  "max": 15.99,
                  "mean": 14.1,
                  "n": 20
                },
                "card_seen": true
              }
            },
            "max_mean_util_pct": 0.0,
            "all_cards_seen": true,
            "attempt": 2,
            "verdict": "quiet: busiest card mean 0.0% <= 5.0%"
          }
        ],
        "passed": true,
        "cards": [
          "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff"
        ]
      },
      "wall_s": 4.6248,
      "calls": [
        {
          "ttft_ms": 343.11,
          "first_token_was_thinking": false,
          "wall_s": 0.4935,
          "eval_count": 2,
          "eval_duration_ns": 149325000,
          "decode_tok_s": 13.394,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 165152000,
          "prefill_tok_s": 4020.539,
          "load_duration_ns": 174415706,
          "total_duration_ns": 491651118,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.652,
          "decode_tok_s_wall_gross": 4.053,
          "num_ctx_option": 65536,
          "wall_latency_ms": 493.52
        },
        {
          "ttft_ms": 316.8,
          "first_token_was_thinking": false,
          "wall_s": 0.4627,
          "eval_count": 2,
          "eval_duration_ns": 144855000,
          "decode_tok_s": 13.807,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 147956000,
          "prefill_tok_s": 4487.821,
          "load_duration_ns": 165086766,
          "total_duration_ns": 460931192,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.852,
          "decode_tok_s_wall_gross": 4.322,
          "num_ctx_option": 65536,
          "wall_latency_ms": 462.81
        },
        {
          "ttft_ms": 331.04,
          "first_token_was_thinking": false,
          "wall_s": 0.4777,
          "eval_count": 2,
          "eval_duration_ns": 145300000,
          "decode_tok_s": 13.765,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 146667000,
          "prefill_tok_s": 4527.262,
          "load_duration_ns": 181555253,
          "total_duration_ns": 475953720,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.819,
          "decode_tok_s_wall_gross": 4.187,
          "num_ctx_option": 65536,
          "wall_latency_ms": 477.78
        },
        {
          "ttft_ms": 305.76,
          "first_token_was_thinking": false,
          "wall_s": 0.4516,
          "eval_count": 2,
          "eval_duration_ns": 144618000,
          "decode_tok_s": 13.83,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 148461000,
          "prefill_tok_s": 4472.555,
          "load_duration_ns": 154260259,
          "total_duration_ns": 450079141,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.858,
          "decode_tok_s_wall_gross": 4.429,
          "num_ctx_option": 65536,
          "wall_latency_ms": 451.65
        },
        {
          "ttft_ms": 320.51,
          "first_token_was_thinking": false,
          "wall_s": 0.4663,
          "eval_count": 2,
          "eval_duration_ns": 144974000,
          "decode_tok_s": 13.796,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 146540000,
          "prefill_tok_s": 4531.186,
          "load_duration_ns": 170337121,
          "total_duration_ns": 464679742,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.859,
          "decode_tok_s_wall_gross": 4.289,
          "num_ctx_option": 65536,
          "wall_latency_ms": 466.39
        },
        {
          "ttft_ms": 301.58,
          "first_token_was_thinking": false,
          "wall_s": 0.449,
          "eval_count": 2,
          "eval_duration_ns": 146199000,
          "decode_tok_s": 13.68,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 146167000,
          "prefill_tok_s": 4542.749,
          "load_duration_ns": 151774232,
          "total_duration_ns": 447085683,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.784,
          "decode_tok_s_wall_gross": 4.454,
          "num_ctx_option": 65536,
          "wall_latency_ms": 449.06
        },
        {
          "ttft_ms": 309.41,
          "first_token_was_thinking": false,
          "wall_s": 0.4553,
          "eval_count": 2,
          "eval_duration_ns": 145121000,
          "decode_tok_s": 13.782,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 145272000,
          "prefill_tok_s": 4570.736,
          "load_duration_ns": 160331149,
          "total_duration_ns": 453597414,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.854,
          "decode_tok_s_wall_gross": 4.393,
          "num_ctx_option": 65536,
          "wall_latency_ms": 455.39
        },
        {
          "ttft_ms": 315.54,
          "first_token_was_thinking": false,
          "wall_s": 0.4643,
          "eval_count": 2,
          "eval_duration_ns": 147759000,
          "decode_tok_s": 13.536,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 149245000,
          "prefill_tok_s": 4449.06,
          "load_duration_ns": 162992337,
          "total_duration_ns": 462670947,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.722,
          "decode_tok_s_wall_gross": 4.308,
          "num_ctx_option": 65536,
          "wall_latency_ms": 464.38
        },
        {
          "ttft_ms": 313.51,
          "first_token_was_thinking": false,
          "wall_s": 0.4604,
          "eval_count": 2,
          "eval_duration_ns": 145801000,
          "decode_tok_s": 13.717,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 144294000,
          "prefill_tok_s": 4601.716,
          "load_duration_ns": 165833860,
          "total_duration_ns": 458844775,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.807,
          "decode_tok_s_wall_gross": 4.344,
          "num_ctx_option": 65536,
          "wall_latency_ms": 460.48
        },
        {
          "ttft_ms": 293.82,
          "first_token_was_thinking": false,
          "wall_s": 0.4432,
          "eval_count": 2,
          "eval_duration_ns": 148372000,
          "decode_tok_s": 13.48,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 148352000,
          "prefill_tok_s": 4475.841,
          "load_duration_ns": 141840229,
          "total_duration_ns": 441295077,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 6.694,
          "decode_tok_s_wall_gross": 4.513,
          "num_ctx_option": 65536,
          "wall_latency_ms": 443.28
        }
      ],
      "errors": [],
      "voided_by_errors": false,
      "latency": {
        "calls_scored": 10,
        "wall_latency_ms": {
          "all": [
            493.52,
            462.81,
            477.78,
            451.65,
            466.39,
            449.06,
            455.39,
            464.38,
            460.48,
            443.28
          ],
          "min": 443.28,
          "median": 461.64,
          "p95": 493.52,
          "max": 493.52
        },
        "ttft_ms": {
          "median": 314.52,
          "p95": 343.11,
          "max": 343.11
        },
        "eval_counts": [
          2,
          2,
          2,
          2,
          2,
          2,
          2,
          2,
          2,
          2
        ]
      },
      "power_mean_w_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 27.81
      },
      "power_mean_w_all_cards": 27.81,
      "thermal_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 4,
          "temp_max_c": 41.0,
          "temp_min_c": 38.0,
          "temp_rise_c": 3.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 8.0,
          "pcie_width_max": 8.0,
          "pcie_width_median": 8.0,
          "fan_mean_pct": null,
          "fan_max_pct": null,
          "clock_floor_mhz": 1672.0,
          "clock_max_mhz": 1717.0,
          "clock_median_mhz": 1672.0,
          "mem_clock_floor_mhz": 9001.0,
          "mem_clock_median_mhz": 9001.0,
          "mem_clock_max_mhz": 9001.0,
          "mem_clock_held": true,
          "memory_bus_width_bits": 256,
          "memory_technology": "GDDR7",
          "memory_bits_per_clock": null,
          "peak_mem_bandwidth_gbs_at_median_clock": null,
          "peak_mem_bandwidth_gbs_at_floor_clock": null,
          "peak_mem_bandwidth_gbs_at_max_clock": null,
          "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
          "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
          "clock_dropped": false,
          "power_limit_w": 95.0,
          "power_mean_w": 27.81,
          "power_pct_of_limit": 29.3,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 0.5,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": true,
          "verdict": "no clock drop during decode"
        }
      },
      "ups_during_level": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      }
    },
    {
      "concurrency": 4,
      "calls_requested": 10,
      "contention_gate": {
        "attempts": [
          {
            "window_s": 10.0,
            "samples": 20,
            "cards": {
              "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
                "utilization_pct": {
                  "min": 0.0,
                  "median": 0.0,
                  "max": 15.0,
                  "mean": 1.5,
                  "n": 20
                },
                "power_w": {
                  "min": 15.78,
                  "median": 15.84,
                  "max": 29.23,
                  "mean": 17.18,
                  "n": 20
                },
                "card_seen": true
              }
            },
            "max_mean_util_pct": 1.5,
            "all_cards_seen": true,
            "attempt": 1,
            "verdict": "quiet: busiest card mean 1.5% <= 5.0%"
          }
        ],
        "passed": true,
        "cards": [
          "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff"
        ]
      },
      "wall_s": 5.4279,
      "calls": [
        {
          "ttft_ms": 309.04,
          "first_token_was_thinking": false,
          "wall_s": 1.1024,
          "eval_count": 2,
          "eval_duration_ns": 791859000,
          "decode_tok_s": 2.526,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 147776000,
          "prefill_tok_s": 4493.287,
          "load_duration_ns": 156555701,
          "total_duration_ns": 1100074294,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 1.261,
          "decode_tok_s_wall_gross": 1.814,
          "num_ctx_option": 65536,
          "wall_latency_ms": 1102.42,
          "call": 1
        },
        {
          "ttft_ms": 3707.82,
          "first_token_was_thinking": false,
          "wall_s": 4.5304,
          "eval_count": 2,
          "eval_duration_ns": 820812000,
          "decode_tok_s": 2.437,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 2607157000,
          "prefill_tok_s": 254.684,
          "load_duration_ns": 176151857,
          "total_duration_ns": 4528397231,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 1.216,
          "decode_tok_s_wall_gross": 0.441,
          "num_ctx_option": 65536,
          "wall_latency_ms": 4530.5,
          "call": 2
        },
        {
          "ttft_ms": 2288.44,
          "first_token_was_thinking": false,
          "wall_s": 3.7082,
          "eval_count": 2,
          "eval_duration_ns": 1418607000,
          "decode_tok_s": 1.41,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 1979754000,
          "prefill_tok_s": 335.395,
          "load_duration_ns": 192092580,
          "total_duration_ns": 3706497895,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 0.704,
          "decode_tok_s_wall_gross": 0.539,
          "num_ctx_option": 65536,
          "wall_latency_ms": 3708.26,
          "call": 3
        },
        {
          "ttft_ms": 4529.81,
          "first_token_was_thinking": false,
          "wall_s": 4.6712,
          "eval_count": 2,
          "eval_duration_ns": 139855000,
          "decode_tok_s": 14.301,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 2240922000,
          "prefill_tok_s": 296.307,
          "load_duration_ns": 170996636,
          "total_duration_ns": 4669111088,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 7.074,
          "decode_tok_s_wall_gross": 0.428,
          "num_ctx_option": 65536,
          "wall_latency_ms": 4671.24,
          "call": 4
        },
        {
          "ttft_ms": 465.76,
          "first_token_was_thinking": false,
          "wall_s": 0.754,
          "eval_count": 2,
          "eval_duration_ns": 287551000,
          "decode_tok_s": 6.955,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 280219000,
          "prefill_tok_s": 2369.575,
          "load_duration_ns": 174254227,
          "total_duration_ns": 752754363,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 3.469,
          "decode_tok_s_wall_gross": 2.653,
          "num_ctx_option": 65536,
          "wall_latency_ms": 754.12,
          "call": 1
        },
        {
          "ttft_ms": 465.31,
          "first_token_was_thinking": false,
          "wall_s": 0.7541,
          "eval_count": 2,
          "eval_duration_ns": 288032000,
          "decode_tok_s": 6.944,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 279527000,
          "prefill_tok_s": 2375.441,
          "load_duration_ns": 170499608,
          "total_duration_ns": 752427649,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 3.463,
          "decode_tok_s_wall_gross": 2.652,
          "num_ctx_option": 65536,
          "wall_latency_ms": 754.17,
          "call": 2
        },
        {
          "ttft_ms": 465.37,
          "first_token_was_thinking": false,
          "wall_s": 0.754,
          "eval_count": 2,
          "eval_duration_ns": 288020000,
          "decode_tok_s": 6.944,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 279158000,
          "prefill_tok_s": 2378.581,
          "load_duration_ns": 155363206,
          "total_duration_ns": 751996690,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 3.465,
          "decode_tok_s_wall_gross": 2.653,
          "num_ctx_option": 65536,
          "wall_latency_ms": 754.05,
          "call": 3
        },
        {
          "ttft_ms": 465.75,
          "first_token_was_thinking": false,
          "wall_s": 0.7546,
          "eval_count": 2,
          "eval_duration_ns": 288046000,
          "decode_tok_s": 6.943,
          "prompt_eval_count": 664,
          "prompt_eval_duration_ns": 279346000,
          "prefill_tok_s": 2376.981,
          "load_duration_ns": 162214225,
          "total_duration_ns": 752512536,
          "streamed_chunks": 1,
          "response_chars": 3,
          "thinking_chars": 0,
          "done_reason": "stop",
          "decode_tok_s_wall": 3.462,
          "decode_tok_s_wall_gross": 2.651,
          "num_ctx_option": 65536,
          "wall_latency_ms": 754.65,
          "call": 4
        }
      ],
      "errors": [],
      "voided_by_errors": false,
      "latency": {
        "calls_scored": 8,
        "wall_latency_ms": {
          "all": [
            1102.42,
            4530.5,
            3708.26,
            4671.24,
            754.12,
            754.17,
            754.05,
            754.65
          ],
          "min": 754.05,
          "median": 928.54,
          "p95": 4671.24,
          "max": 4671.24
        },
        "ttft_ms": {
          "median": 465.75,
          "p95": 4529.81,
          "max": 4529.81
        },
        "eval_counts": [
          2,
          2,
          2,
          2,
          2,
          2,
          2,
          2
        ]
      },
      "power_mean_w_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 66.59
      },
      "power_mean_w_all_cards": 66.59,
      "thermal_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 13,
          "temp_max_c": 44.0,
          "temp_min_c": 39.0,
          "temp_rise_c": 5.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 8.0,
          "pcie_width_max": 8.0,
          "pcie_width_median": 8.0,
          "fan_mean_pct": null,
          "fan_max_pct": null,
          "clock_floor_mhz": 1605.0,
          "clock_max_mhz": 1965.0,
          "clock_median_mhz": 1770.0,
          "mem_clock_floor_mhz": 14001.0,
          "mem_clock_median_mhz": 14001.0,
          "mem_clock_max_mhz": 14001.0,
          "mem_clock_held": true,
          "memory_bus_width_bits": 256,
          "memory_technology": "GDDR7",
          "memory_bits_per_clock": null,
          "peak_mem_bandwidth_gbs_at_median_clock": null,
          "peak_mem_bandwidth_gbs_at_floor_clock": null,
          "peak_mem_bandwidth_gbs_at_max_clock": null,
          "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
          "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
          "clock_dropped": true,
          "power_limit_w": 95.0,
          "power_mean_w": 66.59,
          "power_pct_of_limit": 70.1,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 0.6923,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": true,
          "verdict": "NEITHER cap: the SM clock varied with draw at 70.1% of the cap and the card at 44 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
        }
      },
      "ups_during_level": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      }
    }
  ],
  "errors": [],
  "ollama_version": "0.32.13",
  "size_total": 40204813312,
  "size_vram": 22575149219,
  "fit_verdict": "SPLIT and scored as one: 56.2% VRAM / 43.8% RAM",
  "offload_split": "56.2% VRAM / 43.8% RAM",
  "ups_after_arm": {
    "ups": "pr1500@localhost",
    "read_utc": "2026-09-21T19:16:20Z",
    "available": false,
    "error": "no upsc on this box"
  },
  "finished_utc": "2026-09-21T19:16:20Z",
  "headline": "a doorman call on mistral-small3.2:24b costs 461.64 ms at the median and 493.52 ms at p95, one caller at a time"
}