{
  "model_id": "m5",
  "model_tag": "gemma4:26b",
  "arm": "one",
  "box_class": "desktop-3090",
  "mode": "filled-window",
  "arm_cards": [
    "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08"
  ],
  "card_labels": {
    "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": "index 0 - EVGA - GPU-7aa0be10"
  },
  "card_identity": {
    "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
      "index": 0,
      "bus_id": "00000000:01:00.0",
      "pci_sub_device_id": "0x39753842",
      "vendor": "EVGA",
      "board_per_operator": "EVGA GeForce RTX 3090 XC3 Ultra 24 GB",
      "vbios": "94.02.42.80.C1",
      "serial": null,
      "pcie_link_width_max": 16,
      "pcie_link_width_negotiated": 16,
      "memory_total_mib": 24576,
      "power_default_limit_w": 350.0,
      "power_min_limit_w": 100.0,
      "power_max_limit_w": 366.0,
      "power_limit_at_bench_start_w": 250.0,
      "note": "the single card seated in this box's x16 slot for the control arm. It came up at 280 W from a boot unit written for the box's former pair (/etc/desktop/gpu-power-caps.conf, 2026-08-27) whose other row names a card that is no longer present; the operator set 250 W by hand before this bench and the read-back is recorded per arm."
    }
  },
  "target_uuid": "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08",
  "started_utc": "2026-09-18T00:35:47Z",
  "num_ctx": 131072,
  "fill_fraction": 0.75,
  "target_prompt_tokens": 98304,
  "base_prompt_sha256": "90eedd0c53f9554ae3837674504fcb7090013d9a432a653352183c0f25a7ce5c",
  "num_predict": 256,
  "filled_note": "The context ladder measures what it costs to RESERVE a window. This arm measures what it costs to FILL one. The prompt is grown inside the harness until the server's own prompt_eval_count says the window is about as full as the target fraction asks, and then the same 256 tokens are generated from that depth. Three figures come out that the ladder cannot produce: prefill tokens a second over a real prompt, the time to the first token when the model has that much to read first, and the decode rate AT DEPTH, with a full KV cache behind every token instead of an empty one.",
  "filler_note": "The filled-window prompt is the series' frozen prompt REPEATED, each copy preceded by a numbered header line, until the target token count is reached. It is therefore highly repetitive text, and a real document of the same length may prefill faster or slower: attention over repeated spans is not attention over varied ones, and a KV cache built from repetition is not a harder or an easier case in any direction this bench measured. No estate corpus text, no visitor text and no private document is used or could be: the filler is built inside the harness from one public file whose sha256 is checked first.",
  "pcie_note": "This card negotiated PCIe x16 of a x16-capable generation-3 link -- the x16 slot of the same board on which the pair bench measured a x16 + x4 asymmetry. With one card there is no per-token traffic BETWEEN cards, so the link is not in the decode path the way it was for the split pair; it still carries the model in and the tokens out. Every run records `pcie.link.width.current` UNDER LOAD, because an idle card drops its link to save power and a width read at rest would flatter the result.",
  "thermal_layout_note": "ONE GeForce RTX 3090 24 GB in a consumer desktop, alone on the board's x16 slot -- the same box, the same case and the same airflow in which two GeForce RTX 3080 10 GB cards were measured the night before. A temperature here is a reading of THAT arrangement. It differs from the pair's arrangement in the way that matters most to a temperature: with one card there is no second board warming the air or blocking a face, so this card has open air on both sides where the pair did not. The card is traced on every run.",
  "instance_env": {
    "written_utc": "2026-09-18T00:35:46Z",
    "unit": "bench-3090-one",
    "shape": "one",
    "port": 11470,
    "base_url": "http://127.0.0.1:11470",
    "cuda_visible_devices": "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08",
    "OLLAMA_FLASH_ATTENTION": "1",
    "OLLAMA_KV_CACHE_TYPE": "q8_0",
    "OLLAMA_CONTEXT_LENGTH": 32768,
    "OLLAMA_<parallel-requests>": 1,
    "OLLAMA_MAX_LOADED_MODELS": 1,
    "OLLAMA_KEEP_ALIVE": "0 (the server default; every bench request sends its own keep_alive, which wins, and every arm unloads on the way out)",
    "models_dir": "/usr/share/ollama/.ollama/models",
    "models_dir_writable_by_this_user": false,
    "api_version": {
      "version": "0.32.13"
    },
    "power_limits_at_start_w": "0, 350.00 W;",
    "persistence_mode_at_start": "0, Enabled;",
    "pcie_link_width_at_start": "0, 16;"
  },
  "num_gpu_option": null,
  "num_gpu_note": "ollama decides for itself how many of a model's layers to put on the card. On the pair of 3080s that decision was measurably conservative -- gemma4:26b loaded 75.2% into VRAM and spilled 4.25 GiB to host RAM while about 5 GiB of the two cards' 20 GiB sat unused -- so the pair bench reported `auto` and a forced layer count side by side. On ONE 24 GB card there is no placement decision to make: the expected reading is 100% on the card with no spill. This harness records `options.num_gpu` on every run as `num_gpu_option` and reads the planner's actual placement back from /api/ps, so a spill on a card with room is visible as a finding rather than absorbed into a tok/s figure. A value at or above the model's layer count means every layer.",
  "memory_temp_support": {
    "checked_utc": "2026-09-18T00:35:47Z",
    "paths": {
      "nvidia-smi --query-gpu=temperature.memory": {
        "raw": "0, GPU-7aa0be10-974f-6430-52fe-09018a0e2e08, N/A",
        "readable": false
      },
      "nvidia-smi -q -d TEMPERATURE": {
        "lines": [
          "GPU Shutdown Temp                              : 98 C",
          "GPU Slowdown Temp                              : 95 C",
          "GPU Target Temperature                         : 83 C",
          "Memory Current Temp                            : N/A",
          "Memory Max Operating Temp                      : N/A"
        ],
        "readable": false
      },
      "NVML NVML_FI_DEV_MEMORY_TEMP": {
        "available": true,
        "field_id": 82,
        "cards": {
          "0": {
            "call_rc": 0,
            "field_rc": 3,
            "supported": false,
            "not_supported": true,
            "value_c": null
          }
        }
      }
    },
    "readable": false,
    "verdict": "the memory die's temperature is NOT readable on these cards through any of the three paths asked, so the memory-temperature stop condition could not arm and NO memory temperature is reported anywhere in this bench. The core temperature stop and the driver's own thermal-slowdown reasons are the thermal instrument instead."
  },
  "ups_before_arm": {
    "ups": "pr1500@localhost",
    "read_utc": "2026-09-18T00:35:47Z",
    "available": false,
    "error": "upsc rc=1: Error: Connection failure: Connection refused"
  },
  "runs": [
    {
      "ttft_ms": 1137.62,
      "first_token_was_thinking": false,
      "wall_s": 4.8508,
      "eval_count": 256,
      "eval_duration_ns": 3504824000,
      "decode_tok_s": 73.042,
      "prompt_eval_count": 97992,
      "prompt_eval_duration_ns": 96724000,
      "prefill_tok_s": 1013109.466,
      "load_duration_ns": 468828683,
      "total_duration_ns": 4646979857,
      "streamed_chunks": 256,
      "response_chars": 1039,
      "thinking_chars": 0,
      "done_reason": "length",
      "decode_tok_s_wall": 68.675,
      "decode_tok_s_wall_gross": 52.775,
      "num_ctx_option": 131072,
      "power": {
        "idle_baseline": {
          "seconds": 3.0,
          "samples": 6,
          "power_w": {
            "min": 23.84,
            "median": 24.24,
            "max": 25.65,
            "mean": 24.58,
            "n": 6
          },
          "utilization_pct": {
            "min": 0.0,
            "median": 0.0,
            "max": 0.0,
            "mean": 0.0,
            "n": 6
          },
          "memory_used_mib": {
            "min": 20129.0,
            "median": 20129.0,
            "max": 20129.0,
            "mean": 20129.0,
            "n": 6
          },
          "power_w_per_card": {
            "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 24.58
          },
          "memory_used_mib_per_card": {
            "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 20129.0
          },
          "power_w_all_cards": 24.58,
          "ups_whole_box": {
            "load_pct": null,
            "realpower_w": null,
            "realpower_nominal_w": null,
            "status": null,
            "note": "the UPS's reading of EVERYTHING it feeds -- the whole box, not the cards. The card figures beside it are board power from nvidia-smi. Neither is the other."
          }
        },
        "measured_over": "samples where the GPU was busy",
        "busy_samples": 13,
        "mean_w": 267.09,
        "max_w": 311.28,
        "median_w": 310.58,
        "mean_w_whole_window": 199.66,
        "max_w_whole_window": 311.28,
        "limit_w": 350.0,
        "over_idle_w": 242.51,
        "samples": 18,
        "per_gpu": {
          "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
            "index": "0",
            "name": "NVIDIA GeForce RTX 3090",
            "power_w": {
              "min": 23.65,
              "median": 308.62,
              "max": 311.28,
              "mean": 199.66,
              "n": 18
            },
            "power_limit_w": 350.0,
            "memory_used_mib": {
              "min": 20129.0,
              "median": 20129.0,
              "max": 20129.0,
              "mean": 20129.0,
              "n": 18
            },
            "utilization_pct": {
              "min": 0.0,
              "median": 93.5,
              "max": 94.0,
              "mean": 63.9,
              "n": 18
            },
            "temperature_c": {
              "min": 53.0,
              "median": 61.0,
              "max": 63.0,
              "mean": 59.3,
              "n": 18
            },
            "memory_temperature_c": null,
            "fan_pct": {
              "min": 55.0,
              "median": 56.0,
              "max": 59.0,
              "mean": 56.6,
              "n": 18
            },
            "pcie_width_current": {
              "min": 16.0,
              "median": 16.0,
              "max": 16.0,
              "mean": 16.0,
              "n": 18
            },
            "clocks_sm_mhz": {
              "min": 210.0,
              "median": 1755.0,
              "max": 1920.0,
              "mean": 1405.8,
              "n": 18
            },
            "clocks_mem_mhz": {
              "min": 405.0,
              "median": 9501.0,
              "max": 9501.0,
              "mean": 7479.7,
              "n": 18
            },
            "throttle_sw_power_cap_fraction": 0.7222,
            "throttle_hw_slowdown_fraction": 0.0,
            "throttle_sw_thermal_fraction": 0.0,
            "throttle_hw_thermal_fraction": 0.0,
            "n": 18,
            "busy": {
              "index": "0",
              "name": "NVIDIA GeForce RTX 3090",
              "power_w": {
                "min": 57.2,
                "median": 310.58,
                "max": 311.28,
                "mean": 267.09,
                "n": 13
              },
              "power_limit_w": 350.0,
              "memory_used_mib": {
                "min": 20129.0,
                "median": 20129.0,
                "max": 20129.0,
                "mean": 20129.0,
                "n": 13
              },
              "utilization_pct": {
                "min": 59.0,
                "median": 94.0,
                "max": 94.0,
                "mean": 88.5,
                "n": 13
              },
              "temperature_c": {
                "min": 58.0,
                "median": 62.0,
                "max": 63.0,
                "mean": 61.4,
                "n": 13
              },
              "memory_temperature_c": null,
              "fan_pct": {
                "min": 55.0,
                "median": 56.0,
                "max": 57.0,
                "mean": 55.9,
                "n": 13
              },
              "pcie_width_current": {
                "min": 16.0,
                "median": 16.0,
                "max": 16.0,
                "mean": 16.0,
                "n": 13
              },
              "clocks_sm_mhz": {
                "min": 1710.0,
                "median": 1800.0,
                "max": 1920.0,
                "mean": 1783.8,
                "n": 13
              },
              "clocks_mem_mhz": {
                "min": 9501.0,
                "median": 9501.0,
                "max": 9501.0,
                "mean": 9501.0,
                "n": 13
              },
              "throttle_sw_power_cap_fraction": 1.0,
              "throttle_hw_slowdown_fraction": 0.0,
              "throttle_sw_thermal_fraction": 0.0,
              "throttle_hw_thermal_fraction": 0.0,
              "n": 13
            },
            "busy_samples": 13
          }
        }
      },
      "energy_j_per_1k_tokens": 3656.65,
      "power_mean_w_all_cards": 267.09,
      "energy_j_per_1k_tokens_all_cards": 3656.65,
      "watts_per_tok_s": 3.6567,
      "thermal": {
        "rule_version": 2,
        "measured_over": "samples where the GPU was busy (utilisation > 0)",
        "busy_samples": 13,
        "temp_max_c": 63.0,
        "temp_min_c": 58.0,
        "temp_rise_c": 5.0,
        "memory_temp_max_c": null,
        "memory_temp_median_c": null,
        "memory_temp_readable": false,
        "pcie_width_min": 16.0,
        "pcie_width_max": 16.0,
        "pcie_width_median": 16.0,
        "fan_mean_pct": 55.9,
        "fan_max_pct": 57.0,
        "clock_floor_mhz": 1710.0,
        "clock_max_mhz": 1920.0,
        "clock_median_mhz": 1800.0,
        "mem_clock_floor_mhz": 9501.0,
        "mem_clock_max_mhz": 9501.0,
        "mem_clock_held": true,
        "clock_dropped": true,
        "power_limit_w": 350.0,
        "power_mean_w": 267.09,
        "power_pct_of_limit": 76.3,
        "power_pinned_at_limit": false,
        "sw_power_cap_active_fraction": 1.0,
        "hw_slowdown_active_fraction": 0.0,
        "sw_thermal_active_fraction": 0.0,
        "hw_thermal_active_fraction": 0.0,
        "at_throttling_temperature": false,
        "stop_core_temp_c": 83.0,
        "stop_memory_temp_c": 100.0,
        "hit_core_temp_stop": false,
        "hit_memory_temp_stop": false,
        "throttle_temperature_c": 80.0,
        "temp_rising_within_run": true,
        "verdict": "NEITHER cap: the SM clock varied with draw at 76.3% of the cap and the card at 63 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
      },
      "thermal_per_card": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 13,
          "temp_max_c": 63.0,
          "temp_min_c": 58.0,
          "temp_rise_c": 5.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 16.0,
          "pcie_width_max": 16.0,
          "pcie_width_median": 16.0,
          "fan_mean_pct": 55.9,
          "fan_max_pct": 57.0,
          "clock_floor_mhz": 1710.0,
          "clock_max_mhz": 1920.0,
          "clock_median_mhz": 1800.0,
          "mem_clock_floor_mhz": 9501.0,
          "mem_clock_max_mhz": 9501.0,
          "mem_clock_held": true,
          "clock_dropped": true,
          "power_limit_w": 350.0,
          "power_mean_w": 267.09,
          "power_pct_of_limit": 76.3,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 1.0,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": true,
          "verdict": "NEITHER cap: the SM clock varied with draw at 76.3% of the cap and the card at 63 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
        }
      },
      "memory_used_mib_per_card": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 20129.0
      },
      "pcie_width_under_load": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
          "min": 16.0,
          "median": 16.0,
          "max": 16.0,
          "mean": 16.0,
          "n": 18
        }
      },
      "ups_during_run": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      },
      "stop_conditions_hit": [],
      "run": 1
    },
    {
      "ttft_ms": 1113.07,
      "first_token_was_thinking": false,
      "wall_s": 4.8104,
      "eval_count": 256,
      "eval_duration_ns": 3491193000,
      "decode_tok_s": 73.327,
      "prompt_eval_count": 97992,
      "prompt_eval_duration_ns": 72075000,
      "prefill_tok_s": 1359583.767,
      "load_duration_ns": 473596274,
      "total_duration_ns": 4608874587,
      "streamed_chunks": 256,
      "response_chars": 1039,
      "thinking_chars": 0,
      "done_reason": "length",
      "decode_tok_s_wall": 68.968,
      "decode_tok_s_wall_gross": 53.218,
      "num_ctx_option": 131072,
      "power": {
        "idle_baseline": {
          "seconds": 3.0,
          "samples": 6,
          "power_w": {
            "min": 130.97,
            "median": 131.03,
            "max": 131.08,
            "mean": 131.03,
            "n": 6
          },
          "utilization_pct": {
            "min": 0.0,
            "median": 0.0,
            "max": 0.0,
            "mean": 0.0,
            "n": 6
          },
          "memory_used_mib": {
            "min": 20129.0,
            "median": 20129.0,
            "max": 20129.0,
            "mean": 20129.0,
            "n": 6
          },
          "power_w_per_card": {
            "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 131.03
          },
          "memory_used_mib_per_card": {
            "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 20129.0
          },
          "power_w_all_cards": 131.03,
          "ups_whole_box": {
            "load_pct": null,
            "realpower_w": null,
            "realpower_nominal_w": null,
            "status": null,
            "note": "the UPS's reading of EVERYTHING it feeds -- the whole box, not the cards. The card figures beside it are board power from nvidia-smi. Neither is the other."
          }
        },
        "measured_over": "samples where the GPU was busy",
        "busy_samples": 11,
        "mean_w": 303.62,
        "max_w": 310.96,
        "median_w": 309.79,
        "mean_w_whole_window": 240.28,
        "max_w_whole_window": 310.96,
        "limit_w": 350.0,
        "over_idle_w": 172.59,
        "samples": 18,
        "per_gpu": {
          "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
            "index": "0",
            "name": "NVIDIA GeForce RTX 3090",
            "power_w": {
              "min": 130.8,
              "median": 303.67,
              "max": 310.96,
              "mean": 240.28,
              "n": 18
            },
            "power_limit_w": 350.0,
            "memory_used_mib": {
              "min": 20129.0,
              "median": 20129.0,
              "max": 20129.0,
              "mean": 20129.0,
              "n": 18
            },
            "utilization_pct": {
              "min": 0.0,
              "median": 93.5,
              "max": 95.0,
              "mean": 57.5,
              "n": 18
            },
            "temperature_c": {
              "min": 56.0,
              "median": 61.0,
              "max": 62.0,
              "mean": 59.8,
              "n": 18
            },
            "memory_temperature_c": null,
            "fan_pct": {
              "min": 54.0,
              "median": 54.0,
              "max": 54.0,
              "mean": 54.0,
              "n": 18
            },
            "pcie_width_current": {
              "min": 16.0,
              "median": 16.0,
              "max": 16.0,
              "mean": 16.0,
              "n": 18
            },
            "clocks_sm_mhz": {
              "min": 1725.0,
              "median": 1770.0,
              "max": 1920.0,
              "mean": 1770.8,
              "n": 18
            },
            "clocks_mem_mhz": {
              "min": 9501.0,
              "median": 9501.0,
              "max": 9501.0,
              "mean": 9501.0,
              "n": 18
            },
            "throttle_sw_power_cap_fraction": 0.6111,
            "throttle_hw_slowdown_fraction": 0.0,
            "throttle_sw_thermal_fraction": 0.0,
            "throttle_hw_thermal_fraction": 0.0,
            "n": 18,
            "busy": {
              "index": "0",
              "name": "NVIDIA GeForce RTX 3090",
              "power_w": {
                "min": 252.04,
                "median": 309.79,
                "max": 310.96,
                "mean": 303.62,
                "n": 11
              },
              "power_limit_w": 350.0,
              "memory_used_mib": {
                "min": 20129.0,
                "median": 20129.0,
                "max": 20129.0,
                "mean": 20129.0,
                "n": 11
              },
              "utilization_pct": {
                "min": 93.0,
                "median": 94.0,
                "max": 95.0,
                "mean": 94.1,
                "n": 11
              },
              "temperature_c": {
                "min": 61.0,
                "median": 62.0,
                "max": 62.0,
                "mean": 61.6,
                "n": 11
              },
              "memory_temperature_c": null,
              "fan_pct": {
                "min": 54.0,
                "median": 54.0,
                "max": 54.0,
                "mean": 54.0,
                "n": 11
              },
              "pcie_width_current": {
                "min": 16.0,
                "median": 16.0,
                "max": 16.0,
                "mean": 16.0,
                "n": 11
              },
              "clocks_sm_mhz": {
                "min": 1755.0,
                "median": 1770.0,
                "max": 1920.0,
                "mean": 1789.1,
                "n": 11
              },
              "clocks_mem_mhz": {
                "min": 9501.0,
                "median": 9501.0,
                "max": 9501.0,
                "mean": 9501.0,
                "n": 11
              },
              "throttle_sw_power_cap_fraction": 1.0,
              "throttle_hw_slowdown_fraction": 0.0,
              "throttle_sw_thermal_fraction": 0.0,
              "throttle_hw_thermal_fraction": 0.0,
              "n": 11
            },
            "busy_samples": 11
          }
        }
      },
      "energy_j_per_1k_tokens": 4140.61,
      "power_mean_w_all_cards": 303.62,
      "energy_j_per_1k_tokens_all_cards": 4140.61,
      "watts_per_tok_s": 4.1406,
      "thermal": {
        "rule_version": 2,
        "measured_over": "samples where the GPU was busy (utilisation > 0)",
        "busy_samples": 11,
        "temp_max_c": 62.0,
        "temp_min_c": 61.0,
        "temp_rise_c": 1.0,
        "memory_temp_max_c": null,
        "memory_temp_median_c": null,
        "memory_temp_readable": false,
        "pcie_width_min": 16.0,
        "pcie_width_max": 16.0,
        "pcie_width_median": 16.0,
        "fan_mean_pct": 54.0,
        "fan_max_pct": 54.0,
        "clock_floor_mhz": 1755.0,
        "clock_max_mhz": 1920.0,
        "clock_median_mhz": 1770.0,
        "mem_clock_floor_mhz": 9501.0,
        "mem_clock_max_mhz": 9501.0,
        "mem_clock_held": true,
        "clock_dropped": true,
        "power_limit_w": 350.0,
        "power_mean_w": 303.62,
        "power_pct_of_limit": 86.7,
        "power_pinned_at_limit": false,
        "sw_power_cap_active_fraction": 1.0,
        "hw_slowdown_active_fraction": 0.0,
        "sw_thermal_active_fraction": 0.0,
        "hw_thermal_active_fraction": 0.0,
        "at_throttling_temperature": false,
        "stop_core_temp_c": 83.0,
        "stop_memory_temp_c": 100.0,
        "hit_core_temp_stop": false,
        "hit_memory_temp_stop": false,
        "throttle_temperature_c": 80.0,
        "temp_rising_within_run": false,
        "verdict": "NEITHER cap: the SM clock varied with draw at 86.7% of the cap and the card at 62 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
      },
      "thermal_per_card": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 11,
          "temp_max_c": 62.0,
          "temp_min_c": 61.0,
          "temp_rise_c": 1.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 16.0,
          "pcie_width_max": 16.0,
          "pcie_width_median": 16.0,
          "fan_mean_pct": 54.0,
          "fan_max_pct": 54.0,
          "clock_floor_mhz": 1755.0,
          "clock_max_mhz": 1920.0,
          "clock_median_mhz": 1770.0,
          "mem_clock_floor_mhz": 9501.0,
          "mem_clock_max_mhz": 9501.0,
          "mem_clock_held": true,
          "clock_dropped": true,
          "power_limit_w": 350.0,
          "power_mean_w": 303.62,
          "power_pct_of_limit": 86.7,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 1.0,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": false,
          "verdict": "NEITHER cap: the SM clock varied with draw at 86.7% of the cap and the card at 62 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
        }
      },
      "memory_used_mib_per_card": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 20129.0
      },
      "pcie_width_under_load": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
          "min": 16.0,
          "median": 16.0,
          "max": 16.0,
          "mean": 16.0,
          "n": 18
        }
      },
      "ups_during_run": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      },
      "stop_conditions_hit": [],
      "run": 2
    },
    {
      "ttft_ms": 1112.85,
      "first_token_was_thinking": false,
      "wall_s": 4.806,
      "eval_count": 256,
      "eval_duration_ns": 3488128000,
      "decode_tok_s": 73.392,
      "prompt_eval_count": 97992,
      "prompt_eval_duration_ns": 71954000,
      "prefill_tok_s": 1361870.084,
      "load_duration_ns": 465590239,
      "total_duration_ns": 4605538112,
      "streamed_chunks": 256,
      "response_chars": 1039,
      "thinking_chars": 0,
      "done_reason": "length",
      "decode_tok_s_wall": 69.047,
      "decode_tok_s_wall_gross": 53.267,
      "num_ctx_option": 131072,
      "power": {
        "idle_baseline": {
          "seconds": 3.0,
          "samples": 6,
          "power_w": {
            "min": 130.75,
            "median": 130.79,
            "max": 130.79,
            "mean": 130.78,
            "n": 6
          },
          "utilization_pct": {
            "min": 0.0,
            "median": 0.0,
            "max": 0.0,
            "mean": 0.0,
            "n": 6
          },
          "memory_used_mib": {
            "min": 20129.0,
            "median": 20129.0,
            "max": 20129.0,
            "mean": 20129.0,
            "n": 6
          },
          "power_w_per_card": {
            "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 130.78
          },
          "memory_used_mib_per_card": {
            "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 20129.0
          },
          "power_w_all_cards": 130.78,
          "ups_whole_box": {
            "load_pct": null,
            "realpower_w": null,
            "realpower_nominal_w": null,
            "status": null,
            "note": "the UPS's reading of EVERYTHING it feeds -- the whole box, not the cards. The card figures beside it are board power from nvidia-smi. Neither is the other."
          }
        },
        "measured_over": "samples where the GPU was busy",
        "busy_samples": 12,
        "mean_w": 293.38,
        "max_w": 311.43,
        "median_w": 309.38,
        "mean_w_whole_window": 240.47,
        "max_w_whole_window": 311.43,
        "limit_w": 350.0,
        "over_idle_w": 162.6,
        "samples": 18,
        "per_gpu": {
          "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
            "index": "0",
            "name": "NVIDIA GeForce RTX 3090",
            "power_w": {
              "min": 130.6,
              "median": 305.44,
              "max": 311.43,
              "mean": 240.47,
              "n": 18
            },
            "power_limit_w": 350.0,
            "memory_used_mib": {
              "min": 20129.0,
              "median": 20129.0,
              "max": 20129.0,
              "mean": 20129.0,
              "n": 18
            },
            "utilization_pct": {
              "min": 0.0,
              "median": 94.0,
              "max": 94.0,
              "mean": 62.1,
              "n": 18
            },
            "temperature_c": {
              "min": 56.0,
              "median": 61.0,
              "max": 62.0,
              "mean": 59.6,
              "n": 18
            },
            "memory_temperature_c": null,
            "fan_pct": {
              "min": 53.0,
              "median": 53.0,
              "max": 54.0,
              "mean": 53.3,
              "n": 18
            },
            "pcie_width_current": {
              "min": 16.0,
              "median": 16.0,
              "max": 16.0,
              "mean": 16.0,
              "n": 18
            },
            "clocks_sm_mhz": {
              "min": 1725.0,
              "median": 1770.0,
              "max": 1890.0,
              "mean": 1771.7,
              "n": 18
            },
            "clocks_mem_mhz": {
              "min": 9501.0,
              "median": 9501.0,
              "max": 9501.0,
              "mean": 9501.0,
              "n": 18
            },
            "throttle_sw_power_cap_fraction": 0.6667,
            "throttle_hw_slowdown_fraction": 0.0,
            "throttle_sw_thermal_fraction": 0.0,
            "throttle_hw_thermal_fraction": 0.0,
            "n": 18,
            "busy": {
              "index": "0",
              "name": "NVIDIA GeForce RTX 3090",
              "power_w": {
                "min": 179.46,
                "median": 309.38,
                "max": 311.43,
                "mean": 293.38,
                "n": 12
              },
              "power_limit_w": 350.0,
              "memory_used_mib": {
                "min": 20129.0,
                "median": 20129.0,
                "max": 20129.0,
                "mean": 20129.0,
                "n": 12
              },
              "utilization_pct": {
                "min": 89.0,
                "median": 94.0,
                "max": 94.0,
                "mean": 93.2,
                "n": 12
              },
              "temperature_c": {
                "min": 60.0,
                "median": 61.0,
                "max": 62.0,
                "mean": 61.1,
                "n": 12
              },
              "memory_temperature_c": null,
              "fan_pct": {
                "min": 53.0,
                "median": 53.0,
                "max": 53.0,
                "mean": 53.0,
                "n": 12
              },
              "pcie_width_current": {
                "min": 16.0,
                "median": 16.0,
                "max": 16.0,
                "mean": 16.0,
                "n": 12
              },
              "clocks_sm_mhz": {
                "min": 1755.0,
                "median": 1785.0,
                "max": 1890.0,
                "mean": 1790.0,
                "n": 12
              },
              "clocks_mem_mhz": {
                "min": 9501.0,
                "median": 9501.0,
                "max": 9501.0,
                "mean": 9501.0,
                "n": 12
              },
              "throttle_sw_power_cap_fraction": 1.0,
              "throttle_hw_slowdown_fraction": 0.0,
              "throttle_sw_thermal_fraction": 0.0,
              "throttle_hw_thermal_fraction": 0.0,
              "n": 12
            },
            "busy_samples": 12
          }
        }
      },
      "energy_j_per_1k_tokens": 3997.45,
      "power_mean_w_all_cards": 293.38,
      "energy_j_per_1k_tokens_all_cards": 3997.45,
      "watts_per_tok_s": 3.9974,
      "thermal": {
        "rule_version": 2,
        "measured_over": "samples where the GPU was busy (utilisation > 0)",
        "busy_samples": 12,
        "temp_max_c": 62.0,
        "temp_min_c": 60.0,
        "temp_rise_c": 2.0,
        "memory_temp_max_c": null,
        "memory_temp_median_c": null,
        "memory_temp_readable": false,
        "pcie_width_min": 16.0,
        "pcie_width_max": 16.0,
        "pcie_width_median": 16.0,
        "fan_mean_pct": 53.0,
        "fan_max_pct": 53.0,
        "clock_floor_mhz": 1755.0,
        "clock_max_mhz": 1890.0,
        "clock_median_mhz": 1785.0,
        "mem_clock_floor_mhz": 9501.0,
        "mem_clock_max_mhz": 9501.0,
        "mem_clock_held": true,
        "clock_dropped": true,
        "power_limit_w": 350.0,
        "power_mean_w": 293.38,
        "power_pct_of_limit": 83.8,
        "power_pinned_at_limit": false,
        "sw_power_cap_active_fraction": 1.0,
        "hw_slowdown_active_fraction": 0.0,
        "sw_thermal_active_fraction": 0.0,
        "hw_thermal_active_fraction": 0.0,
        "at_throttling_temperature": false,
        "stop_core_temp_c": 83.0,
        "stop_memory_temp_c": 100.0,
        "hit_core_temp_stop": false,
        "hit_memory_temp_stop": false,
        "throttle_temperature_c": 80.0,
        "temp_rising_within_run": false,
        "verdict": "NEITHER cap: the SM clock varied with draw at 83.8% of the cap and the card at 62 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
      },
      "thermal_per_card": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 12,
          "temp_max_c": 62.0,
          "temp_min_c": 60.0,
          "temp_rise_c": 2.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 16.0,
          "pcie_width_max": 16.0,
          "pcie_width_median": 16.0,
          "fan_mean_pct": 53.0,
          "fan_max_pct": 53.0,
          "clock_floor_mhz": 1755.0,
          "clock_max_mhz": 1890.0,
          "clock_median_mhz": 1785.0,
          "mem_clock_floor_mhz": 9501.0,
          "mem_clock_max_mhz": 9501.0,
          "mem_clock_held": true,
          "clock_dropped": true,
          "power_limit_w": 350.0,
          "power_mean_w": 293.38,
          "power_pct_of_limit": 83.8,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 1.0,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": false,
          "verdict": "NEITHER cap: the SM clock varied with draw at 83.8% of the cap and the card at 62 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
        }
      },
      "memory_used_mib_per_card": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 20129.0
      },
      "pcie_width_under_load": {
        "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": {
          "min": 16.0,
          "median": 16.0,
          "max": 16.0,
          "mean": 16.0,
          "n": 18
        }
      },
      "ups_during_run": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      },
      "stop_conditions_hit": [],
      "run": 3
    }
  ],
  "errors": [],
  "ps_after_load": [
    {
      "name": "gemma4:26b",
      "model": "gemma4:26b",
      "size": 17826494544,
      "digest": "5571076f3d70050487b26b341705799e0ab29b808164f90d20d4cf84f699d251",
      "details": {
        "parent_model": "",
        "format": "gguf",
        "family": "gemma4",
        "families": [
          "gemma4"
        ],
        "parameter_size": "25.8B",
        "quantization_level": "Q4_K_M"
      },
      "expires_at": "2026-09-18T00:45:51.886664069Z",
      "size_vram": 17826494544,
      "context_length": 131072
    }
  ],
  "size_total": 17826494544,
  "size_vram": 17826494544,
  "context_length_loaded": 131072,
  "fits_fully_in_vram": true,
  "fit_verdict": "fits: fully resident on the card",
  "memory_used_mib_per_card": {
    "GPU-7aa0be10-974f-6430-52fe-09018a0e2e08": 20129.0
  },
  "fill": {
    "target_tokens": 98304,
    "base_prompt_tokens": 536,
    "num_ctx": 131072,
    "filler_note": "The filled-window prompt is the series' frozen prompt REPEATED, each copy preceded by a numbered header line, until the target token count is reached. It is therefore highly repetitive text, and a real document of the same length may prefill faster or slower: attention over repeated spans is not attention over varied ones, and a KV cache built from repetition is not a harder or an easier case in any direction this bench measured. No estate corpus text, no visitor text and no private document is used or could be: the filler is built inside the harness from one public file whose sha256 is checked first.",
    "rounds": [
      {
        "round": 1,
        "copies": 183,
        "measured_tokens": 97992,
        "verdict": "at target"
      }
    ],
    "copies": 183,
    "measured_tokens": 97992,
    "prompt_chars": 388948,
    "fraction_of_target": 0.9968
  },
  "prompt_sha256": "f5461ceaa65b645619007d9c3b6f7004b2191c9c218eeb07c1f9d146cdbf89ec",
  "prompt_sha256_note": "the digest of the FILLED prompt actually sent -- built inside the harness from the frozen prompt whose own digest is base_prompt_sha256",
  "prompt_chars": 388948,
  "measured_prompt_tokens": 97992,
  "window_occupancy": 0.7476,
  "cold_prefill": {
    "prompt_eval_count": 98023,
    "prompt_eval_duration_ns": 43171529000,
    "prefill_tok_s": 2270.547,
    "ttft_ms": 44686.01,
    "note": "ONE pass with a unique header prepended, so the KV prefix cache MISSES and the server really reads the whole prompt. This is the prefill rate; the figure in the scored runs below is a cached-prefix number and is labelled as one."
  },
  "prefill_note": "TWO prefill figures are reported and they answer different questions. `cold_prefill` is one pass with a unique header prepended so the prefix cache misses -- the rate at which this pair actually reads a prompt of this length. The scored runs' `prefill_tok_s` all send the SAME prompt, so from the second request the prefix is cached and the figure is a cache hit, not a read. The scored runs' TTFT and decode ARE meaningful: they are what a caller sees on a repeat query at this depth.",
  "warmup_discarded": {
    "ttft_ms": 1822.61,
    "first_token_was_thinking": false,
    "wall_s": 5.5602,
    "eval_count": 256,
    "eval_duration_ns": 3529137000,
    "decode_tok_s": 72.539,
    "prompt_eval_count": 97992,
    "prompt_eval_duration_ns": 132340000,
    "prefill_tok_s": 740456.4,
    "load_duration_ns": 462811105,
    "total_duration_ns": 5356491456,
    "streamed_chunks": 256,
    "response_chars": 1056,
    "thinking_chars": 0,
    "done_reason": "length",
    "decode_tok_s_wall": 68.225,
    "decode_tok_s_wall_gross": 46.041,
    "num_ctx_option": 131072
  },
  "peak_runner_rss_gib": 3.34,
  "spread_gate": {
    "limit": 0.15,
    "values": [
      73.042,
      73.327,
      73.392
    ],
    "spread_fraction": 0.0048,
    "verdict": "spread 0.5% of the median within the 15% gate"
  },
  "voided_by_gate": false,
  "derived": {
    "prompt_eval_count": {
      "min": 97992,
      "median": 97992,
      "max": 97992,
      "mean": 97992.0,
      "n": 3
    },
    "prefill_tok_s": {
      "min": 1013109.466,
      "median": 1359583.767,
      "max": 1361870.084,
      "mean": 1244854.439,
      "n": 3
    },
    "ttft_ms": {
      "min": 1112.85,
      "median": 1113.07,
      "max": 1137.62,
      "mean": 1121.18,
      "n": 3
    },
    "decode_tok_s": {
      "min": 73.042,
      "median": 73.327,
      "max": 73.392,
      "mean": 73.254,
      "n": 3
    },
    "decode_tok_s_wall": {
      "min": 68.675,
      "median": 68.968,
      "max": 69.047,
      "mean": 68.897,
      "n": 3
    },
    "wall_s": {
      "min": 4.806,
      "median": 4.8104,
      "max": 4.8508,
      "mean": 4.8224,
      "n": 3
    },
    "power_mean_w": {
      "min": 267.09,
      "median": 293.38,
      "max": 303.62,
      "mean": 288.03,
      "n": 3
    },
    "power_mean_w_all_cards": {
      "min": 267.09,
      "median": 293.38,
      "max": 303.62,
      "mean": 288.03,
      "n": 3
    },
    "energy_j_per_1k_tokens": {
      "min": 3656.65,
      "median": 3997.45,
      "max": 4140.61,
      "mean": 3931.57,
      "n": 3
    },
    "energy_j_per_1k_tokens_all_cards": {
      "min": 3656.65,
      "median": 3997.45,
      "max": 4140.61,
      "mean": 3931.57,
      "n": 3
    },
    "temp_max_c": {
      "min": 62.0,
      "median": 62.0,
      "max": 63.0,
      "mean": 62.3,
      "n": 3
    }
  },
  "ups_after_arm": {
    "ups": "pr1500@localhost",
    "read_utc": "2026-09-18T00:38:30Z",
    "available": false,
    "error": "upsc rc=1: Error: Connection failure: Connection refused"
  },
  "finished_utc": "2026-09-18T00:38:30Z",
  "prompt_path": "/workshop/estate-bench/cpu-only-2026-08-25/frozen-p512.txt"
}