{
  "model_id": "m5",
  "model_tag": "gemma4:26b",
  "arm": "one",
  "box_class": "gpu-5090-laptop-24g",
  "mode": "filled-window",
  "arm_cards": [
    "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff"
  ],
  "model_density": "moe",
  "card_labels": {
    "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": "soldered (mobile; no socket \u2014 link width is traced under load, never assumed) at 00000000:01:00.0 - vendor unread - GPU-edff232c"
  },
  "card_identity": {
    "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
      "index": null,
      "bus_id": "00000000:01:00.0",
      "pci_sub_device_id": null,
      "vendor": null,
      "board_per_operator": null,
      "vbios": null,
      "serial": null,
      "pcie_link_width_max": null,
      "pcie_link_width_negotiated": null,
      "memory_total_mib": null,
      "memory_bus_width_bits": null,
      "clocks_max_sm_mhz": null,
      "clocks_max_memory_mhz": null,
      "power_default_limit_w": null,
      "power_min_limit_w": null,
      "power_max_limit_w": null,
      "power_limit_at_bench_start_w": null,
      "note": "the single NVIDIA GeForce RTX 5090 Laptop GPU 24 GB SOLDERED to this machine's board at 00000000:01:00.0. There is no seat, no partner and no card to swap. \u26a0 THE TRAP ON THIS BOX IS NOT A STALE BOOT UNIT, IT IS A RUNNING DAEMON: nvidia-powerd.service (NVIDIA Dynamic Boost) floats this board's power limit between its default and its maximum against the CPU's draw, continuously and without asking, and it has been running since 2026-09-08. The lesson every leg of this ladder carries \u2014 a cap read once at the start of a night is not a cap \u2014 is true here in its strongest form: such a reading is a sample of a moving signal. Every arm reads the limit back in the same call as its own data, every stage reads it again at its close, and a stage that finds it moved STOPS rather than relabelling its file. The second trap is a CLOCK LOCK: ai-perf.service runs `nvidia-smi -lgc 1200,2550` at boot on this box and the cards that produced the rows this bench compares against ran unlocked, so a floored clock changes what a power cap means and every record carries clocks.sm min and max so the confound is visible rather than implied."
    }
  },
  "target_uuid": "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff",
  "started_utc": "2026-09-22T00:23:09Z",
  "num_ctx": 131072,
  "fill_fraction": 0.75,
  "target_prompt_tokens": 98304,
  "base_prompt_sha256": "90eedd0c53f9554ae3837674504fcb7090013d9a432a653352183c0f25a7ce5c",
  "num_predict": 256,
  "filled_note": "The context ladder measures what it costs to RESERVE a window. This arm measures what it costs to FILL one. The prompt is grown inside the harness until the server's own prompt_eval_count says the window is about as full as the target fraction asks, and then the same 256 tokens are generated from that depth. Three figures come out that the ladder cannot produce: prefill tokens a second over a real prompt, the time to the first token when the model has that much to read first, and the decode rate AT DEPTH, with a full KV cache behind every token instead of an empty one.",
  "filler_note": "The filled-window prompt is the series' frozen prompt REPEATED, each copy preceded by a numbered header line, until the target token count is reached. It is therefore highly repetitive text, and a real document of the same length may prefill faster or slower: attention over repeated spans is not attention over varied ones, and a KV cache built from repetition is not a harder or an easier case in any direction this bench measured. No estate corpus text, no visitor text and no private document is used or could be: the filler is built inside the harness from one public file whose sha256 is checked first.",
  "pcie_note": "This card negotiated PCIe x16 of a x16-capable generation-3 link -- the same x16 slot the 3090 used, and the slot on which the pair bench measured a x16 + x4 asymmetry. With one card there is no per-token traffic BETWEEN cards, so the link is not in the decode path the way it was for the split pair; it still carries the model in and the tokens out. Every run records `pcie.link.width.current` UNDER LOAD, because an idle card drops its link to save power and a width read at rest would flatter the result.",
  "thermal_layout_note": "ONE GeForce RTX 3090 Ti 24 GB in a consumer desktop, alone on the board's x16 slot -- the same box, the same case and the same airflow in which one GeForce RTX 3090 24 GB was measured earlier the same day and two GeForce RTX 3080 10 GB cards the night before. A temperature here is a reading of THAT arrangement. Against the single 3090 the arrangement is the same one, which is what makes the two cards' thermal rows comparable; against the PAIR it is not, because with one card there is no second board warming the air or blocking a face. The card is traced on every run.",
  "instance_env": {
    "kind": "instance-env-receipt",
    "box_class": "gpu-5090-laptop-24g",
    "target_uuid": "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff",
    "written_utc": "2026-09-22T00:23:09Z",
    "unit": "bench-5090laptop-one",
    "shape": "one",
    "port": 11470,
    "base_url": "http://127.0.0.1:11470",
    "cuda_visible_devices": "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff",
    "OLLAMA_FLASH_ATTENTION": "1",
    "OLLAMA_KV_CACHE_TYPE": "q8_0",
    "OLLAMA_CONTEXT_LENGTH": 32768,
    "OLLAMA_<parallel-requests>": 1,
    "OLLAMA_MAX_LOADED_MODELS": 1,
    "OLLAMA_KEEP_ALIVE": "0 (the server default; every bench request sends its own keep_alive, which wins, and every arm unloads on the way out)",
    "models_dir": "/usr/share/ollama/.ollama/models",
    "models_dir_writable_by_this_user": false,
    "api_version": {
      "version": "0.32.13"
    },
    "ollama_binary": "/workshop/bench-laptop-5090-2026-09-21/ollama-0.32.13/bin/ollama",
    "ollama_client_version": "0.32.13",
    "ollama_version_pin": "0.32.13",
    "disclosed_tenants": "nomic-embed-text:latest on the SYSTEM ollama (:<the runtime's default port>, OLLAMA_KEEP_ALIVE=-1, ~323 MB VRAM) \u2014 disclosed, never unloaded",
    "power_and_clocks_at_start_csv": "[N/A], 150.00 W, 95.00 W, 175.00 W, 1192 MHz, 810 MHz, 3090 MHz",
    "power_and_clocks_at_start_fields": "power.limit,enforced.power.limit,power.default_limit,power.max_limit,clocks.sm,clocks.mem,clocks.max.sm",
    "nvidia_powerd_at_start": "active",
    "clock_lock_unit_at_start": "active",
    "clock_lock_unit_name": "ai-perf.service",
    "clock_lock_range_declared": "1200,2550",
    "clock_lock_unit_state_is_not_evidence": "ai-perf.service is Type=oneshot and reads 'active (exited)' for the whole uptime whatever happens to the card afterwards. The lock's real state is read from the CARD (clocks.sm against the lock's floor) and is recorded in this leg's clock-lock receipt.",
    "power_limits_at_start_w": "0, [N/A], 150.00 W;",
    "persistence_mode_at_start": "0, Enabled;",
    "pcie_link_width_at_start": "0, 8;"
  },
  "num_gpu_option": null,
  "num_gpu_note": "ollama decides for itself how many of a model's layers to put on the card. On the pair of 3080s that decision was measurably conservative -- gemma4:26b loaded 75.2% into VRAM and spilled 4.25 GiB to host RAM while about 5 GiB of the two cards' 20 GiB sat unused -- so the pair bench reported `auto` and a forced layer count side by side. On ONE 24 GB card there is no placement decision to make: the expected reading is 100% on the card with no spill. This harness records `options.num_gpu` on every run as `num_gpu_option` and reads the planner's actual placement back from /api/ps, so a spill on a card with room is visible as a finding rather than absorbed into a tok/s figure. A value at or above the model's layer count means every layer. ON THIS CARD THE QUESTION INVERTS. The pair bench forced `num_gpu` UP to recover cards the planner had left half empty; on a 12 GB board the model cannot fit however high the count goes, so forcing it up is not a recovery -- it is either ignored or an out-of-memory refusal, and either is a result. What is worth measuring is whether the planner leaves VRAM on the table on a card it CANNOT fill: so the probe BRACKETS the planner's own choice rather than jumping to the layer count. The planner's num_gpu is read back from the server, the model's own block count is read from /api/show (never typed -- the pair bench knew gemma4:26b had 31 layers because it READ it, and this bench has no receipt for the dense model's), and the probe walks UPWARD from the planner's choice in the steps PREREG SS5.4 registers, stopping at the FIRST refusal and recording it. An out-of-memory answer is written into the result file as that rung's outcome, never as a discarded run.",
  "memory_temp_support": {
    "checked_utc": "2026-09-22T00:23:09Z",
    "paths": {
      "nvidia-smi --query-gpu=temperature.memory": {
        "raw": "0, GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff, N/A",
        "readable": false
      },
      "nvidia-smi -q -d TEMPERATURE": {
        "lines": [
          "GPU Target Temperature                         : 87 C",
          "Memory Current Temp                            : N/A",
          "Memory Max Operating T.Limit Temp              : N/A"
        ],
        "readable": false
      },
      "NVML NVML_FI_DEV_MEMORY_TEMP": {
        "available": true,
        "field_id": 82,
        "cards": {
          "0": {
            "call_rc": 0,
            "field_rc": 3,
            "supported": false,
            "not_supported": true,
            "value_c": null
          }
        }
      }
    },
    "readable": false,
    "verdict": "the memory die's temperature is NOT readable on these cards through any of the three paths asked, so the memory-temperature stop condition could not arm and NO memory temperature is reported anywhere in this bench. The core temperature stop and the driver's own thermal-slowdown reasons are the thermal instrument instead."
  },
  "ups_before_arm": {
    "ups": "pr1500@localhost",
    "read_utc": "2026-09-22T00:23:09Z",
    "available": false,
    "error": "no upsc on this box"
  },
  "runs": [
    {
      "ttft_ms": 745.29,
      "first_token_was_thinking": false,
      "wall_s": 4.4142,
      "eval_count": 256,
      "eval_duration_ns": 3544652000,
      "decode_tok_s": 72.221,
      "prompt_eval_count": 97992,
      "prompt_eval_duration_ns": 75359000,
      "prefill_tok_s": 1300335.726,
      "load_duration_ns": 291418672,
      "total_duration_ns": 4292383422,
      "streamed_chunks": 256,
      "response_chars": 1053,
      "thinking_chars": 0,
      "done_reason": "length",
      "decode_tok_s_wall": 69.502,
      "decode_tok_s_wall_gross": 57.994,
      "num_ctx_option": 131072,
      "power": {
        "idle_baseline": {
          "seconds": 3.0,
          "samples": 6,
          "power_w": {
            "min": 12.1,
            "median": 12.37,
            "max": 12.67,
            "mean": 12.38,
            "n": 6
          },
          "utilization_pct": {
            "min": 0.0,
            "median": 0.0,
            "max": 0.0,
            "mean": 0.0,
            "n": 6
          },
          "memory_used_mib": {
            "min": 19847.0,
            "median": 19847.0,
            "max": 19847.0,
            "mean": 19847.0,
            "n": 6
          },
          "power_w_per_card": {
            "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 12.38
          },
          "memory_used_mib_per_card": {
            "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 19847.0
          },
          "power_w_all_cards": 12.38,
          "ups_whole_box": {
            "load_pct": null,
            "realpower_w": null,
            "realpower_nominal_w": null,
            "status": null,
            "note": "the UPS's reading of EVERYTHING it feeds -- the whole box, not the cards. The card figures beside it are board power from nvidia-smi. Neither is the other."
          }
        },
        "measured_over": "samples where the GPU was busy",
        "busy_samples": 13,
        "mean_w": 100.55,
        "max_w": 124.98,
        "median_w": 124.45,
        "mean_w_whole_window": 84.05,
        "max_w_whole_window": 124.98,
        "limit_w": 150.0,
        "over_idle_w": 88.17,
        "samples": 16,
        "per_gpu": {
          "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
            "index": "0",
            "name": "NVIDIA GeForce RTX 5090 Laptop GPU",
            "power_w": {
              "min": 12.54,
              "median": 124.26,
              "max": 124.98,
              "mean": 84.05,
              "n": 16
            },
            "power_limit_w": 150.0,
            "power_limit_field": "enforced.power.limit",
            "power_limit_enforced_w": 150.0,
            "cap": "shipped",
            "power_posture": "shipped",
            "power_limit_enforced_stats_w": {
              "min": 150.0,
              "median": 150.0,
              "max": 150.0,
              "mean": 150.0,
              "n": 16
            },
            "power_posture_note": "shipped -- nvidia-powerd running and no cap set: the board floats between its power.default_limit and its power.max_limit against the CPU's draw, DURING the scored run. `power_limit_enforced_stats_w` is that limit reduced over this run's own 2 Hz trace and is the only honest cap cell; `power_limit_enforced_w` is the FIRST sample and is kept only so the two can be compared. A single wattage in a cap column for this posture is a fabrication. AND this box's boot-time clock lock (ai-perf.service, -lgc 1200,2550) is IN FORCE, which is this posture's declared condition rather than a confound: it is how the machine boots. The verdict is read FROM THE CARD and lives in the pass's clock-lock receipt. Rows measured here are comparable with this machine's other two postures and with nothing else in the estate -- PREREG.md Amendment 4.",
            "cap_cell": "150 W, flat",
            "cap_cell_note": "the enforced limit read 150 W on every one of 16 samples of this run's 2 Hz trace. Under shipped that is a FINDING, not a range.",
            "memory_used_mib": {
              "min": 19847.0,
              "median": 19847.0,
              "max": 19847.0,
              "mean": 19847.0,
              "n": 16
            },
            "utilization_pct": {
              "min": 0.0,
              "median": 95.0,
              "max": 96.0,
              "mean": 75.4,
              "n": 16
            },
            "temperature_c": {
              "min": 43.0,
              "median": 49.0,
              "max": 49.0,
              "mean": 47.4,
              "n": 16
            },
            "memory_temperature_c": null,
            "fan_pct": null,
            "pcie_width_current": {
              "min": 8.0,
              "median": 8.0,
              "max": 8.0,
              "mean": 8.0,
              "n": 16
            },
            "clocks_sm_mhz": {
              "min": 1192.0,
              "median": 1905.0,
              "max": 2017.0,
              "mean": 1735.0,
              "n": 16
            },
            "clocks_mem_mhz": {
              "min": 810.0,
              "median": 14001.0,
              "max": 14001.0,
              "mean": 11527.7,
              "n": 16
            },
            "throttle_sw_power_cap_fraction": 0.8125,
            "throttle_hw_slowdown_fraction": 0.0,
            "throttle_sw_thermal_fraction": 0.0,
            "throttle_hw_thermal_fraction": 0.0,
            "n": 16,
            "busy": {
              "index": "0",
              "name": "NVIDIA GeForce RTX 5090 Laptop GPU",
              "power_w": {
                "min": 17.55,
                "median": 124.45,
                "max": 124.98,
                "mean": 100.55,
                "n": 13
              },
              "power_limit_w": 150.0,
              "power_limit_field": "enforced.power.limit",
              "power_limit_enforced_w": 150.0,
              "cap": "shipped",
              "power_posture": "shipped",
              "power_limit_enforced_stats_w": {
                "min": 150.0,
                "median": 150.0,
                "max": 150.0,
                "mean": 150.0,
                "n": 13
              },
              "power_posture_note": "shipped -- nvidia-powerd running and no cap set: the board floats between its power.default_limit and its power.max_limit against the CPU's draw, DURING the scored run. `power_limit_enforced_stats_w` is that limit reduced over this run's own 2 Hz trace and is the only honest cap cell; `power_limit_enforced_w` is the FIRST sample and is kept only so the two can be compared. A single wattage in a cap column for this posture is a fabrication. AND this box's boot-time clock lock (ai-perf.service, -lgc 1200,2550) is IN FORCE, which is this posture's declared condition rather than a confound: it is how the machine boots. The verdict is read FROM THE CARD and lives in the pass's clock-lock receipt. Rows measured here are comparable with this machine's other two postures and with nothing else in the estate -- PREREG.md Amendment 4.",
              "cap_cell": "150 W, flat",
              "cap_cell_note": "the enforced limit read 150 W on every one of 13 samples of this run's 2 Hz trace. Under shipped that is a FINDING, not a range.",
              "memory_used_mib": {
                "min": 19847.0,
                "median": 19847.0,
                "max": 19847.0,
                "mean": 19847.0,
                "n": 13
              },
              "utilization_pct": {
                "min": 81.0,
                "median": 95.0,
                "max": 96.0,
                "mean": 92.8,
                "n": 13
              },
              "temperature_c": {
                "min": 45.0,
                "median": 49.0,
                "max": 49.0,
                "mean": 48.5,
                "n": 13
              },
              "memory_temperature_c": null,
              "fan_pct": null,
              "pcie_width_current": {
                "min": 8.0,
                "median": 8.0,
                "max": 8.0,
                "mean": 8.0,
                "n": 13
              },
              "clocks_sm_mhz": {
                "min": 1215.0,
                "median": 1905.0,
                "max": 2017.0,
                "mean": 1860.3,
                "n": 13
              },
              "clocks_mem_mhz": {
                "min": 14001.0,
                "median": 14001.0,
                "max": 14001.0,
                "mean": 14001.0,
                "n": 13
              },
              "throttle_sw_power_cap_fraction": 1.0,
              "throttle_hw_slowdown_fraction": 0.0,
              "throttle_sw_thermal_fraction": 0.0,
              "throttle_hw_thermal_fraction": 0.0,
              "n": 13
            },
            "busy_samples": 13
          }
        }
      },
      "energy_j_per_1k_tokens": 1392.25,
      "power_mean_w_all_cards": 100.55,
      "energy_j_per_1k_tokens_all_cards": 1392.25,
      "watts_per_tok_s": 1.3923,
      "thermal": {
        "rule_version": 2,
        "measured_over": "samples where the GPU was busy (utilisation > 0)",
        "busy_samples": 13,
        "temp_max_c": 49.0,
        "temp_min_c": 45.0,
        "temp_rise_c": 4.0,
        "memory_temp_max_c": null,
        "memory_temp_median_c": null,
        "memory_temp_readable": false,
        "pcie_width_min": 8.0,
        "pcie_width_max": 8.0,
        "pcie_width_median": 8.0,
        "fan_mean_pct": null,
        "fan_max_pct": null,
        "clock_floor_mhz": 1215.0,
        "clock_max_mhz": 2017.0,
        "clock_median_mhz": 1905.0,
        "mem_clock_floor_mhz": 14001.0,
        "mem_clock_median_mhz": 14001.0,
        "mem_clock_max_mhz": 14001.0,
        "mem_clock_held": true,
        "memory_bus_width_bits": 256,
        "memory_technology": "GDDR7",
        "memory_bits_per_clock": null,
        "peak_mem_bandwidth_gbs_at_median_clock": null,
        "peak_mem_bandwidth_gbs_at_floor_clock": null,
        "peak_mem_bandwidth_gbs_at_max_clock": null,
        "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
        "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
        "clock_dropped": true,
        "power_limit_w": 150.0,
        "power_mean_w": 100.55,
        "power_pct_of_limit": 67.0,
        "power_pinned_at_limit": false,
        "sw_power_cap_active_fraction": 1.0,
        "hw_slowdown_active_fraction": 0.0,
        "sw_thermal_active_fraction": 0.0,
        "hw_thermal_active_fraction": 0.0,
        "at_throttling_temperature": false,
        "stop_core_temp_c": 83.0,
        "stop_memory_temp_c": 100.0,
        "hit_core_temp_stop": false,
        "hit_memory_temp_stop": false,
        "throttle_temperature_c": 80.0,
        "temp_rising_within_run": true,
        "verdict": "NEITHER cap: the SM clock varied with draw at 67.0% of the cap and the card at 49 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
      },
      "thermal_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 13,
          "temp_max_c": 49.0,
          "temp_min_c": 45.0,
          "temp_rise_c": 4.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 8.0,
          "pcie_width_max": 8.0,
          "pcie_width_median": 8.0,
          "fan_mean_pct": null,
          "fan_max_pct": null,
          "clock_floor_mhz": 1215.0,
          "clock_max_mhz": 2017.0,
          "clock_median_mhz": 1905.0,
          "mem_clock_floor_mhz": 14001.0,
          "mem_clock_median_mhz": 14001.0,
          "mem_clock_max_mhz": 14001.0,
          "mem_clock_held": true,
          "memory_bus_width_bits": 256,
          "memory_technology": "GDDR7",
          "memory_bits_per_clock": null,
          "peak_mem_bandwidth_gbs_at_median_clock": null,
          "peak_mem_bandwidth_gbs_at_floor_clock": null,
          "peak_mem_bandwidth_gbs_at_max_clock": null,
          "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
          "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
          "clock_dropped": true,
          "power_limit_w": 150.0,
          "power_mean_w": 100.55,
          "power_pct_of_limit": 67.0,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 1.0,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": true,
          "verdict": "NEITHER cap: the SM clock varied with draw at 67.0% of the cap and the card at 49 C. On a memory-bound decode the clock follows the work, and nothing here was limiting it"
        }
      },
      "decode_bandwidth": {
        "model_tag": "gemma4:26b",
        "model_density": "moe",
        "weight_bytes": 17987581215,
        "peak_bandwidth_gbs_at_observed_clock": null,
        "decode_gbs": null,
        "fraction_of_peak_pct": null,
        "basis": null,
        "refused_reason": "gemma4:26b is moe: this bench has no receipt for how many bytes a token actually reads, so no bytes-per-second figure is computed. An MoE reads a subset of its weights per token and the subset is not measured here."
      },
      "memory_used_mib_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 19847.0
      },
      "pcie_width_under_load": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "min": 8.0,
          "median": 8.0,
          "max": 8.0,
          "mean": 8.0,
          "n": 16
        }
      },
      "ups_during_run": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      },
      "stop_conditions_hit": [],
      "run": 1
    },
    {
      "ttft_ms": 672.85,
      "first_token_was_thinking": false,
      "wall_s": 4.3358,
      "eval_count": 256,
      "eval_duration_ns": 3537653000,
      "decode_tok_s": 72.364,
      "prompt_eval_count": 97992,
      "prompt_eval_duration_ns": 72450000,
      "prefill_tok_s": 1352546.584,
      "load_duration_ns": 258709219,
      "total_duration_ns": 4213351623,
      "streamed_chunks": 256,
      "response_chars": 1053,
      "thinking_chars": 0,
      "done_reason": "length",
      "decode_tok_s_wall": 69.616,
      "decode_tok_s_wall_gross": 59.043,
      "num_ctx_option": 131072,
      "power": {
        "idle_baseline": {
          "seconds": 3.0,
          "samples": 6,
          "power_w": {
            "min": 11.81,
            "median": 11.84,
            "max": 11.9,
            "mean": 11.85,
            "n": 6
          },
          "utilization_pct": {
            "min": 0.0,
            "median": 0.0,
            "max": 0.0,
            "mean": 0.0,
            "n": 6
          },
          "memory_used_mib": {
            "min": 19847.0,
            "median": 19847.0,
            "max": 19847.0,
            "mean": 19847.0,
            "n": 6
          },
          "power_w_per_card": {
            "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 11.85
          },
          "memory_used_mib_per_card": {
            "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 19847.0
          },
          "power_w_all_cards": 11.85,
          "ups_whole_box": {
            "load_pct": null,
            "realpower_w": null,
            "realpower_nominal_w": null,
            "status": null,
            "note": "the UPS's reading of EVERYTHING it feeds -- the whole box, not the cards. The card figures beside it are board power from nvidia-smi. Neither is the other."
          }
        },
        "measured_over": "samples where the GPU was busy",
        "busy_samples": 12,
        "mean_w": 108.96,
        "max_w": 124.4,
        "median_w": 123.97,
        "mean_w_whole_window": 85.31,
        "max_w_whole_window": 124.4,
        "limit_w": 150.0,
        "over_idle_w": 97.11,
        "samples": 16,
        "per_gpu": {
          "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
            "index": "0",
            "name": "NVIDIA GeForce RTX 5090 Laptop GPU",
            "power_w": {
              "min": 11.73,
              "median": 123.84,
              "max": 124.4,
              "mean": 85.31,
              "n": 16
            },
            "power_limit_w": 150.0,
            "power_limit_field": "enforced.power.limit",
            "power_limit_enforced_w": 150.0,
            "cap": "shipped",
            "power_posture": "shipped",
            "power_limit_enforced_stats_w": {
              "min": 150.0,
              "median": 150.0,
              "max": 150.0,
              "mean": 150.0,
              "n": 16
            },
            "power_posture_note": "shipped -- nvidia-powerd running and no cap set: the board floats between its power.default_limit and its power.max_limit against the CPU's draw, DURING the scored run. `power_limit_enforced_stats_w` is that limit reduced over this run's own 2 Hz trace and is the only honest cap cell; `power_limit_enforced_w` is the FIRST sample and is kept only so the two can be compared. A single wattage in a cap column for this posture is a fabrication. AND this box's boot-time clock lock (ai-perf.service, -lgc 1200,2550) is IN FORCE, which is this posture's declared condition rather than a confound: it is how the machine boots. The verdict is read FROM THE CARD and lives in the pass's clock-lock receipt. Rows measured here are comparable with this machine's other two postures and with nothing else in the estate -- PREREG.md Amendment 4.",
            "cap_cell": "150 W, flat",
            "cap_cell_note": "the enforced limit read 150 W on every one of 16 samples of this run's 2 Hz trace. Under shipped that is a FINDING, not a range.",
            "memory_used_mib": {
              "min": 19847.0,
              "median": 19847.0,
              "max": 19847.0,
              "mean": 19847.0,
              "n": 16
            },
            "utilization_pct": {
              "min": 0.0,
              "median": 95.0,
              "max": 95.0,
              "mean": 71.2,
              "n": 16
            },
            "temperature_c": {
              "min": 40.0,
              "median": 46.0,
              "max": 47.0,
              "mean": 44.8,
              "n": 16
            },
            "memory_temperature_c": null,
            "fan_pct": null,
            "pcie_width_current": {
              "min": 8.0,
              "median": 8.0,
              "max": 8.0,
              "mean": 8.0,
              "n": 16
            },
            "clocks_sm_mhz": {
              "min": 1192.0,
              "median": 1905.0,
              "max": 1920.0,
              "mean": 1760.8,
              "n": 16
            },
            "clocks_mem_mhz": {
              "min": 810.0,
              "median": 14001.0,
              "max": 14001.0,
              "mean": 11527.7,
              "n": 16
            },
            "throttle_sw_power_cap_fraction": 0.75,
            "throttle_hw_slowdown_fraction": 0.0,
            "throttle_sw_thermal_fraction": 0.0,
            "throttle_hw_thermal_fraction": 0.0,
            "n": 16,
            "busy": {
              "index": "0",
              "name": "NVIDIA GeForce RTX 5090 Laptop GPU",
              "power_w": {
                "min": 53.94,
                "median": 123.97,
                "max": 124.4,
                "mean": 108.96,
                "n": 12
              },
              "power_limit_w": 150.0,
              "power_limit_field": "enforced.power.limit",
              "power_limit_enforced_w": 150.0,
              "cap": "shipped",
              "power_posture": "shipped",
              "power_limit_enforced_stats_w": {
                "min": 150.0,
                "median": 150.0,
                "max": 150.0,
                "mean": 150.0,
                "n": 12
              },
              "power_posture_note": "shipped -- nvidia-powerd running and no cap set: the board floats between its power.default_limit and its power.max_limit against the CPU's draw, DURING the scored run. `power_limit_enforced_stats_w` is that limit reduced over this run's own 2 Hz trace and is the only honest cap cell; `power_limit_enforced_w` is the FIRST sample and is kept only so the two can be compared. A single wattage in a cap column for this posture is a fabrication. AND this box's boot-time clock lock (ai-perf.service, -lgc 1200,2550) is IN FORCE, which is this posture's declared condition rather than a confound: it is how the machine boots. The verdict is read FROM THE CARD and lives in the pass's clock-lock receipt. Rows measured here are comparable with this machine's other two postures and with nothing else in the estate -- PREREG.md Amendment 4.",
              "cap_cell": "150 W, flat",
              "cap_cell_note": "the enforced limit read 150 W on every one of 12 samples of this run's 2 Hz trace. Under shipped that is a FINDING, not a range.",
              "memory_used_mib": {
                "min": 19847.0,
                "median": 19847.0,
                "max": 19847.0,
                "mean": 19847.0,
                "n": 12
              },
              "utilization_pct": {
                "min": 95.0,
                "median": 95.0,
                "max": 95.0,
                "mean": 95.0,
                "n": 12
              },
              "temperature_c": {
                "min": 45.0,
                "median": 46.0,
                "max": 47.0,
                "mean": 46.0,
                "n": 12
              },
              "memory_temperature_c": null,
              "fan_pct": null,
              "pcie_width_current": {
                "min": 8.0,
                "median": 8.0,
                "max": 8.0,
                "mean": 8.0,
                "n": 12
              },
              "clocks_sm_mhz": {
                "min": 1897.0,
                "median": 1905.0,
                "max": 1920.0,
                "mean": 1907.9,
                "n": 12
              },
              "clocks_mem_mhz": {
                "min": 14001.0,
                "median": 14001.0,
                "max": 14001.0,
                "mean": 14001.0,
                "n": 12
              },
              "throttle_sw_power_cap_fraction": 1.0,
              "throttle_hw_slowdown_fraction": 0.0,
              "throttle_sw_thermal_fraction": 0.0,
              "throttle_hw_thermal_fraction": 0.0,
              "n": 12
            },
            "busy_samples": 12
          }
        }
      },
      "energy_j_per_1k_tokens": 1505.71,
      "power_mean_w_all_cards": 108.96,
      "energy_j_per_1k_tokens_all_cards": 1505.71,
      "watts_per_tok_s": 1.5057,
      "thermal": {
        "rule_version": 2,
        "measured_over": "samples where the GPU was busy (utilisation > 0)",
        "busy_samples": 12,
        "temp_max_c": 47.0,
        "temp_min_c": 45.0,
        "temp_rise_c": 2.0,
        "memory_temp_max_c": null,
        "memory_temp_median_c": null,
        "memory_temp_readable": false,
        "pcie_width_min": 8.0,
        "pcie_width_max": 8.0,
        "pcie_width_median": 8.0,
        "fan_mean_pct": null,
        "fan_max_pct": null,
        "clock_floor_mhz": 1897.0,
        "clock_max_mhz": 1920.0,
        "clock_median_mhz": 1905.0,
        "mem_clock_floor_mhz": 14001.0,
        "mem_clock_median_mhz": 14001.0,
        "mem_clock_max_mhz": 14001.0,
        "mem_clock_held": true,
        "memory_bus_width_bits": 256,
        "memory_technology": "GDDR7",
        "memory_bits_per_clock": null,
        "peak_mem_bandwidth_gbs_at_median_clock": null,
        "peak_mem_bandwidth_gbs_at_floor_clock": null,
        "peak_mem_bandwidth_gbs_at_max_clock": null,
        "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
        "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
        "clock_dropped": false,
        "power_limit_w": 150.0,
        "power_mean_w": 108.96,
        "power_pct_of_limit": 72.6,
        "power_pinned_at_limit": false,
        "sw_power_cap_active_fraction": 1.0,
        "hw_slowdown_active_fraction": 0.0,
        "sw_thermal_active_fraction": 0.0,
        "hw_thermal_active_fraction": 0.0,
        "at_throttling_temperature": false,
        "stop_core_temp_c": 83.0,
        "stop_memory_temp_c": 100.0,
        "hit_core_temp_stop": false,
        "hit_memory_temp_stop": false,
        "throttle_temperature_c": 80.0,
        "temp_rising_within_run": false,
        "verdict": "no clock drop during decode"
      },
      "thermal_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 12,
          "temp_max_c": 47.0,
          "temp_min_c": 45.0,
          "temp_rise_c": 2.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 8.0,
          "pcie_width_max": 8.0,
          "pcie_width_median": 8.0,
          "fan_mean_pct": null,
          "fan_max_pct": null,
          "clock_floor_mhz": 1897.0,
          "clock_max_mhz": 1920.0,
          "clock_median_mhz": 1905.0,
          "mem_clock_floor_mhz": 14001.0,
          "mem_clock_median_mhz": 14001.0,
          "mem_clock_max_mhz": 14001.0,
          "mem_clock_held": true,
          "memory_bus_width_bits": 256,
          "memory_technology": "GDDR7",
          "memory_bits_per_clock": null,
          "peak_mem_bandwidth_gbs_at_median_clock": null,
          "peak_mem_bandwidth_gbs_at_floor_clock": null,
          "peak_mem_bandwidth_gbs_at_max_clock": null,
          "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
          "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
          "clock_dropped": false,
          "power_limit_w": 150.0,
          "power_mean_w": 108.96,
          "power_pct_of_limit": 72.6,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 1.0,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": false,
          "verdict": "no clock drop during decode"
        }
      },
      "decode_bandwidth": {
        "model_tag": "gemma4:26b",
        "model_density": "moe",
        "weight_bytes": 17987581215,
        "peak_bandwidth_gbs_at_observed_clock": null,
        "decode_gbs": null,
        "fraction_of_peak_pct": null,
        "basis": null,
        "refused_reason": "gemma4:26b is moe: this bench has no receipt for how many bytes a token actually reads, so no bytes-per-second figure is computed. An MoE reads a subset of its weights per token and the subset is not measured here."
      },
      "memory_used_mib_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 19847.0
      },
      "pcie_width_under_load": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "min": 8.0,
          "median": 8.0,
          "max": 8.0,
          "mean": 8.0,
          "n": 16
        }
      },
      "ups_during_run": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      },
      "stop_conditions_hit": [],
      "run": 2
    },
    {
      "ttft_ms": 678.53,
      "first_token_was_thinking": false,
      "wall_s": 4.3494,
      "eval_count": 256,
      "eval_duration_ns": 3536408000,
      "decode_tok_s": 72.39,
      "prompt_eval_count": 97992,
      "prompt_eval_duration_ns": 74039000,
      "prefill_tok_s": 1323518.686,
      "load_duration_ns": 265397359,
      "total_duration_ns": 4216843519,
      "streamed_chunks": 256,
      "response_chars": 1053,
      "thinking_chars": 0,
      "done_reason": "length",
      "decode_tok_s_wall": 69.466,
      "decode_tok_s_wall_gross": 58.859,
      "num_ctx_option": 131072,
      "power": {
        "idle_baseline": {
          "seconds": 3.0,
          "samples": 6,
          "power_w": {
            "min": 11.82,
            "median": 11.84,
            "max": 11.97,
            "mean": 11.88,
            "n": 6
          },
          "utilization_pct": {
            "min": 0.0,
            "median": 0.0,
            "max": 0.0,
            "mean": 0.0,
            "n": 6
          },
          "memory_used_mib": {
            "min": 19847.0,
            "median": 19847.0,
            "max": 19847.0,
            "mean": 19847.0,
            "n": 6
          },
          "power_w_per_card": {
            "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 11.88
          },
          "memory_used_mib_per_card": {
            "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 19847.0
          },
          "power_w_all_cards": 11.88,
          "ups_whole_box": {
            "load_pct": null,
            "realpower_w": null,
            "realpower_nominal_w": null,
            "status": null,
            "note": "the UPS's reading of EVERYTHING it feeds -- the whole box, not the cards. The card figures beside it are board power from nvidia-smi. Neither is the other."
          }
        },
        "measured_over": "samples where the GPU was busy",
        "busy_samples": 12,
        "mean_w": 112.83,
        "max_w": 124.66,
        "median_w": 124.11,
        "mean_w_whole_window": 88.06,
        "max_w_whole_window": 124.66,
        "limit_w": 150.0,
        "over_idle_w": 100.95,
        "samples": 16,
        "per_gpu": {
          "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
            "index": "0",
            "name": "NVIDIA GeForce RTX 5090 Laptop GPU",
            "power_w": {
              "min": 11.79,
              "median": 123.9,
              "max": 124.66,
              "mean": 88.06,
              "n": 16
            },
            "power_limit_w": 150.0,
            "power_limit_field": "enforced.power.limit",
            "power_limit_enforced_w": 150.0,
            "cap": "shipped",
            "power_posture": "shipped",
            "power_limit_enforced_stats_w": {
              "min": 150.0,
              "median": 150.0,
              "max": 150.0,
              "mean": 150.0,
              "n": 16
            },
            "power_posture_note": "shipped -- nvidia-powerd running and no cap set: the board floats between its power.default_limit and its power.max_limit against the CPU's draw, DURING the scored run. `power_limit_enforced_stats_w` is that limit reduced over this run's own 2 Hz trace and is the only honest cap cell; `power_limit_enforced_w` is the FIRST sample and is kept only so the two can be compared. A single wattage in a cap column for this posture is a fabrication. AND this box's boot-time clock lock (ai-perf.service, -lgc 1200,2550) is IN FORCE, which is this posture's declared condition rather than a confound: it is how the machine boots. The verdict is read FROM THE CARD and lives in the pass's clock-lock receipt. Rows measured here are comparable with this machine's other two postures and with nothing else in the estate -- PREREG.md Amendment 4.",
            "cap_cell": "150 W, flat",
            "cap_cell_note": "the enforced limit read 150 W on every one of 16 samples of this run's 2 Hz trace. Under shipped that is a FINDING, not a range.",
            "memory_used_mib": {
              "min": 19847.0,
              "median": 19847.0,
              "max": 19847.0,
              "mean": 19847.0,
              "n": 16
            },
            "utilization_pct": {
              "min": 0.0,
              "median": 95.0,
              "max": 96.0,
              "mean": 71.5,
              "n": 16
            },
            "temperature_c": {
              "min": 38.0,
              "median": 44.0,
              "max": 45.0,
              "mean": 43.1,
              "n": 16
            },
            "memory_temperature_c": null,
            "fan_pct": null,
            "pcie_width_current": {
              "min": 8.0,
              "median": 8.0,
              "max": 8.0,
              "mean": 8.0,
              "n": 16
            },
            "clocks_sm_mhz": {
              "min": 1192.0,
              "median": 1908.5,
              "max": 1957.0,
              "mean": 1744.3,
              "n": 16
            },
            "clocks_mem_mhz": {
              "min": 810.0,
              "median": 14001.0,
              "max": 14001.0,
              "mean": 11527.7,
              "n": 16
            },
            "throttle_sw_power_cap_fraction": 0.75,
            "throttle_hw_slowdown_fraction": 0.0,
            "throttle_sw_thermal_fraction": 0.0,
            "throttle_hw_thermal_fraction": 0.0,
            "n": 16,
            "busy": {
              "index": "0",
              "name": "NVIDIA GeForce RTX 5090 Laptop GPU",
              "power_w": {
                "min": 49.53,
                "median": 124.11,
                "max": 124.66,
                "mean": 112.83,
                "n": 12
              },
              "power_limit_w": 150.0,
              "power_limit_field": "enforced.power.limit",
              "power_limit_enforced_w": 150.0,
              "cap": "shipped",
              "power_posture": "shipped",
              "power_limit_enforced_stats_w": {
                "min": 150.0,
                "median": 150.0,
                "max": 150.0,
                "mean": 150.0,
                "n": 12
              },
              "power_posture_note": "shipped -- nvidia-powerd running and no cap set: the board floats between its power.default_limit and its power.max_limit against the CPU's draw, DURING the scored run. `power_limit_enforced_stats_w` is that limit reduced over this run's own 2 Hz trace and is the only honest cap cell; `power_limit_enforced_w` is the FIRST sample and is kept only so the two can be compared. A single wattage in a cap column for this posture is a fabrication. AND this box's boot-time clock lock (ai-perf.service, -lgc 1200,2550) is IN FORCE, which is this posture's declared condition rather than a confound: it is how the machine boots. The verdict is read FROM THE CARD and lives in the pass's clock-lock receipt. Rows measured here are comparable with this machine's other two postures and with nothing else in the estate -- PREREG.md Amendment 4.",
              "cap_cell": "150 W, flat",
              "cap_cell_note": "the enforced limit read 150 W on every one of 12 samples of this run's 2 Hz trace. Under shipped that is a FINDING, not a range.",
              "memory_used_mib": {
                "min": 19847.0,
                "median": 19847.0,
                "max": 19847.0,
                "mean": 19847.0,
                "n": 12
              },
              "utilization_pct": {
                "min": 94.0,
                "median": 95.5,
                "max": 96.0,
                "mean": 95.3,
                "n": 12
              },
              "temperature_c": {
                "min": 43.0,
                "median": 44.5,
                "max": 45.0,
                "mean": 44.4,
                "n": 12
              },
              "memory_temperature_c": null,
              "fan_pct": null,
              "pcie_width_current": {
                "min": 8.0,
                "median": 8.0,
                "max": 8.0,
                "mean": 8.0,
                "n": 12
              },
              "clocks_sm_mhz": {
                "min": 1897.0,
                "median": 1912.0,
                "max": 1957.0,
                "mean": 1912.8,
                "n": 12
              },
              "clocks_mem_mhz": {
                "min": 14001.0,
                "median": 14001.0,
                "max": 14001.0,
                "mean": 14001.0,
                "n": 12
              },
              "throttle_sw_power_cap_fraction": 1.0,
              "throttle_hw_slowdown_fraction": 0.0,
              "throttle_sw_thermal_fraction": 0.0,
              "throttle_hw_thermal_fraction": 0.0,
              "n": 12
            },
            "busy_samples": 12
          }
        }
      },
      "energy_j_per_1k_tokens": 1558.64,
      "power_mean_w_all_cards": 112.83,
      "energy_j_per_1k_tokens_all_cards": 1558.64,
      "watts_per_tok_s": 1.5586,
      "thermal": {
        "rule_version": 2,
        "measured_over": "samples where the GPU was busy (utilisation > 0)",
        "busy_samples": 12,
        "temp_max_c": 45.0,
        "temp_min_c": 43.0,
        "temp_rise_c": 2.0,
        "memory_temp_max_c": null,
        "memory_temp_median_c": null,
        "memory_temp_readable": false,
        "pcie_width_min": 8.0,
        "pcie_width_max": 8.0,
        "pcie_width_median": 8.0,
        "fan_mean_pct": null,
        "fan_max_pct": null,
        "clock_floor_mhz": 1897.0,
        "clock_max_mhz": 1957.0,
        "clock_median_mhz": 1912.0,
        "mem_clock_floor_mhz": 14001.0,
        "mem_clock_median_mhz": 14001.0,
        "mem_clock_max_mhz": 14001.0,
        "mem_clock_held": true,
        "memory_bus_width_bits": 256,
        "memory_technology": "GDDR7",
        "memory_bits_per_clock": null,
        "peak_mem_bandwidth_gbs_at_median_clock": null,
        "peak_mem_bandwidth_gbs_at_floor_clock": null,
        "peak_mem_bandwidth_gbs_at_max_clock": null,
        "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
        "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
        "clock_dropped": false,
        "power_limit_w": 150.0,
        "power_mean_w": 112.83,
        "power_pct_of_limit": 75.2,
        "power_pinned_at_limit": false,
        "sw_power_cap_active_fraction": 1.0,
        "hw_slowdown_active_fraction": 0.0,
        "sw_thermal_active_fraction": 0.0,
        "hw_thermal_active_fraction": 0.0,
        "at_throttling_temperature": false,
        "stop_core_temp_c": 83.0,
        "stop_memory_temp_c": 100.0,
        "hit_core_temp_stop": false,
        "hit_memory_temp_stop": false,
        "throttle_temperature_c": 80.0,
        "temp_rising_within_run": false,
        "verdict": "no clock drop during decode"
      },
      "thermal_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "rule_version": 2,
          "measured_over": "samples where the GPU was busy (utilisation > 0)",
          "busy_samples": 12,
          "temp_max_c": 45.0,
          "temp_min_c": 43.0,
          "temp_rise_c": 2.0,
          "memory_temp_max_c": null,
          "memory_temp_median_c": null,
          "memory_temp_readable": false,
          "pcie_width_min": 8.0,
          "pcie_width_max": 8.0,
          "pcie_width_median": 8.0,
          "fan_mean_pct": null,
          "fan_max_pct": null,
          "clock_floor_mhz": 1897.0,
          "clock_max_mhz": 1957.0,
          "clock_median_mhz": 1912.0,
          "mem_clock_floor_mhz": 14001.0,
          "mem_clock_median_mhz": 14001.0,
          "mem_clock_max_mhz": 14001.0,
          "mem_clock_held": true,
          "memory_bus_width_bits": 256,
          "memory_technology": "GDDR7",
          "memory_bits_per_clock": null,
          "peak_mem_bandwidth_gbs_at_median_clock": null,
          "peak_mem_bandwidth_gbs_at_floor_clock": null,
          "peak_mem_bandwidth_gbs_at_max_clock": null,
          "peak_mem_bandwidth_withheld_because": "this harness's peak-bandwidth formula assumes two bits per memory clock per pin, which is exact for GDDR6 and GDDR6X and is checked against three published bandwidths. It has no receipt for the bits-per-clock of this board's memory, so the DERIVED bytes-per-second ceiling is withheld rather than computed from an unverified constant. The memory clock itself is a reading and is published unchanged.",
          "peak_mem_bandwidth_formula": "GB/s = clocks.mem MHz x 1e6 x <bits-per-clock for THIS board's memory> x (bus bits / 8) / 1e9; a CEILING at the clock that was observed, never an achieved rate. The bits-per-clock is registered per board and is NOT defaulted.",
          "clock_dropped": false,
          "power_limit_w": 150.0,
          "power_mean_w": 112.83,
          "power_pct_of_limit": 75.2,
          "power_pinned_at_limit": false,
          "sw_power_cap_active_fraction": 1.0,
          "hw_slowdown_active_fraction": 0.0,
          "sw_thermal_active_fraction": 0.0,
          "hw_thermal_active_fraction": 0.0,
          "at_throttling_temperature": false,
          "stop_core_temp_c": 83.0,
          "stop_memory_temp_c": 100.0,
          "hit_core_temp_stop": false,
          "hit_memory_temp_stop": false,
          "throttle_temperature_c": 80.0,
          "temp_rising_within_run": false,
          "verdict": "no clock drop during decode"
        }
      },
      "decode_bandwidth": {
        "model_tag": "gemma4:26b",
        "model_density": "moe",
        "weight_bytes": 17987581215,
        "peak_bandwidth_gbs_at_observed_clock": null,
        "decode_gbs": null,
        "fraction_of_peak_pct": null,
        "basis": null,
        "refused_reason": "gemma4:26b is moe: this bench has no receipt for how many bytes a token actually reads, so no bytes-per-second figure is computed. An MoE reads a subset of its weights per token and the subset is not measured here."
      },
      "memory_used_mib_per_card": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 19847.0
      },
      "pcie_width_under_load": {
        "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": {
          "min": 8.0,
          "median": 8.0,
          "max": 8.0,
          "mean": 8.0,
          "n": 16
        }
      },
      "ups_during_run": {
        "ups": "pr1500@localhost",
        "samples": 0,
        "load_pct": null,
        "whole_box_realpower_w": null,
        "whole_box_note": "the UPS's reading of everything it feeds, not the cards; nvidia-smi's per-card watts are a different measurement of a smaller thing",
        "stop_threshold_pct": 80.0,
        "breached": false
      },
      "stop_conditions_hit": [],
      "run": 3
    }
  ],
  "errors": [],
  "model_store_size_bytes": 17987581215,
  "model_store_size_gib": 16.75,
  "ps_after_load": [
    {
      "name": "gemma4:26b",
      "model": "gemma4:26b",
      "size": 17826494544,
      "digest": "5571076f3d70050487b26b341705799e0ab29b808164f90d20d4cf84f699d251",
      "details": {
        "parent_model": "",
        "format": "gguf",
        "family": "gemma4",
        "families": [
          "gemma4"
        ],
        "parameter_size": "25.8B",
        "quantization_level": "Q4_K_M"
      },
      "expires_at": "2026-09-22T00:33:13.938046291Z",
      "size_vram": 17826494544,
      "context_length": 131072
    }
  ],
  "size_total": 17826494544,
  "size_vram": 17826494544,
  "context_length_loaded": 131072,
  "fits_fully_in_vram": true,
  "fit_verdict": "fits: fully resident on the card",
  "memory_used_mib_per_card": {
    "GPU-edff232c-7dbf-2bac-07fb-921a7c9eecff": 19845.0
  },
  "fill": {
    "target_tokens": 98304,
    "base_prompt_tokens": 536,
    "num_ctx": 131072,
    "filler_note": "The filled-window prompt is the series' frozen prompt REPEATED, each copy preceded by a numbered header line, until the target token count is reached. It is therefore highly repetitive text, and a real document of the same length may prefill faster or slower: attention over repeated spans is not attention over varied ones, and a KV cache built from repetition is not a harder or an easier case in any direction this bench measured. No estate corpus text, no visitor text and no private document is used or could be: the filler is built inside the harness from one public file whose sha256 is checked first.",
    "rounds": [
      {
        "round": 1,
        "copies": 183,
        "measured_tokens": 97992,
        "verdict": "at target"
      }
    ],
    "copies": 183,
    "measured_tokens": 97992,
    "prompt_chars": 388948,
    "fraction_of_target": 0.9968
  },
  "prompt_sha256": "f5461ceaa65b645619007d9c3b6f7004b2191c9c218eeb07c1f9d146cdbf89ec",
  "prompt_sha256_note": "the digest of the FILLED prompt actually sent -- built inside the harness from the frozen prompt whose own digest is base_prompt_sha256",
  "prompt_chars": 388948,
  "measured_prompt_tokens": 97992,
  "window_occupancy": 0.7476,
  "cold_prefill": {
    "prompt_eval_count": 98025,
    "prompt_eval_duration_ns": 37547511000,
    "prefill_tok_s": 2610.692,
    "ttft_ms": 38696.9,
    "note": "ONE pass with a unique header prepended, so the KV prefix cache MISSES and the server really reads the whole prompt. This is the prefill rate; the figure in the scored runs below is a cached-prefix number and is labelled as one."
  },
  "prefill_note": "TWO prefill figures are reported and they answer different questions. `cold_prefill` is one pass with a unique header prepended so the prefix cache misses -- the rate at which this pair actually reads a prompt of this length. The scored runs' `prefill_tok_s` all send the SAME prompt, so from the second request the prefix is cached and the figure is a cache hit, not a read. The scored runs' TTFT and decode ARE meaningful: they are what a caller sees on a repeat query at this depth.",
  "warmup_discarded": {
    "ttft_ms": 1387.85,
    "first_token_was_thinking": false,
    "wall_s": 5.0702,
    "eval_count": 256,
    "eval_duration_ns": 3547111000,
    "decode_tok_s": 72.171,
    "prompt_eval_count": 97992,
    "prompt_eval_duration_ns": 120492000,
    "prefill_tok_s": 813265.611,
    "load_duration_ns": 278792302,
    "total_duration_ns": 4937065999,
    "streamed_chunks": 256,
    "response_chars": 1068,
    "thinking_chars": 0,
    "done_reason": "length",
    "decode_tok_s_wall": 69.25,
    "decode_tok_s_wall_gross": 50.492,
    "num_ctx_option": 131072
  },
  "peak_runner_rss_gib": 3.41,
  "spread_gate": {
    "limit": 0.15,
    "values": [
      72.221,
      72.364,
      72.39
    ],
    "spread_fraction": 0.0023,
    "verdict": "spread 0.2% of the median within the 15% gate"
  },
  "voided_by_gate": false,
  "derived": {
    "prompt_eval_count": {
      "min": 97992,
      "median": 97992,
      "max": 97992,
      "mean": 97992.0,
      "n": 3
    },
    "prefill_tok_s": {
      "min": 1300335.726,
      "median": 1323518.686,
      "max": 1352546.584,
      "mean": 1325466.999,
      "n": 3
    },
    "ttft_ms": {
      "min": 672.85,
      "median": 678.53,
      "max": 745.29,
      "mean": 698.89,
      "n": 3
    },
    "decode_tok_s": {
      "min": 72.221,
      "median": 72.364,
      "max": 72.39,
      "mean": 72.325,
      "n": 3
    },
    "decode_tok_s_wall": {
      "min": 69.466,
      "median": 69.502,
      "max": 69.616,
      "mean": 69.528,
      "n": 3
    },
    "wall_s": {
      "min": 4.3358,
      "median": 4.3494,
      "max": 4.4142,
      "mean": 4.3665,
      "n": 3
    },
    "power_mean_w": {
      "min": 100.55,
      "median": 108.96,
      "max": 112.83,
      "mean": 107.45,
      "n": 3
    },
    "power_mean_w_all_cards": {
      "min": 100.55,
      "median": 108.96,
      "max": 112.83,
      "mean": 107.45,
      "n": 3
    },
    "energy_j_per_1k_tokens": {
      "min": 1392.25,
      "median": 1505.71,
      "max": 1558.64,
      "mean": 1485.53,
      "n": 3
    },
    "energy_j_per_1k_tokens_all_cards": {
      "min": 1392.25,
      "median": 1505.71,
      "max": 1558.64,
      "mean": 1485.53,
      "n": 3
    },
    "temp_max_c": {
      "min": 45.0,
      "median": 47.0,
      "max": 49.0,
      "mean": 47.0,
      "n": 3
    }
  },
  "ups_after_arm": {
    "ups": "pr1500@localhost",
    "read_utc": "2026-09-22T00:25:58Z",
    "available": false,
    "error": "no upsc on this box"
  },
  "finished_utc": "2026-09-22T00:25:58Z",
  "prompt_path": "/workshop/bench-laptop-5090-2026-09-21/harness-llm/inputs/frozen-p512.txt"
}