{
  "schema_version": 1,
  "measurement_kind": "engine_microbenchmark",
  "publisher": "GPU Server Hub / Bestconnect",
  "test_date": "2026-09-26",
  "hardware": [
    {
      "gpu": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
      "gpu_reported_memory_mib": 97887.0,
      "gpu_default_limit_w": 300.0,
      "driver": "595.91.07",
      "cpu": [
        "AMD EPYC 4565P 16-Core Processor"
      ],
      "logical_cpus": 32,
      "linux_visible_ram_bytes": 100271353856,
      "physical_system_ram_gb": 96,
      "planned_system_ram_gb_not_measured": 192,
      "kernel": "6.12.0-211.56.1.el10_2.0.1.x86_64"
    }
  ],
  "limitations": [
    "Measured physical system RAM is 96 GB; the planned/offered 192 GB configuration was not measured.",
    "Results compare particular models and settings, not general answer quality or a controlled effect of parameter count.",
    "Engine throughput excludes tokenization/sampling; HTTP latency is loopback-only and excludes WAN.",
    "Telemetry samples can miss peaks between samples; board power is not whole-server AC power."
  ],
  "retained_failed_runs": [
    {
      "run": "20260926T075146Z-server-small-9451dffe",
      "model": "Qwen3-8B",
      "classification": "incomplete_stream",
      "record_sha256": "09d1b2bc405572e0da5f76eef7980eb2bf8bab64bf64b90a24707a2fa03fdd2c",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 48.49844729801407,
      "completed_case_groups": 3,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability."
    },
    {
      "run": "20260926T080932Z-server-small-317a8559",
      "model": "Qwen3-8B",
      "classification": "thermal_safety_guard",
      "record_sha256": "f5d30ca1a9567add927c6127dfdac70f74e816f09c45bc48c4b3f20ef186e4c2",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 70.75511266198009,
      "completed_case_groups": 4,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 7.0,
        "power.draw": 299.93
      },
      "guard_recorded_at": "2026-09-26T08:10:45.553634+00:00"
    },
    {
      "run": "20260926T084305Z-server-medium-36eb432a",
      "model": "Qwen3-32B",
      "classification": "thermal_safety_guard",
      "record_sha256": "b88f8c537500b2d04e04a9bcf069bc1eb4a747fcb62601a5f446ada896e0b161",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 111.19253667499288,
      "completed_case_groups": 3,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 7.0,
        "power.draw": 300.01,
        "fan.speed": 56.0,
        "clocks_event_reasons.sw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.hw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.sw_power_cap": "Active"
      },
      "guard_recorded_at": "2026-09-26T08:45:01.234302+00:00"
    },
    {
      "run": "20260926T085119Z-large-2d370bbb",
      "model": "Qwen2.5-72B-Instruct",
      "classification": "thermal_safety_guard",
      "record_sha256": "5b615f2ac8a9b6db53e38439a49e329997adc9b421eb650096f02580f9dd614b",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": null,
      "completed_case_groups": 0,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "completed_engine_cases": [
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 2048
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 2048
        },
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 8192
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 8192
        }
      ]
    },
    {
      "run": "20260926T091352Z-server-large-71da7c9f",
      "model": "Qwen2.5-72B-Instruct",
      "classification": "thermal_safety_guard",
      "record_sha256": "459962eeebeb8dff83f7e9ad34c145c73513d5cf67f3d05b3b14b60fb42326b0",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 118.36200489901239,
      "completed_case_groups": 1,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 8.0,
        "power.draw": 300.0,
        "fan.speed": 56.0,
        "clocks_event_reasons.sw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.hw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.sw_power_cap": "Active"
      },
      "guard_recorded_at": "2026-09-26T09:15:58.169964+00:00"
    },
    {
      "run": "20260926T092242Z-server-medium-b5a86db7",
      "model": "Qwen3-32B",
      "classification": "thermal_safety_guard",
      "record_sha256": "239fdab8b1038912667720835c50e8d8cfba223d08cbb326fa4d0d706462fc6b",
      "attempted_duration_seconds": 14400,
      "elapsed_campaign_seconds": 105.10982441101805,
      "completed_case_groups": 0,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 7.0,
        "power.draw": 300.01,
        "fan.speed": 56.0,
        "clocks_event_reasons.sw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.hw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.sw_power_cap": "Active"
      },
      "guard_recorded_at": "2026-09-26T09:24:31.801668+00:00"
    }
  ],
  "runs": [
    {
      "run": "20260926T075005Z-small-92ae4526",
      "started_at": "2026-09-26T07:50:05.786311+00:00",
      "finished_at": "2026-09-26T07:51:04.544041+00:00",
      "record_sha256": "219d1a325ad38d7343166aa158f9bc6398af489795b5bc7e86eee573e99b6070",
      "model": {
        "name": "Qwen3-8B",
        "repository": "Qwen/Qwen3-8B-GGUF",
        "revision": "7c41481f57cb95916b40956ab2f0b139b296d974",
        "license": "apache-2.0",
        "quantization": "Q4_K_M",
        "expected_total_bytes": 5027783488,
        "expected_hashes": [
          "d98cdcbd03e17ce47681435b5150e34c1417f50b5c0019dd560e4882c5745785"
        ]
      },
      "tool": {
        "name": "llama.cpp",
        "version": "0.5.0-dev",
        "release": "b11146",
        "commit": "7fe450e19305b828c199d602c23a8337aaa1f03b",
        "release_channel": "pinned tested pre-release, not a stable release; verified binary --version"
      },
      "settings": {
        "batch": 2048,
        "flash_attention": "on",
        "generation_prefilled_depth": "same as prompt-token count",
        "generation_tokens": 256,
        "kv_cache": "f16",
        "prompt_token_counts": [
          2048,
          8192,
          32768
        ],
        "repetitions": 3,
        "requested_gpu_layers": 999,
        "threads": 16,
        "ubatch": 512,
        "warmup": "llama-bench built-in enabled"
      },
      "methodology": "llama-bench microbenchmark: synthetic tokens, built-in warmup, 3 measured repetitions. Excludes tokenization, sampling, network, and application latency. OS cache is not dropped.",
      "cases": [
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 2048,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 8190735360,
              "model_size": 5021827072,
              "n_prompt": 2048,
              "n_gen": 0,
              "n_depth": 0,
              "avg_ts": 10911.980256,
              "stddev_ts": 108.481695,
              "avg_ns": 187696013,
              "stddev_ns": 1869302,
              "samples_ts": [
                10797.7,
                11013.5,
                10924.7
              ],
              "samples_ns": [
                189670369,
                185953367,
                187464303
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 2,
            "mean_sampled_board_power_w": 124.07000000000001,
            "peak_board_power_w": 234.24,
            "peak_temperature_c": 32.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 2048,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 8190735360,
              "model_size": 5021827072,
              "n_prompt": 0,
              "n_gen": 256,
              "n_depth": 2048,
              "avg_ts": 202.208294,
              "stddev_ts": 0.12964,
              "avg_ns": 1266021612,
              "stddev_ns": 812950,
              "samples_ts": [
                202.12,
                202.357,
                202.148
              ],
              "samples_ns": [
                1266576961,
                1265090450,
                1266397427
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 4,
            "mean_sampled_board_power_w": 232.41,
            "peak_board_power_w": 299.0,
            "peak_temperature_c": 38.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 8192,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 8190735360,
              "model_size": 5021827072,
              "n_prompt": 8192,
              "n_gen": 0,
              "n_depth": 0,
              "avg_ts": 9885.186206,
              "stddev_ts": 6.29609,
              "avg_ns": 828715013,
              "stddev_ns": 527783,
              "samples_ts": [
                9891.74,
                9884.64,
                9879.18
              ],
              "samples_ns": [
                828166000,
                828760404,
                829218635
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 4,
            "mean_sampled_board_power_w": 197.3125,
            "peak_board_power_w": 301.91,
            "peak_temperature_c": 43.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 8192,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 8190735360,
              "model_size": 5021827072,
              "n_prompt": 0,
              "n_gen": 256,
              "n_depth": 8192,
              "avg_ts": 172.677061,
              "stddev_ts": 0.044,
              "avg_ns": 1482536301,
              "stddev_ns": 377732,
              "samples_ts": [
                172.639,
                172.725,
                172.667
              ],
              "samples_ns": [
                1482862293,
                1482122349,
                1482624261
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 5,
            "mean_sampled_board_power_w": 190.956,
            "peak_board_power_w": 275.12,
            "peak_temperature_c": 45.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 32768,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 8190735360,
              "model_size": 5021827072,
              "n_prompt": 32768,
              "n_gen": 0,
              "n_depth": 0,
              "avg_ts": 7033.490848,
              "stddev_ts": 16.049198,
              "avg_ns": 4658869192,
              "stddev_ns": 10638265,
              "samples_ts": [
                7047.52,
                7036.97,
                7015.99
              ],
              "samples_ns": [
                4649581348,
                4656550740,
                4670475488
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 11,
            "mean_sampled_board_power_w": 269.57272727272726,
            "peak_board_power_w": 300.12,
            "peak_temperature_c": 58.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 32768,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 8190735360,
              "model_size": 5021827072,
              "n_prompt": 0,
              "n_gen": 256,
              "n_depth": 32768,
              "avg_ts": 111.731538,
              "stddev_ts": 0.148368,
              "avg_ns": 2291208966,
              "stddev_ns": 3043048,
              "samples_ts": [
                111.878,
                111.736,
                111.581
              ],
              "samples_ns": [
                2288212996,
                2291117620,
                2294296283
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 9,
            "mean_sampled_board_power_w": 240.81333333333336,
            "peak_board_power_w": 305.99,
            "peak_temperature_c": 61.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        }
      ]
    },
    {
      "run": "20260926T082248Z-medium-753b9f7e",
      "started_at": "2026-09-26T08:22:48.898059+00:00",
      "finished_at": "2026-09-26T08:38:57.201646+00:00",
      "record_sha256": "4db121364190230fa977ba749da9b957870729a5b9f8f685cc947615fec20ff9",
      "model": {
        "name": "Qwen3-32B",
        "repository": "Qwen/Qwen3-32B-GGUF",
        "revision": "938a7432affaec9157f883a87164e2646ae17555",
        "license": "apache-2.0",
        "quantization": "Q4_K_M",
        "expected_total_bytes": 19762149024,
        "expected_hashes": [
          "efd971561896866f0e910cce52761ca77b1b138090c7f15fe284676d57d1f689"
        ]
      },
      "tool": {
        "name": "llama.cpp",
        "version": "0.5.0-dev",
        "release": "b11146",
        "commit": "7fe450e19305b828c199d602c23a8337aaa1f03b",
        "release_channel": "pinned tested pre-release, not a stable release; verified binary --version"
      },
      "settings": {
        "batch": 2048,
        "flash_attention": "on",
        "generation_prefilled_depth": "same as prompt-token count",
        "generation_tokens": 256,
        "kv_cache": "f16",
        "prompt_token_counts": [
          2048,
          8192,
          32768
        ],
        "repetitions": 3,
        "requested_gpu_layers": 999,
        "threads": 16,
        "ubatch": 512,
        "warmup": "llama-bench built-in enabled",
        "cooldown_between_cases_to_c": 40
      },
      "methodology": "llama-bench microbenchmark: synthetic tokens, built-in warmup, 3 measured repetitions. Excludes tokenization, sampling, network, and application latency. OS cache is not dropped.",
      "cases": [
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 2048,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 32762123264,
              "model_size": 19756174336,
              "n_prompt": 2048,
              "n_gen": 0,
              "n_depth": 0,
              "avg_ts": 2699.481636,
              "stddev_ts": 8.151925,
              "avg_ns": 758668791,
              "stddev_ns": 2294334,
              "samples_ts": [
                2690.26,
                2705.72,
                2702.47
              ],
              "samples_ns": [
                761265227,
                756914656,
                757826490
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 5,
            "mean_sampled_board_power_w": 148.928,
            "peak_board_power_w": 300.08,
            "peak_temperature_c": 50.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 2048,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 32762123264,
              "model_size": 19756174336,
              "n_prompt": 0,
              "n_gen": 256,
              "n_depth": 2048,
              "avg_ts": 54.295146,
              "stddev_ts": 0.163058,
              "avg_ns": 4714998616,
              "stddev_ns": 14171600,
              "samples_ts": [
                54.4413,
                54.3249,
                54.1193
              ],
              "samples_ns": [
                4702314940,
                4712386791,
                4730294119
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 11,
            "mean_sampled_board_power_w": 231.33636363636361,
            "peak_board_power_w": 300.94,
            "peak_temperature_c": 55.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 8192,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 32762123264,
              "model_size": 19756174336,
              "n_prompt": 8192,
              "n_gen": 0,
              "n_depth": 0,
              "avg_ts": 2483.777392,
              "stddev_ts": 1.851664,
              "avg_ns": 3298203398,
              "stddev_ns": 2460782,
              "samples_ts": [
                2485.37,
                2481.74,
                2484.22
              ],
              "samples_ns": [
                3296092052,
                3300903917,
                3297614227
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 10,
            "mean_sampled_board_power_w": 225.711,
            "peak_board_power_w": 300.11,
            "peak_temperature_c": 57.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 8192,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 32762123264,
              "model_size": 19756174336,
              "n_prompt": 0,
              "n_gen": 256,
              "n_depth": 8192,
              "avg_ts": 53.175617,
              "stddev_ts": 0.106647,
              "avg_ns": 4814249519,
              "stddev_ns": 9654339,
              "samples_ts": [
                53.2849,
                53.1702,
                53.0718
              ],
              "samples_ns": [
                4804366031,
                4814726499,
                4823656029
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 12,
            "mean_sampled_board_power_w": 245.9283333333333,
            "peak_board_power_w": 301.05,
            "peak_temperature_c": 56.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 32768,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 32762123264,
              "model_size": 19756174336,
              "n_prompt": 32768,
              "n_gen": 0,
              "n_depth": 0,
              "avg_ts": 1839.811209,
              "stddev_ts": 18.681279,
              "avg_ns": 17811748857,
              "stddev_ns": 181083037,
              "samples_ts": [
                1857.72,
                1841.28,
                1820.44
              ],
              "samples_ns": [
                17638864638,
                17796336406,
                18000045527
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 38,
            "mean_sampled_board_power_w": 283.53263157894736,
            "peak_board_power_w": 300.37,
            "peak_temperature_c": 80.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 32768,
          "measurements": [
            {
              "build_commit": "7fe450e19",
              "model_n_params": 32762123264,
              "model_size": 19756174336,
              "n_prompt": 0,
              "n_gen": 256,
              "n_depth": 32768,
              "avg_ts": 43.739135,
              "stddev_ts": 0.178479,
              "avg_ns": 5852947173,
              "stddev_ns": 23911784,
              "samples_ts": [
                43.8967,
                43.7754,
                43.5453
              ],
              "samples_ns": [
                5831874619,
                5848032430,
                5878934470
              ]
            }
          ],
          "sampled_telemetry": {
            "interval_seconds": 2,
            "sample_count": 22,
            "mean_sampled_board_power_w": 254.24909090909088,
            "peak_board_power_w": 300.28,
            "peak_temperature_c": 65.0,
            "validity_warning": "2-second samples include startup/teardown and may miss short active phases. Do not infer model memory or steady-state power from these microbenchmark samples; GPU memory is intentionally omitted."
          }
        }
      ]
    }
  ],
  "selection_policy": "Latest complete validated run per pinned model; recorded cooling conditions are retained for each run.",
  "cohort_limitations": [
    "Inter-case cooling differs between the recorded model runs. These are observations of each configuration, not a controlled model-to-model comparison."
  ]
}
