{
  "schema_version": 1,
  "measurement_kind": "loopback_streaming_api",
  "publisher": "GPU Server Hub / Bestconnect",
  "test_date": "2026-09-26",
  "hardware": [
    {
      "gpu": "NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition",
      "gpu_reported_memory_mib": 97887.0,
      "gpu_default_limit_w": 300.0,
      "driver": "595.91.07",
      "cpu": [
        "AMD EPYC 4565P 16-Core Processor"
      ],
      "logical_cpus": 32,
      "linux_visible_ram_bytes": 100271353856,
      "physical_system_ram_gb": 96,
      "planned_system_ram_gb_not_measured": 192,
      "kernel": "6.12.0-211.56.1.el10_2.0.1.x86_64"
    }
  ],
  "limitations": [
    "Measured physical system RAM is 96 GB; the planned/offered 192 GB configuration was not measured.",
    "Results compare particular models and settings, not general answer quality or a controlled effect of parameter count.",
    "Engine throughput excludes tokenization/sampling; HTTP latency is loopback-only and excludes WAN.",
    "Telemetry samples can miss peaks between samples; board power is not whole-server AC power."
  ],
  "retained_failed_runs": [
    {
      "run": "20260926T075146Z-server-small-9451dffe",
      "model": "Qwen3-8B",
      "classification": "incomplete_stream",
      "record_sha256": "09d1b2bc405572e0da5f76eef7980eb2bf8bab64bf64b90a24707a2fa03fdd2c",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 48.49844729801407,
      "completed_case_groups": 3,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability."
    },
    {
      "run": "20260926T080932Z-server-small-317a8559",
      "model": "Qwen3-8B",
      "classification": "thermal_safety_guard",
      "record_sha256": "f5d30ca1a9567add927c6127dfdac70f74e816f09c45bc48c4b3f20ef186e4c2",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 70.75511266198009,
      "completed_case_groups": 4,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 7.0,
        "power.draw": 299.93
      },
      "guard_recorded_at": "2026-09-26T08:10:45.553634+00:00"
    },
    {
      "run": "20260926T084305Z-server-medium-36eb432a",
      "model": "Qwen3-32B",
      "classification": "thermal_safety_guard",
      "record_sha256": "b88f8c537500b2d04e04a9bcf069bc1eb4a747fcb62601a5f446ada896e0b161",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 111.19253667499288,
      "completed_case_groups": 3,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 7.0,
        "power.draw": 300.01,
        "fan.speed": 56.0,
        "clocks_event_reasons.sw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.hw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.sw_power_cap": "Active"
      },
      "guard_recorded_at": "2026-09-26T08:45:01.234302+00:00"
    },
    {
      "run": "20260926T085119Z-large-2d370bbb",
      "model": "Qwen2.5-72B-Instruct",
      "classification": "thermal_safety_guard",
      "record_sha256": "5b615f2ac8a9b6db53e38439a49e329997adc9b421eb650096f02580f9dd614b",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": null,
      "completed_case_groups": 0,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "completed_engine_cases": [
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 2048
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 2048
        },
        {
          "mode": "prefill",
          "prompt_or_prefilled_tokens": 8192
        },
        {
          "mode": "decode",
          "prompt_or_prefilled_tokens": 8192
        }
      ]
    },
    {
      "run": "20260926T091352Z-server-large-71da7c9f",
      "model": "Qwen2.5-72B-Instruct",
      "classification": "thermal_safety_guard",
      "record_sha256": "459962eeebeb8dff83f7e9ad34c145c73513d5cf67f3d05b3b14b60fb42326b0",
      "attempted_duration_seconds": 0,
      "elapsed_campaign_seconds": 118.36200489901239,
      "completed_case_groups": 1,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 8.0,
        "power.draw": 300.0,
        "fan.speed": 56.0,
        "clocks_event_reasons.sw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.hw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.sw_power_cap": "Active"
      },
      "guard_recorded_at": "2026-09-26T09:15:58.169964+00:00"
    },
    {
      "run": "20260926T092242Z-server-medium-b5a86db7",
      "model": "Qwen3-32B",
      "classification": "thermal_safety_guard",
      "record_sha256": "239fdab8b1038912667720835c50e8d8cfba223d08cbb326fa4d0d706462fc6b",
      "attempted_duration_seconds": 14400,
      "elapsed_campaign_seconds": 105.10982441101805,
      "completed_case_groups": 0,
      "meaning": "Not a successful stability run. A harness guard stop does not establish GPU thermal shutdown or driver instability.",
      "guard_sample": {
        "temperature.gpu": 85.0,
        "temperature.gpu.tlimit": 7.0,
        "power.draw": 300.01,
        "fan.speed": 56.0,
        "clocks_event_reasons.sw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.hw_thermal_slowdown": "Not Active",
        "clocks_event_reasons.sw_power_cap": "Active"
      },
      "guard_recorded_at": "2026-09-26T09:24:31.801668+00:00"
    }
  ],
  "runs": [
    {
      "run": "20260926T081648Z-server-small-b038388d",
      "started_at": "2026-09-26T08:16:48.758633+00:00",
      "finished_at": "2026-09-26T08:17:39.917640+00:00",
      "record_sha256": "bbb7daf8509b8eb03cf9d1f203a506e9eef1b6638be440cc1c3d937308aa9dee",
      "model": {
        "name": "Qwen3-8B",
        "repository": "Qwen/Qwen3-8B-GGUF",
        "revision": "7c41481f57cb95916b40956ab2f0b139b296d974",
        "license": "apache-2.0",
        "quantization": "Q4_K_M",
        "expected_total_bytes": 5027783488,
        "expected_hashes": [
          "d98cdcbd03e17ce47681435b5150e34c1417f50b5c0019dd560e4882c5745785"
        ]
      },
      "tool": {
        "name": "llama.cpp",
        "version": "0.5.0-dev",
        "release": "b11146",
        "commit": "7fe450e19305b828c199d602c23a8337aaa1f03b",
        "release_channel": "pinned tested pre-release, not a stable release; verified binary --version"
      },
      "settings": {
        "generated_tokens": 256,
        "prompt_cache": false,
        "temperature": 0,
        "seed_base": 20260926,
        "kv_cache": "f16",
        "network_scope": "loopback-only",
        "requested_gpu_layers": 999,
        "fit": "off",
        "flash_attention": "on",
        "threads": 16,
        "batch": 2048,
        "ubatch": 512,
        "contexts": [
          2048,
          8192
        ],
        "concurrency": [
          1,
          4
        ],
        "repetitions": 3,
        "duration_seconds": 0,
        "server_slots": 4,
        "context_per_slot": 8704,
        "total_context_tokens": 34816
      },
      "methodology": "Local HTTP streaming /completion: time from client request start to first generated-token event and complete response. One unmeasured warmup per case. Public authored text is tokenized then repeated/truncated to exact token count; fixed-length output ignores EOS. No quality evaluation or WAN latency claim.",
      "groups": [
        {
          "context": 2048,
          "concurrency": 1,
          "rounds": 3,
          "requests": 3,
          "ttft_median_seconds": 0.1929314169974532,
          "ttft_p95_seconds": 0.1932711687986739,
          "response_median_seconds": 1.4946947820135392,
          "response_p95_seconds": 1.4949681975209388,
          "aggregate_tps_mean": 171.21679196417597
        },
        {
          "context": 2048,
          "concurrency": 4,
          "rounds": 3,
          "requests": 12,
          "ttft_median_seconds": 0.6956389364931965,
          "ttft_p95_seconds": 0.8180397215473931,
          "response_median_seconds": 3.26163692199043,
          "response_p95_seconds": 3.289774447915261,
          "aggregate_tps_mean": 312.64948497374877
        },
        {
          "context": 8192,
          "concurrency": 1,
          "rounds": 3,
          "requests": 3,
          "ttft_median_seconds": 0.8452908120234497,
          "ttft_p95_seconds": 0.8467865651851753,
          "response_median_seconds": 2.372962160006864,
          "response_p95_seconds": 2.377301402887679,
          "aggregate_tps_mean": 107.84496019626651
        },
        {
          "context": 8192,
          "concurrency": 4,
          "rounds": 3,
          "requests": 12,
          "ttft_median_seconds": 2.4947381555102766,
          "ttft_p95_seconds": 3.5840266332859754,
          "response_median_seconds": 6.562111150007695,
          "response_p95_seconds": 6.638645357001224,
          "aggregate_tps_mean": 154.86149766264384
        }
      ],
      "measured_campaign_seconds": 49.12098955901456,
      "sampled_telemetry": {
        "sample_count": 26,
        "interval_seconds": 2,
        "temperature_c": {
          "minimum": 40.0,
          "maximum": 73.0,
          "mean": 60.57692307692308
        },
        "thermal_margin_c": {
          "minimum": 19.0,
          "maximum": 52.0,
          "mean": 31.23076923076923
        },
        "board_power_w": {
          "minimum": 14.95,
          "maximum": 300.25,
          "mean": 280.31538461538463
        },
        "gpu_memory_used_mib": {
          "minimum": 2.0,
          "maximum": 10114.0,
          "mean": 9723.153846153846
        },
        "gpu_utilization_percent": {
          "minimum": 0.0,
          "maximum": 100.0,
          "mean": 89.03846153846153
        }
      },
      "warm_cache_startup_observation": {
        "ready_seconds": 2.032999703020323,
        "poll_interval_seconds": 1,
        "methodology": "Process creation until first successful health response; includes runtime, weights and context allocation. Model hashes were read before startup: warm OS file cache, NOT a cold NVMe throughput benchmark."
      }
    }
  ]
}
