{"schemaVersion":1,"publisher":"OnPremBench by Understand Tech","scope":"Sourced catalogue records; not a live price feed or measured capacity ranking","reuse":"Original sources and third-party image rights remain applicable. No blanket license is granted for third-party material.","machines":[{"id":"nvidia-dgx-spark","brand":"NVIDIA","name":"DGX Spark","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"4 TB NVMe","form":"Desktop AI system","image":"/images/nvidia-dgx-spark-pny.png","source":"https://docs.nvidia.com/dgx/dgx-spark/hardware.html","tag":"UT field story","description":"A compact NVIDIA development system for exploring local models, applications and workflows.","notes":["Memory capacity does not establish interactive speed or simultaneous request capacity.","UT has documented a Spark deployment. Confirm exact software versions and support scope."],"managed":"documented","offer":{"amount":4699,"currency":"USD","region":"United States","url":"https://marketplace.nvidia.com/en-us/enterprise/personal-ai-supercomputers/dgx-spark/","configuration":"128 GB unified / 4 TB NVMe","tax":"Tax treatment not stated","date":"2026-09-07"},"reviewed":"2026-09-13","availability":"Listed for order · recheck with supplier","dimensions":"150 × 150 × 50.5 mm","weight":"1.2 kg","power":"240 W adapter · 140 W SoC TDP","network":"10 GbE · ConnectX-7 · Wi-Fi 7","configuration":"128 GB unified / 4 TB NVMe","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"nvidia-dgx-station","brand":"NVIDIA","name":"DGX Station","chip":"GB300 Grace Blackwell Ultra Desktop Superchip","memory":252,"memoryText":"496 GB CPU + 252 GB GPU · 748 GB coherent","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s · NVLink-C2C 900 GB/s","platform":"NVIDIA GB300","os":"NVIDIA DGX OS (Ubuntu) · Windows edition announced for Q4 2026","storage":"4 × M.2 PCIe Gen 5 slots · drives configured by the partner","form":"Deskside workstation","image":"/images/nvidia-dgx-station.jpg","source":"https://www.nvidia.com/en-us/products/workstations/dgx-station/","tag":"GB300 reference design","description":"NVIDIA’s reference design for the GB300 deskside class. Built and sold by ASUS, Dell, GIGABYTE, HP, MSI and Supermicro; every GB300 workstation in this catalogue is a version of this machine.","notes":["748 GB coherent memory is 252 GB HBM3e on the GPU plus 496 GB LPDDR5X on the CPU, joined by NVLink-C2C. The two tiers have very different bandwidths; large models spill from the fast tier into the slow one.","NVIDIA does not sell the DGX Station directly: it is ordered through partner manufacturers, each with its own chassis, drives, warranty and price. Partner listings observed in 2026 ran from about $85,000 to $175,000 depending on configuration.","NVIDIA specifies 1,600 W total system power and a 20 A circuit. Confirm OEM and regional installation requirements before choosing an office outlet. The GB300 can be paired with an RTX PRO 6000 Blackwell workstation GPU for graphics and additional compute.","DGX Station for Windows was announced on 31 May 2026 for Q4 2026 availability from the same partners."],"managed":"review","offer":null,"reviewed":"2026-09-17","availability":"Order through partner manufacturers · configuration and price set by each partner","dimensions":"Set by each partner’s chassis","weight":"Set by each partner’s chassis","power":"1,600 W platform system-power specification · 20 A circuit specified by NVIDIA","network":"ConnectX-8 SuperNIC · 2 × 400 GbE QSFP112 · 10 GbE · 1 GbE BMC","configuration":"GB300 · 748 GB coherent · 4 × M.2 Gen 5","status":"Reference design","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"nvidia-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"nvidia-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"nvidia-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"nvidia-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"dell-pro-max-gb300","brand":"Dell","name":"Pro Max with GB300","chip":"GB300 Grace Blackwell Ultra","memory":252,"memoryText":"496 GB CPU + 252 GB GPU","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s","platform":"NVIDIA GB300","os":"Ubuntu 24.04 Pro","storage":"4 × 4 TB NVMe · listed configuration","form":"Deskside workstation","image":"/images/dell-pro-max-gb300.png","source":"https://www.dell.com/en-uk/shop/desktop-computers/dell-pro-max-with-gb300/spd/dell-pro-max-fct6263-desktop/xcto_fct6263_emea","tag":"GB300 deskside","description":"A deskside reference platform for investigating much larger workloads and enterprise deployment requirements.","notes":["748 GB coherent memory comprises 252 GB HBM3e GPU plus 496 GB LPDDR5X CPU memory. These tiers have different bandwidths.","Regional snapshots cover specific configurations with RTX PRO 2000, four 4 TB drives and 12-month ProSupport. They are not entry prices or directly comparable landed costs.","UT’s founder reports a deployment of this system; comparable reproducible measurements are not yet published in this lab."],"managed":"documented","offer":{"amount":175499.98,"currency":"USD","region":"United States","url":"https://www.dell.com/en-us/shop/desktop-computers/dell-pro-max-with-gb300/spd/dell-pro-max-fct6263-desktop/xcto_fct6263_usx","configuration":"FCT6263 · GB300 + RTX PRO 2000 / 4 × 4 TB / 12-month ProSupport","tax":"Tax and shipping not established in source extract","date":"2026-09-13"},"reviewed":"2026-09-13","availability":"Listed for order · recheck with supplier","dimensions":"Confirm selected configuration","weight":"Not verified","power":"1,600 W PSU · output depends on input voltage · Dell specifies a 20 A circuit","network":"Confirm selected configuration","configuration":"FCT6263 · GB300 + RTX PRO 2000 / 4 × 4 TB / 1yr ProSupport","status":"Catalogued","architecture":"arm64","offers":[{"amount":175499.98,"currency":"USD","region":"United States","url":"https://www.dell.com/en-us/shop/desktop-computers/dell-pro-max-with-gb300/spd/dell-pro-max-fct6263-desktop/xcto_fct6263_usx","configuration":"FCT6263 · GB300 + RTX PRO 2000 / 4 × 4 TB / 12-month ProSupport","tax":"Tax and shipping not established in source extract","date":"2026-09-13"},{"amount":150217.25,"currency":"GBP","region":"United Kingdom","url":"https://www.dell.com/en-uk/shop/desktop-computers/dell-pro-max-with-gb300/spd/dell-pro-max-fct6263-desktop/xcto_fct6263_emea","configuration":"FCT6263 · GB300 + RTX PRO 2000 / 4 × 4 TB / 1yr ProSupport","tax":"Excludes UK VAT · £180,260.70 incl. 20% VAT","date":"2026-09-07"}],"evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"dell-pro-max-gb300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"dell-pro-max-gb300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"dell-pro-max-gb300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"dell-pro-max-gb300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"offer":null,"reviewed":"2026-09-09","managed":"explore","status":"Catalogued","dimensions":"Confirm with manufacturer","weight":"Not verified","power":"No comparable wall-power measurement","network":"See manufacturer configuration","storage":"Confirm with manufacturer","bandwidth":"See source; no lab measurement","id":"amd-ryzen-ai-halo","brand":"AMD","name":"Ryzen AI Halo","chip":"Ryzen AI Max+ 395 · Radeon integrated graphics","memory":128,"memoryText":"128 GB unified","platform":"AMD Ryzen AI Max","os":"Windows / Linux · ROCm","form":"Desktop AI system","image":"/images/amd-ryzen-ai-halo.jpg","source":"https://www.amd.com/en/products/processors/desktops/ryzen/ryzen-ai-halo.html","tag":"AMD compact AI","description":"A complete compact AMD developer platform with unified memory and a preconfigured local AI software environment.","notes":["The reference configuration is Ryzen AI Max+ 395 with 128 GB unified memory.","AMD lists a PRO 495 version supporting 192 GB as coming soon; it is a separate configuration.","The Lab has not tested this system or validated the UT stack."],"availability":"Manufacturer lists US purchase option; confirm stock","configuration":"Ryzen AI Max+ 395 / 128 GB unified","architecture":"x86-64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"ryzen-gpt-oss-20b","selectedHardwareId":"amd-ryzen-ai-halo","referencePlatform":"AMD Ryzen AI Max","evidence":{"origin":"AMD","status":"Vendor guide","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"model":{"name":"gpt-oss-20b","checkpoint":"ggml-org/gpt-oss-20b-GGUF","revision":null,"precision":"MXFP4 · GGUF"},"serving":{"runtime":"llama.cpp / ROCm","environment":"llama.cpp b8407 / ROCm 7.2.1 / Ubuntu 24.04","backend":"ROCm / HIP","contextTokens":2048,"contextEvidence":"Context setting in AMD’s test example.","settings":"File: gpt-oss-20b-mxfp4.gguf · -ngl 99 · -fa on","instructions":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["AMD guide version is recorded; confirm the current compatibility matrix before installing.","GPU memory allocation and OEM firmware differ.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"ryzen-qwen38-27b","selectedHardwareId":"amd-ryzen-ai-halo","referencePlatform":"Ryzen AI Max+ 395","evidence":{"origin":"AMD","status":"Vendor-reported setup","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"model":{"name":"Qwen3.8 27B","checkpoint":null,"revision":null,"precision":"GGUF · exact quantization not specified"},"serving":{"runtime":"llama.cpp / LM Studio","environment":"Windows / version not specified in source","backend":"Vulkan","contextTokens":null,"contextEvidence":"Not specified in the source; set and measure your workload.","settings":"AMD suggests MTP = 4 draft tokens on Max+ 395; disable Try mmap in LM Studio.","instructions":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Model filename, quantization and runtime revision remain to be pinned.","Published single-stream speed is not a shared-user capacity result.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"offer":null,"reviewed":"2026-09-13","managed":"explore","status":"Catalogued","dimensions":"340 × 215 × 279 mm","weight":"11 kg · confirm card population","power":"120 W TDP · not a whole-system wall-power measurement","network":"See manufacturer configuration","storage":"Confirm with manufacturer","bandwidth":"See source; no lab measurement","id":"aetina-aip-fr68","brand":"Aetina","name":"MegaEdge AIP-FR68","chip":"Qualcomm Cloud AI 100 Ultra · up to 2 cards","memory":128,"memoryText":"Up to 2 accelerator cards","platform":"Qualcomm Cloud AI","os":"Linux · Qualcomm AI Inference Suite for On-Prem","form":"Deskside workstation","image":"/images/aetina-aip-fr68.png","source":"https://www.aetina.com/products-detail.php?i=640","tag":"Qualcomm on-prem AI","description":"A complete OEM workstation for on-premises inference, using Qualcomm Cloud AI accelerators and the AI Inference Suite.","notes":["Supports up to two Cloud AI 100 Ultra cards. Confirm the accelerator population and memory with the supplier.","128 GB is the memory-screen reference for one Cloud AI 100 Ultra card; availability to a compiled model depends on the runtime.","Qualcomm model conversion and runtime support require validation. A CUDA result does not establish compatibility.","The AIP-FR68-A2 orderable bare system excludes CPU, RAM, SSD, GPU card and power supply. Obtain a complete accelerator bundle quote."],"availability":"Supplier inquiry · confirm regional availability","configuration":"AIP-FR68 / up to 2 Cloud AI 100 Ultra cards; exact build on request","architecture":"x86 host + accelerator","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"cloud-ai-llama31-8b","selectedHardwareId":"aetina-aip-fr68","referencePlatform":"Qualcomm Cloud AI accelerator family","evidence":{"origin":"Qualcomm / QEfficient","status":"Upstream model support","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://github.com/quic/efficient-transformers"},"model":{"name":"Llama 3.1 8B Instruct","checkpoint":"meta-llama/Llama-3.1-8B-Instruct","revision":null,"precision":"Compiled artifact · precision to select"},"serving":{"runtime":"QEfficient / Cloud AI SDK","environment":"Linux / matched QEfficient and SDK releases","backend":"Qualcomm Cloud AI compiler/runtime","contextTokens":null,"contextEvidence":"Context and batch sizes must be chosen for compilation.","settings":"Convert the checkpoint, compile for the actual card population and retain the compiler configuration.","instructions":"https://github.com/quic/efficient-transformers"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Library support does not confirm a compiled result on the Aetina system.","Confirm card generation, SDK, memory and licensing with the supplier.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"cloud-ai-gpt-oss-20b","selectedHardwareId":"aetina-aip-fr68","referencePlatform":"Qualcomm Cloud AI accelerator family","evidence":{"origin":"Qualcomm / QEfficient","status":"Upstream model support","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://github.com/quic/efficient-transformers"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"Compiled artifact · precision to select"},"serving":{"runtime":"QEfficient / Cloud AI SDK","environment":"Linux / matched QEfficient and SDK releases","backend":"Qualcomm Cloud AI compiler/runtime","contextTokens":null,"contextEvidence":"Context and batch sizes must be chosen for compilation.","settings":"Convert the checkpoint, compile for the actual card population and retain the compiler configuration.","instructions":"https://github.com/quic/efficient-transformers"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Library support does not confirm a compiled result on the Aetina system.","Confirm card generation, SDK, memory and licensing with the supplier.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"offer":null,"reviewed":"2026-09-13","managed":"explore","status":"Catalogued","dimensions":"Confirm with manufacturer","weight":"Not verified","power":"No comparable wall-power measurement","network":"2.5 GbE · Wi-Fi 6 (supplier tri-band listing) · Bluetooth 5.3","storage":"64 GB eMMC · NVMe expansion","bandwidth":"See source; no lab measurement","id":"arduino-ventuno-q","brand":"Arduino","name":"VENTUNO Q","chip":"Qualcomm Dragonwing IQ8 · STM32H5","memory":16,"memoryText":"16 GB shared","platform":"Qualcomm Dragonwing IQ8","os":"Ubuntu / Zephyr on MCU · Debian announced as coming soon","form":"Edge AI computer","image":"/images/arduino-ventuno-q.png","source":"https://www.arduino.cc/product-ventuno-q/","tag":"Physical AI · preorder","description":"An edge AI computer that brings local models, computer vision and real-time control onto one board.","notes":["Manufacturer specifies 40 dense TOPS and 16 GB shared LPDDR5 memory. TOPS is not an LLM tokens-per-second benchmark.","A physical-AI evaluation needs the actual sensor, camera or actuator setup as well as the compute board.","The Lab’s generic LLM memory heuristic is conservative and is not a prediction of NPU model support."],"availability":"Manufacturer preorder page · confirm shipping date","configuration":"Dragonwing IQ8 / STM32H5 / 16 GB RAM / 64 GB eMMC","architecture":"Arm + MCU","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[]},{"offer":null,"reviewed":"2026-09-09","managed":"explore","status":"Catalogued","dimensions":"Confirm with manufacturer","weight":"Not verified","power":"No comparable wall-power measurement","network":"See manufacturer configuration","storage":"Confirm with manufacturer","bandwidth":"See source; no lab measurement","id":"amd-threadripper-halo-station","brand":"AMD","name":"Threadripper Halo Station","chip":"Threadripper PRO 9995WX · Instinct MI350P","memory":144,"memoryText":"Up to 576 GB GPU + 2 TB CPU","platform":"AMD Instinct","os":"ROCm · final OEM configuration pending","form":"Deskside workstation","image":"/images/amd-threadripper-halo-station.png","source":"https://www.amd.com/en/products/workstations/amd-threadripper-halo-station.html","tag":"Announced · 2027","description":"AMD’s deskside AI workstation prototype pairs a 96-core Threadripper PRO with Instinct HBM accelerators. Manufacturer launch target: 2027.","notes":["AMD describes this as a prototype first shown at IFA 2026, coming in 2027. No verified orderable price.","576 GB is aggregate GPU memory across four 144 GB accelerators. It is not one unified GPU allocation; model sharding is required.","The model memory screen uses a single 144 GB accelerator. CPU RAM and other accelerators are excluded from that screen."],"availability":"Prototype announced · coming in 2027","configuration":"Maximum published configuration: 96-core CPU / 4 × 144 GB MI350P / up to 2 TB CPU RAM","architecture":"x86-64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[]},{"id":"asus-ascent-gx10","brand":"ASUS","name":"Ascent GX10","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"1 TB SSD · listed offer","form":"Desktop AI system","image":"/images/asus-ascent-gx10-angle.png","source":"https://www.asus.com/networking-iot-servers/desktop-ai-supercomputer/ultra-small-ai-supercomputers/asus-ascent-gx10/techspec/","tag":"Supplier stock pending","description":"An ASUS desktop AI system with unified memory and the NVIDIA software ecosystem.","notes":["Compare exact storage configuration and regional support coverage.","Validate the complete runtime and workload before a production choice."],"managed":"review","offer":{"amount":4499.95,"currency":"EUR","region":"France · LDLC","url":"https://www.ldlc.com/fiche/PB00716803.html","configuration":"90MS0371-M00030 · 128 GB / 1 TB / DGX OS","tax":"Displayed consumer price · tax treatment not explicit in reviewed extract","date":"2026-09-07"},"reviewed":"2026-09-13","availability":"Out of stock at LDLC on review date","dimensions":"150 × 150 × 51 mm","weight":"1.48 kg","power":"240 W adapter","network":"10 GbE · ConnectX-7 · Wi-Fi 7","configuration":"90MS0371-M00030 · 128 GB / 1 TB / DGX OS","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"asus-ascent-gx10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"asus-ascent-gx10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"asus-ascent-gx10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"asus-ascent-gx10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"asus-ascent-gx10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"asus-ascent-gx10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"dell-pro-max-gb10","brand":"Dell","name":"Pro Max with GB10","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s · GB10 platform","platform":"NVIDIA GB10","os":"NVIDIA DGX OS 7","storage":"4 TB PCIe Gen4","form":"Desktop AI system","image":"/images/dell-pro-max-gb10.png","source":"https://www.dell.com/en-us/shop/desktop-computers/dell-pro-max-with-gb10/spd/dell-pro-max-fcm1253-micro/xcto_fcm1253_usx","tag":"GB10 desktop","description":"Dell’s compact GB10 platform for AI development with a familiar enterprise procurement path.","notes":["Confirm region, storage, warranty and exact shipping configuration with Dell.","A shared chip does not imply identical thermals, service terms or measured performance."],"managed":"review","offer":{"amount":8224.39,"currency":"USD","region":"United States","url":"https://www.dell.com/en-us/shop/desktop-computers/dell-pro-max-with-gb10/spd/dell-pro-max-fcm1253-micro/xcto_fcm1253_usx","configuration":"FCM1253 · 128 GB / 4 TB / 12-month basic support","tax":"Tax treatment not stated","date":"2026-09-07"},"reviewed":"2026-09-07","availability":"Listed for order · recheck with supplier","dimensions":"Confirm selected configuration","weight":"Not verified","power":"280 W USB-C adapter","network":"Confirm selected configuration","configuration":"FCM1253 · 128 GB / 4 TB / 12-month basic support","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"dell-pro-max-gb10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"dell-pro-max-gb10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"dell-pro-max-gb10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"dell-pro-max-gb10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"dell-pro-max-gb10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"dell-pro-max-gb10","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"hp-zgx-nano","brand":"HP","name":"ZGX Nano G1n","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s · GB10 platform","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"2 TB PCIe Gen4 NVMe","form":"Desktop AI system","image":"/images/hp-zgx-nano-g1n.webp","source":"https://www.hp.com/us-en/shop/pdp/hp-zgx-nano-g1n-ai-station-d10nput-aba","tag":"GB10 desktop","description":"A compact HP AI development system built on the NVIDIA GB10 platform.","notes":["Product configuration and availability depend on region and SKU.","UT compatibility and managed service coverage require review.","HP's US store did not expose a usable price on the review date; request a quote for current pricing."],"managed":"review","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"Confirm selected configuration","weight":"Not verified","power":"240 W external USB-C adapter","network":"ConnectX-7 200 GbE · Wi-Fi 7","configuration":"D10NPUT#ABA · 128 GB / 2 TB","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"hp-zgx-nano","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"hp-zgx-nano","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"hp-zgx-nano","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"hp-zgx-nano","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"hp-zgx-nano","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"hp-zgx-nano","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"lenovo-thinkstation-pgx","brand":"Lenovo","name":"ThinkStation PGX","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"Up to 4 TB","form":"Desktop AI system","image":"/images/lenovo-thinkstation-pgx-news.jpg","source":"https://psref.lenovo.com/syspool/Sys/PDF/ThinkStation/ThinkStation_PGX/ThinkStation_PGX_Spec.pdf","tag":"GB10 desktop","description":"Lenovo’s compact GB10 offering for local AI development and experimentation.","notes":["Specifications vary by configured SKU and region.","Managed deployment through UT needs a compatibility and support review."],"managed":"review","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"Confirm selected configuration","weight":"Not verified","power":"240 W external adapter","network":"Confirm selected configuration","configuration":"128 GB GB10 family · select storage SKU","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"lenovo-thinkstation-pgx","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"lenovo-thinkstation-pgx","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"lenovo-thinkstation-pgx","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"lenovo-thinkstation-pgx","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"lenovo-thinkstation-pgx","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"lenovo-thinkstation-pgx","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"acer-veriton-gn100","brand":"Acer","name":"Veriton GN100","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"2 TB NVMe M.2","form":"Desktop AI system","image":"/images/acer-gn100.png","source":"https://store.acer.com/en-us/veriton-ai-mini-workstation-vgn100-ud13","tag":"GB10 desktop","description":"A complete, palm-sized GB10 workstation with a 2 TB reference configuration and NVIDIA’s development stack.","notes":["The listed US offer is the VGN100-UD13 configuration with 128 GB unified memory and a 2 TB NVMe drive; other regional SKUs may differ in storage and bundled software.","Acer positions the GN100 on NVIDIA's GB10 reference platform. Vendor and community GB10 guidance applies to the platform; no Lab measurement exists for this exact chassis.","Memory capacity alone does not establish speed, runtime compatibility or production capacity."],"managed":"review","offer":{"amount":5499.99,"currency":"USD","region":"United States","url":"https://store.acer.com/en-us/veriton-ai-mini-workstation-vgn100-ud13","configuration":"VGN100-UD13 · 128 GB / 2 TB","tax":"Tax treatment not stated","date":"2026-09-07"},"reviewed":"2026-09-07","availability":"In stock at source on review date","dimensions":"150 × 150 × 50.8 mm · rounded from inches","weight":"Not verified","power":"170 W power-supply listing from Acer · confirm the delivered adapter","network":"ConnectX-7 · Wi-Fi 7 · Bluetooth 5.4","configuration":"VGN100-UD13 · 128 GB / 2 TB","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"acer-veriton-gn100","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"acer-veriton-gn100","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"acer-veriton-gn100","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"acer-veriton-gn100","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"acer-veriton-gn100","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"acer-veriton-gn100","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"gigabyte-ai-top-atom","brand":"GIGABYTE","name":"AI TOP ATOM","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"1 TB Gen4 / 4 TB Gen4 / 4 TB Gen5","form":"Desktop AI system","image":"/images/gigabyte-atom.png","source":"https://www.gigabyte.com/ge/AI-TOP-PC/GIGABYTE-AI-TOP-ATOM/sp","tag":"GB10 desktop","description":"A one-liter GB10 computer with high-speed networking and three documented storage variants.","notes":["GIGABYTE publishes three ATAGB10 variants differing in storage (1 TB Gen4, 4 TB Gen4, 4 TB Gen5). Confirm which variant a regional seller offers; no usable public price was verified on the review date.","The ATOM shares NVIDIA's GB10 platform. Platform guidance applies; no Lab measurement exists for this exact chassis.","Memory capacity alone does not establish speed, runtime compatibility or production capacity."],"managed":"review","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"150 × 150 × 50.5 mm","weight":"1.2 kg","power":"240 W adapter","network":"10 GbE · ConnectX-7 · Wi-Fi 7","configuration":"ATAGB10-9000 / 9001 / 9002 family","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"gigabyte-ai-top-atom","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"gigabyte-ai-top-atom","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"gigabyte-ai-top-atom","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"gigabyte-ai-top-atom","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"gigabyte-ai-top-atom","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"gigabyte-ai-top-atom","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"msi-edgexpert","brand":"MSI","name":"EdgeXpert","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"4 TB PCIe Gen4 NVMe","form":"Desktop AI system","image":"/images/msi-edgexpert.png","source":"https://us-store.msi.com/EdgeXpert-AI","tag":"GB10 desktop","description":"A compact GB10 workstation offered in several SSD and cable bundles. This listing uses the single-unit 4 TB Gen4 configuration.","notes":["The US store offer is the EdgeXpert-11SUS single unit with 128 GB unified memory and a 4 TB Gen4 drive. MSI also markets paired-unit configurations; confirm networking accessories for a two-node setup.","The EdgeXpert shares NVIDIA's GB10 platform. Platform guidance applies; no Lab measurement exists for this exact chassis.","Memory capacity alone does not establish speed, runtime compatibility or production capacity."],"managed":"review","offer":{"amount":5979,"currency":"USD","region":"United States","url":"https://us-store.msi.com/EdgeXpert-AI","configuration":"EdgeXpert-11SUS · 128 GB / 4 TB Gen4 · one unit","tax":"Tax treatment not stated","date":"2026-09-07"},"reviewed":"2026-09-07","availability":"Listed for order · recheck with supplier","dimensions":"Confirm selected configuration","weight":"Not verified","power":"Approximately 240 W external USB-C adapter","network":"Confirm selected configuration","configuration":"EdgeXpert-11SUS · 128 GB / 4 TB Gen4 · one unit","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"msi-edgexpert","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"msi-edgexpert","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"msi-edgexpert","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"msi-edgexpert","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"msi-edgexpert","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"msi-edgexpert","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"asus-expertcenter-et900n-g3","brand":"ASUS","name":"ExpertCenter Pro ET900N G3","chip":"GB300 Grace Blackwell Ultra","memory":252,"memoryText":"496 GB CPU + 252 GB GPU","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s","platform":"NVIDIA GB300","os":"Ubuntu with NVIDIA AI Developer Tools","storage":"2 × 2 TB pre-installed OS drives · 4 M.2 slots","form":"Deskside workstation","image":"/images/asus-et900n-g3.png","source":"https://www.asus.com/us/displays-desktops/workstations/performance/expertcenter-pro-et900n-g3/techspec/","tag":"GB300 deskside","description":"A liquid-cooled GB300 deskside workstation with separate high-bandwidth GPU and large CPU memory tiers.","notes":["252 GB GPU and 496 GB CPU memory are coherent but do not have equal bandwidth.","The manufacturer specifies two pre-installed 2 TB OS drives. Confirm usable RAID capacity and optional data drives."],"managed":"review","offer":null,"reviewed":"2026-09-13","availability":"Check regional availability","dimensions":"584 × 232 × 565 mm","weight":"27 kg net · 32 kg gross","power":"1600 W Titanium PSU · regional voltage and power cord requirements apply","network":"ConnectX-8 dual QSFP112 · 10 GbE · BMC management","configuration":"Manufacturer platform · exact SKU to configure","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"asus-expertcenter-et900n-g3","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"asus-expertcenter-et900n-g3","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"asus-expertcenter-et900n-g3","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"asus-expertcenter-et900n-g3","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"msi-xpertstation-ws300","brand":"MSI","name":"XpertStation WS300","chip":"GB300 Grace Blackwell Ultra","memory":252,"memoryText":"496 GB CPU + 252 GB GPU","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s","platform":"NVIDIA GB300","os":"Ubuntu with NVIDIA AI Developer Tools","storage":"2 × 2 TB SSD · seller reference","form":"Deskside workstation","image":"/images/msi-ws300.png","source":"https://us.msi.com/Landing/NVIDIA-DGX-STATION","tag":"Preorder · GB300","description":"MSI’s liquid-cooled deskside GB300 system, with remote management and high-speed interconnects for larger local experiments.","notes":["The EUR figure is a German reseller preorder for the WS300T60L with two 2 TB SSDs, quoted before final configuration. It is not directly comparable with the Dell GB300 US snapshot, which includes an RTX PRO 2000 card, four 4 TB drives and 12-month ProSupport.","The WS300 shares NVIDIA's GB300 platform. Platform guidance applies; no Lab measurement exists for this exact chassis.","MSI and the reseller list differing PSU efficiency and chassis dimensions. This page uses MSI's platform specifications; confirm the exact WS300T60L revision in the quote."],"managed":"review","offer":{"amount":93277.31,"currency":"EUR","region":"Germany · buyzero / pi3g","url":"https://buyzero.de/en/products/msi-xpertstation-ws300-ws300t60l-nvidia-dgx-station-gb300","configuration":"WS300T60L · GB300 / 2 × 2 TB SSD · confirm final quote","tax":"Net price · €111,000 gross listed by seller","date":"2026-09-07"},"reviewed":"2026-09-07","availability":"Preorder at buyzero · seller estimates 10–12 weeks","dimensions":"245 × 528.4 × 595 mm · W × H × D","weight":"Not verified","power":"1600 W 80 PLUS Titanium PSU","network":"2 × 400 GbE QSFP · 10 GbE · BMC","configuration":"WS300T60L · GB300 / 2 × 2 TB SSD · confirm final quote","status":"Preorder","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"msi-xpertstation-ws300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"msi-xpertstation-ws300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"msi-xpertstation-ws300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"msi-xpertstation-ws300","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"hp-zgx-fury","brand":"HP","name":"ZGX Fury","chip":"GB300 Grace Blackwell Ultra","memory":252,"memoryText":"496 GB CPU + 252 GB GPU","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s","platform":"NVIDIA GB300","os":"Ubuntu with NVIDIA AI Developer Tools","storage":"Configure with supplier","form":"Deskside workstation","image":"/images/hp-zgx-fury.png","source":"https://www.hp.com/us-en/newsroom/blogs/2026/unlock-ai-with-zgx-fury.html","tag":"GB300 deskside","description":"HP’s GB300 deskside AI station, pairing large coherent memory with the HP ZGX software toolkit.","notes":["HP states 748 GB coherent memory. Confirm the exact local configuration and delivery schedule with HP.","Manufacturer maximum-model claims depend on quantization and workload; they are not lab benchmarks."],"managed":"review","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"Confirm selected configuration","weight":"Not verified","power":"Not verified · measure actual consumption","network":"Confirm selected configuration","configuration":"GB300 platform · request full regional SKU","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"hp-zgx-fury","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"hp-zgx-fury","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"hp-zgx-fury","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"hp-zgx-fury","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"gigabyte-w775-v10","brand":"GIGABYTE","name":"W775-V10-L01","chip":"GB300 Grace Blackwell Ultra","memory":252,"memoryText":"496 GB CPU + 252 GB GPU","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s","platform":"NVIDIA GB300","os":"Ubuntu with NVIDIA AI Developer Tools","storage":"4 M.2 slots · SSDs selected at integration","form":"Deskside workstation","image":"/images/gigabyte-w775.png","source":"https://www.gigabyte.com/at/Enterprise/Tower-Server/W775-V10-L01","tag":"Integration required","description":"A GB300 workstation platform with closed-loop cooling, BMC management and configurable storage. Order as an integrated system through a supplier.","notes":["This vendor reference is a barebone workstation platform. A supplier must specify SSDs, OS, display GPU and the final ready-to-use configuration.","252 GB GPU plus 496 GB CPU memory; power and circuit requirements need checking before installation."],"managed":"review","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"245 × 500.4 × 531 mm · W × H × D","weight":"29.2 kg net","power":"1,600 W Platinum PSU · NVIDIA platform guidance specifies a 20 A circuit","network":"2 × 400 GbE QSFP · 10 GbE · management LAN","configuration":"6NW775V10MR000L01 · barebone platform; integration required","status":"Configured platform","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"gigabyte-w775-v10","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"gigabyte-w775-v10","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"gigabyte-w775-v10","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"gigabyte-w775-v10","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"supermicro-ai-station","brand":"Supermicro","name":"Super AI Station · ARS-511GD","chip":"GB300 Grace Blackwell Ultra","memory":252,"memoryText":"496 GB CPU + 252 GB GPU","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s","platform":"NVIDIA GB300","os":"Ubuntu with NVIDIA AI Developer Tools","storage":"4 M.2 slots · configured drives","form":"Tower / 5U rack","image":"/images/supermicro-ars511.jpg","source":"https://www.supermicro.com/en/products/system/workstation/tower/ars-511gd-nb-lcc","tag":"GB300 deskside","description":"A GB300 system that moves from tower to 5U rack with an optional kit, bridging deskside experiments and an on-premises installation.","notes":["Optional rackmount kit; confirm rails, rack depth, circuit and acoustic suitability with the integrator.","748 GB combines 252 GB GPU and 496 GB CPU memory. An optional display GPU has its own separate memory."],"managed":"review","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"454.7 × 218.4 × 701 mm · enclosure","weight":"40 kg net","power":"1600 W Titanium PSU","network":"2 × 400 GbE · 10 GbE · IPMI / Redfish","configuration":"ARS-511GD-NB-LCC · integrated configuration on request","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"supermicro-ai-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"supermicro-ai-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"supermicro-ai-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"supermicro-ai-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"exxact-valence-dgx-station","brand":"Exxact","name":"Valence DGX Station","chip":"GB300 Grace Blackwell Ultra Desktop Superchip","memory":252,"memoryText":"496 GB CPU + 252 GB GPU","bandwidth":"GPU 7.1 TB/s · CPU 396 GB/s","platform":"NVIDIA GB300","os":"NVIDIA DGX OS (Ubuntu)","storage":"2 × PCIe 5.0 M.2 + 2 × PCIe 6.0 M.2 slots · drives configured at order","form":"Deskside workstation","image":"/images/exxact-dgx-station.jpg","source":"https://www.exxactcorp.com/Exxact-VWS-158270643-E158270643","tag":"GB300 · published starting price","description":"A US system integrator’s build of the NVIDIA DGX Station reference design, with a published starting price and an online configurator.","notes":["One of the few GB300 workstations with a public starting price. The starting configuration excludes drives, the optional display GPU and support options; the configurator sets the final price.","Same GB300 superchip, memory and networking as every DGX Station; the chassis, cooling and warranty are Exxact’s.","Supports one double-wide graphics card for display alongside the GB300."],"managed":"explore","offer":{"amount":95150,"currency":"USD","region":"United States","url":"https://www.exxactcorp.com/Exxact-VWS-158270643-E158270643","configuration":"VWS-158270643 · GB300 · starting configuration before drives and options","tax":"Starting price · tax and shipping not stated","date":"2026-09-17"},"offers":[{"amount":95150,"currency":"USD","region":"United States","url":"https://www.exxactcorp.com/Exxact-VWS-158270643-E158270643","configuration":"VWS-158270643 · GB300 · starting configuration before drives and options","tax":"Starting price · tax and shipping not stated","date":"2026-09-17"}],"reviewed":"2026-09-17","availability":"Configure and quote online · lead time set by the seller","dimensions":"Not published in source extract","weight":"Not published in source extract","power":"1,600 W NVIDIA platform system-power specification · installed PSU needs confirmation","network":"ConnectX-8 · 2 × 400 GbE QSFP112 · 10 GbE · 1 GbE BMC","configuration":"VWS-158270643 · GB300 · starting configuration","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"station-qwen-8b","selectedHardwareId":"exxact-valence-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"Qwen/Qwen3-8B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-35b","selectedHardwareId":"exxact-valence-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 35B A3B","checkpoint":"Qwen/Qwen3.6-35B-A3B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Hybrid architecture: use its runtime-specific cache guidance.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-qwen-27b","selectedHardwareId":"exxact-valence-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3.6 27B","checkpoint":"Qwen/Qwen3.6-27B","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"station-llama-70b","selectedHardwareId":"exxact-valence-dgx-station","referencePlatform":"NVIDIA DGX Station","evidence":{"origin":"NVIDIA","status":"Vendor starting point","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"meta-llama/Llama-3.3-70B-Instruct","revision":null,"precision":"Checkpoint default · confirm dtype"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"hp-z8-fury-g6i","brand":"HP","name":"Z8 Fury G6i","chip":"Intel Xeon 600 · RTX PRO 6000 Blackwell Max-Q option","memory":96,"memoryText":"96 GB per GPU · up to 4 GPUs","bandwidth":"Confirm selected configuration","platform":"NVIDIA RTX PRO","os":"Windows 11 Pro · confirm Linux option","storage":"Up to 104 TB · configured drives","form":"Expandable tower","image":"/images/hp-z8-fury-g6i.png","source":"https://www.hp.com/us-en/workstations/z8-fury.html","tag":"Expandable RTX PRO","description":"An expandable Xeon workstation supporting up to four RTX PRO 6000 Blackwell Max-Q GPUs. Configure it for a workstation or a larger shared inference service.","notes":["Up to 2 TB DDR5 ECC system RAM is separate from GPU memory. The memory screen uses one 96 GB GPU.","Multiple GPUs do not automatically form one memory pool. Framework support, sharding and interconnect overhead must be evaluated."],"managed":"explore","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"439.1 × 219.4 × 558.8 mm","weight":"From 22.2 kg","power":"1,350 / 1,700 W single PSU by voltage · up to 2,700 W dual aggregate capacity","network":"Confirm selected configuration","configuration":"1 × 96 GB RTX PRO 6000 Max-Q screening reference · other components configurable","status":"Catalogued","architecture":"x86-64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"rtx-gpt-oss-20b","selectedHardwareId":"hp-z8-fury-g6i","referencePlatform":"RTX PRO workstation / selected 96 GB GPU","evidence":{"origin":"NVIDIA","status":"Vendor support statement","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"Ollama / llama.cpp","environment":"Version and OS to pin for the selected workstation","backend":"CUDA","contextTokens":null,"contextEvidence":"No exact workstation context or concurrency test is provided.","settings":"Confirm the selected runtime’s gpt-oss support and GPU offload. Workstation Edition and Max-Q need separate measurements.","instructions":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Source establishes model support, not a pinned OEM recipe.","Do not transfer datacenter throughput figures to this workstation.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"rtx-gpt-oss-120b","selectedHardwareId":"hp-z8-fury-g6i","referencePlatform":"RTX PRO workstation / selected 96 GB GPU","evidence":{"origin":"NVIDIA","status":"Vendor support statement","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"Ollama / llama.cpp","environment":"Version and OS to pin for the selected workstation","backend":"CUDA","contextTokens":null,"contextEvidence":"No exact workstation context or concurrency test is provided.","settings":"Confirm the selected runtime’s gpt-oss support and GPU offload. Workstation Edition and Max-Q need separate measurements.","instructions":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Source establishes model support, not a pinned OEM recipe.","Do not transfer datacenter throughput figures to this workstation.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"lenovo-thinkstation-p7","brand":"Lenovo","name":"ThinkStation P7","chip":"Intel Xeon W · RTX PRO 6000 Blackwell option","memory":96,"memoryText":"96 GB per GPU · up to 3 Max-Q GPUs","bandwidth":"Confirm selected configuration","platform":"NVIDIA RTX PRO","os":"Windows / Ubuntu options · selected SKU","storage":"Configure with supplier","form":"Expandable tower","image":"/images/lenovo-p7.png","source":"https://psref.lenovo.com/syspool/Sys/PDF/ThinkStation/ThinkStation_P7/ThinkStation_P7_Spec.pdf","tag":"Expandable RTX PRO","description":"An enterprise tower offering a route to single- or multi-GPU local AI with configurable Xeon processors and professional graphics.","notes":["September 2026 PSREF supports up to three RTX PRO 6000 Blackwell Max-Q GPUs or one 600 W Workstation Edition GPU, with orderability still to confirm.","Memory screening uses a single 96 GB GPU, not an assumed pool across three cards."],"managed":"explore","offer":null,"reviewed":"2026-09-07","availability":"GPU support documented · Blackwell orderability to confirm","dimensions":"Confirm selected configuration","weight":"Not verified","power":"1,000 W or 1,400 W PSU · regional voltage derating applies","network":"Confirm selected configuration","configuration":"1 × 96 GB RTX PRO 6000 screening reference · configured platform","status":"Catalogued","architecture":"x86-64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"rtx-gpt-oss-20b","selectedHardwareId":"lenovo-thinkstation-p7","referencePlatform":"RTX PRO workstation / selected 96 GB GPU","evidence":{"origin":"NVIDIA","status":"Vendor support statement","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"Ollama / llama.cpp","environment":"Version and OS to pin for the selected workstation","backend":"CUDA","contextTokens":null,"contextEvidence":"No exact workstation context or concurrency test is provided.","settings":"Confirm the selected runtime’s gpt-oss support and GPU offload. Workstation Edition and Max-Q need separate measurements.","instructions":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Source establishes model support, not a pinned OEM recipe.","Do not transfer datacenter throughput figures to this workstation.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"rtx-gpt-oss-120b","selectedHardwareId":"lenovo-thinkstation-p7","referencePlatform":"RTX PRO workstation / selected 96 GB GPU","evidence":{"origin":"NVIDIA","status":"Vendor support statement","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"Ollama / llama.cpp","environment":"Version and OS to pin for the selected workstation","backend":"CUDA","contextTokens":null,"contextEvidence":"No exact workstation context or concurrency test is provided.","settings":"Confirm the selected runtime’s gpt-oss support and GPU offload. Workstation Edition and Max-Q need separate measurements.","instructions":"https://developer.nvidia.com/blog/delivering-1-5-m-tps-inference-on-nvidia-gb200-nvl72-nvidia-accelerates-openai-gpt-oss-models-from-cloud-to-edge/"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Source establishes model support, not a pinned OEM recipe.","Do not transfer datacenter throughput figures to this workstation.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"framework-desktop","brand":"Framework","name":"Desktop · Ryzen AI Max+","chip":"Ryzen AI Max+ 395 · Radeon 8060S","memory":128,"memoryText":"128 GB shared system memory","bandwidth":"Confirm selected configuration","platform":"AMD Ryzen AI Max","os":"Bring your own OS · Windows / Linux options","storage":"2 M.2 PCIe 4.0 sockets · choose SSDs","form":"Mini workstation","image":"/images/framework-desktop.png","source":"https://frame.work/desktop/?tab=specs","tag":"DIY kit · extras required","description":"A 4.5-liter desktop with standard replaceable components and an AMD shared-memory architecture. The desktop kit requires assembly and an OS.","notes":["The 128 GB memory is soldered. GPU-available memory depends on OS and configuration.","The listed $3,449 is the 128 GB system kit selection, not a ready-to-run total. SSD, OS, fan, cable and tiles are additional selections."],"managed":"explore","offer":{"amount":3449,"currency":"USD","region":"United States","url":"https://frame.work/products/desktop-diy-amd-aimax300/configuration/new","configuration":"128 GB Max+ 395 system kit only · SSD, OS, fan, cable and tiles extra","tax":"Tax treatment not stated","date":"2026-09-07"},"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"Confirm selected configuration","weight":"Not verified","power":"400 W PSU · processor 120 W sustained / 140 W boost","network":"5 GbE · Wi-Fi 7","configuration":"128 GB Max+ 395 desktop kit · storage and OS extra","status":"DIY kit","architecture":"x86-64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"ryzen-gpt-oss-20b","selectedHardwareId":"framework-desktop","referencePlatform":"AMD Ryzen AI Max","evidence":{"origin":"AMD","status":"Vendor guide","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"model":{"name":"gpt-oss-20b","checkpoint":"ggml-org/gpt-oss-20b-GGUF","revision":null,"precision":"MXFP4 · GGUF"},"serving":{"runtime":"llama.cpp / ROCm","environment":"llama.cpp b8407 / ROCm 7.2.1 / Ubuntu 24.04","backend":"ROCm / HIP","contextTokens":2048,"contextEvidence":"Context setting in AMD’s test example.","settings":"File: gpt-oss-20b-mxfp4.gguf · -ngl 99 · -fa on","instructions":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["AMD guide version is recorded; confirm the current compatibility matrix before installing.","GPU memory allocation and OEM firmware differ.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"ryzen-qwen38-27b","selectedHardwareId":"framework-desktop","referencePlatform":"Ryzen AI Max+ 395","evidence":{"origin":"AMD","status":"Vendor-reported setup","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"model":{"name":"Qwen3.8 27B","checkpoint":null,"revision":null,"precision":"GGUF · exact quantization not specified"},"serving":{"runtime":"llama.cpp / LM Studio","environment":"Windows / version not specified in source","backend":"Vulkan","contextTokens":null,"contextEvidence":"Not specified in the source; set and measure your workload.","settings":"AMD suggests MTP = 4 draft tokens on Max+ 395; disable Try mmap in LM Studio.","instructions":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Model filename, quantization and runtime revision remain to be pinned.","Published single-stream speed is not a shared-user capacity result.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"minisforum-ms-s1-max","brand":"MINISFORUM","name":"MS-S1 MAX","chip":"Ryzen AI Max+ 395 · Radeon 8060S","memory":128,"memoryText":"128 GB shared system memory","bandwidth":"Confirm selected configuration","platform":"AMD Ryzen AI Max","os":"Confirm OS option","storage":"2 TB SSD","form":"Mini / optional 2U rack","image":"/images/minisforum-mss1.png","source":"https://store.minisforum.com/products/minisforum-ms-s1-max-mini-pc","tag":"AMD compact","description":"A compact AMD workstation with dual 10 GbE, PCIe expansion and an optional path to a 2U rack arrangement.","notes":["The selected 128 GB offer lists estimated mid-September shipping. Confirm delivery before ordering.","PCIe x16 physical slot is wired PCIe 4.0 x4; do not assume x16 bandwidth. Shared memory allocation depends on runtime and OS."],"managed":"explore","offer":{"amount":3799,"currency":"USD","region":"United States","url":"https://store.minisforum.com/products/minisforum-ms-s1-max-mini-pc","configuration":"128 GB Max AI Compute Edition / 2 TB SSD / US adapter","tax":"Tax treatment not stated","date":"2026-09-07"},"reviewed":"2026-09-07","availability":"Estimated shipping mid-September · supplier statement","dimensions":"Confirm selected configuration","weight":"Not verified","power":"320 W built-in PSU · processor 130 W sustained / 160 W peak","network":"Dual 10 GbE · USB4 v2 · Wi-Fi 7","configuration":"128 GB Max AI Compute Edition / 2 TB SSD / US adapter","status":"Catalogued","architecture":"x86-64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"ryzen-gpt-oss-20b","selectedHardwareId":"minisforum-ms-s1-max","referencePlatform":"AMD Ryzen AI Max","evidence":{"origin":"AMD","status":"Vendor guide","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"model":{"name":"gpt-oss-20b","checkpoint":"ggml-org/gpt-oss-20b-GGUF","revision":null,"precision":"MXFP4 · GGUF"},"serving":{"runtime":"llama.cpp / ROCm","environment":"llama.cpp b8407 / ROCm 7.2.1 / Ubuntu 24.04","backend":"ROCm / HIP","contextTokens":2048,"contextEvidence":"Context setting in AMD’s test example.","settings":"File: gpt-oss-20b-mxfp4.gguf · -ngl 99 · -fa on","instructions":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["AMD guide version is recorded; confirm the current compatibility matrix before installing.","GPU memory allocation and OEM firmware differ.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"ryzen-qwen38-27b","selectedHardwareId":"minisforum-ms-s1-max","referencePlatform":"Ryzen AI Max+ 395","evidence":{"origin":"AMD","status":"Vendor-reported setup","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"model":{"name":"Qwen3.8 27B","checkpoint":null,"revision":null,"precision":"GGUF · exact quantization not specified"},"serving":{"runtime":"llama.cpp / LM Studio","environment":"Windows / version not specified in source","backend":"Vulkan","contextTokens":null,"contextEvidence":"Not specified in the source; set and measure your workload.","settings":"AMD suggests MTP = 4 draft tokens on Max+ 395; disable Try mmap in LM Studio.","instructions":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Model filename, quantization and runtime revision remain to be pinned.","Published single-stream speed is not a shared-user capacity result.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"hp-z2-mini-g1a","brand":"HP","name":"Z2 Mini G1a","chip":"Ryzen AI Max+ PRO 395 · Radeon 8060S","memory":128,"memoryText":"128 GB system memory","bandwidth":"Verify selected configuration","platform":"AMD Ryzen AI Max","os":"Windows 11 Pro · reference","storage":"1 TB · reference","form":"Mini workstation","image":"/images/hp-z2-mini-g1a.webp","source":"https://www.hp.com/in-en/products/workstations/product-details/product-specifications/2104054511","tag":"AMD compact","description":"A compact AMD workstation offering a different architecture for local AI exploration.","notes":["Reference SKU E07RFPA. GPU-accessible memory depends on configuration; do not assume all system memory is available to the model.","Runtime and full UT software support require validation."],"managed":"explore","offer":null,"reviewed":"2026-09-07","availability":"Check regional availability","dimensions":"Confirm selected configuration","weight":"Not verified","power":"300 W internal power unit (E07RFPA)","network":"Confirm selected configuration","configuration":"E07RFPA · 128 GB / 1 TB","status":"Catalogued","architecture":"x86-64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"ryzen-gpt-oss-20b","selectedHardwareId":"hp-z2-mini-g1a","referencePlatform":"AMD Ryzen AI Max","evidence":{"origin":"AMD","status":"Vendor guide","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"model":{"name":"gpt-oss-20b","checkpoint":"ggml-org/gpt-oss-20b-GGUF","revision":null,"precision":"MXFP4 · GGUF"},"serving":{"runtime":"llama.cpp / ROCm","environment":"llama.cpp b8407 / ROCm 7.2.1 / Ubuntu 24.04","backend":"ROCm / HIP","contextTokens":2048,"contextEvidence":"Context setting in AMD’s test example.","settings":"File: gpt-oss-20b-mxfp4.gguf · -ngl 99 · -fa on","instructions":"https://rocm.docs.amd.com/projects/radeon-ryzen/en/docs-7.2.1/docs/advanced/advancedryz/linux/llm/llamacpp.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["AMD guide version is recorded; confirm the current compatibility matrix before installing.","GPU memory allocation and OEM firmware differ.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"ryzen-qwen38-27b","selectedHardwareId":"hp-z2-mini-g1a","referencePlatform":"Ryzen AI Max+ 395","evidence":{"origin":"AMD","status":"Vendor-reported setup","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"model":{"name":"Qwen3.8 27B","checkpoint":null,"revision":null,"precision":"GGUF · exact quantization not specified"},"serving":{"runtime":"llama.cpp / LM Studio","environment":"Windows / version not specified in source","backend":"Vulkan","contextTokens":null,"contextEvidence":"Not specified in the source; set and measure your workload.","settings":"AMD suggests MTP = 4 draft tokens on Max+ 395; disable Try mmap in LM Studio.","instructions":"https://www.amd.com/en/blogs/2026/run-qwen-3-8-27b-on-amd-ryzen-ai-max-and-radeon-graphics-cards-day-0.html"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Model filename, quantization and runtime revision remain to be pinned.","Published single-stream speed is not a shared-user capacity result.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"apple-mac-studio-m5","brand":"Apple","name":"Mac Studio · M5 Ultra","chip":"Apple M5 Ultra · 30-core CPU / 64-core GPU","memory":96,"memoryText":"96 GB unified · selected model","bandwidth":"Not verified for selected M5 Ultra","platform":"Apple Silicon","os":"macOS","storage":"1 TB · reference","form":"Mini workstation","image":"/images/mac-studio-m5.jpg","source":"https://www.apple.com/shop/buy-mac/mac-studio/m5-ultra-chip-30-core-cpu-64-core-gpu-96gb-memory-1tb-storage","tag":"Preorder · Apple Silicon","description":"An M5 Ultra preorder configuration for exploring local inference in the Metal and MLX ecosystem. Exact runtime support remains to be checked.","notes":["Apple store lists this configuration for pre-order, available from 22 September. No usable price was exposed in the reviewed store page.","This is a new configuration. Runtime compatibility, actual GPU memory allowance and performance have not been validated by this lab."],"managed":"explore","offer":null,"reviewed":"2026-09-13","availability":"Apple lists availability from 22 September 2026","dimensions":"Confirm current M5 chassis specifications","weight":"Not verified","power":"480 W maximum continuous power rating · not typical consumption","network":"Confirm current M5 configuration","configuration":"M5 Ultra · 30-core CPU / 64-core GPU / 96 GB / 1 TB","status":"Preorder","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"mac-llama32-3b","selectedHardwareId":"apple-mac-studio-m5","referencePlatform":"Apple Silicon / macOS","evidence":{"origin":"MLX-LM maintainers","status":"Maintainer example","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://github.com/ml-explore/mlx-lm"},"model":{"name":"Llama 3.2 3B Instruct","checkpoint":"mlx-community/Llama-3.2-3B-Instruct-4bit","revision":null,"precision":"4-bit · MLX"},"serving":{"runtime":"MLX-LM","environment":"macOS / pin mlx-lm and mlx versions","backend":"MLX / Metal","contextTokens":null,"contextEvidence":"No Mac Studio load-test context is published in this record.","settings":"Use the MLX-converted checkpoint; record prompt length, cache settings and generation limits.","instructions":"https://github.com/ml-explore/mlx-lm"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Ecosystem-level guidance; no M3 Ultra or M5 Ultra reproduction is attached.","Available GPU memory and runtime support must be checked on the exact Mac.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"mac-mistral-7b","selectedHardwareId":"apple-mac-studio-m5","referencePlatform":"Apple Silicon / macOS","evidence":{"origin":"MLX-LM maintainers","status":"Maintainer example","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://github.com/ml-explore/mlx-lm"},"model":{"name":"Mistral 7B Instruct v0.3","checkpoint":"mlx-community/Mistral-7B-Instruct-v0.3-4bit","revision":null,"precision":"4-bit · MLX"},"serving":{"runtime":"MLX-LM","environment":"macOS / pin mlx-lm and mlx versions","backend":"MLX / Metal","contextTokens":null,"contextEvidence":"No Mac Studio load-test context is published in this record.","settings":"Use the MLX-converted checkpoint; record prompt length, cache settings and generation limits.","instructions":"https://github.com/ml-explore/mlx-lm"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Ecosystem-level guidance; no M3 Ultra or M5 Ultra reproduction is attached.","Available GPU memory and runtime support must be checked on the exact Mac.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"mac-qwen3-8b","selectedHardwareId":"apple-mac-studio-m5","referencePlatform":"Apple Silicon / macOS","evidence":{"origin":"MLX Community","status":"Community model card","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://huggingface.co/mlx-community/Qwen3-8B-4bit"},"model":{"name":"Qwen3 8B","checkpoint":"mlx-community/Qwen3-8B-4bit","revision":null,"precision":"4-bit · MLX"},"serving":{"runtime":"MLX-LM","environment":"macOS / pin mlx-lm and mlx versions","backend":"MLX / Metal","contextTokens":null,"contextEvidence":"No Mac Studio load-test context is published in this record.","settings":"Use the MLX-converted checkpoint; record prompt length, cache settings and generation limits.","instructions":"https://github.com/ml-explore/mlx-lm"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Ecosystem-level guidance; no M3 Ultra or M5 Ultra reproduction is attached.","Available GPU memory and runtime support must be checked on the exact Mac.","No exact-OEM benchmark or cost per user is published in this record."]}]},{"id":"apple-mac-studio","brand":"Apple","name":"Mac Studio · M3 Ultra","chip":"Apple M3 Ultra","memory":96,"memoryText":"96 GB unified · reference","bandwidth":"819 GB/s","platform":"Apple Silicon","os":"macOS","storage":"1 TB · reference","form":"Mini workstation","image":"/images/apple-mac-studio-2025.jpg","source":"https://www.apple.com/tm/mac-studio/specs/","tag":"Earlier generation","description":"An M3 Ultra reference configuration for exploring local inference in the Metal and MLX ecosystem.","notes":["This is a specific M3 Ultra configuration, not a claim that it is the latest Mac Studio.","CUDA workloads need a different platform. UT’s complete managed stack is not validated here."],"managed":"explore","offer":null,"reviewed":"2026-09-07","availability":"Earlier generation · check reseller stock","dimensions":"197 × 197 × 95 mm","weight":"Not verified","power":"480 W maximum continuous power rating · not typical consumption","network":"Thunderbolt 5 · 10 GbE","configuration":"M3 Ultra · 28-core CPU / 60-core GPU / 96 GB / 1 TB","status":"Earlier generation","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"mac-llama32-3b","selectedHardwareId":"apple-mac-studio","referencePlatform":"Apple Silicon / macOS","evidence":{"origin":"MLX-LM maintainers","status":"Maintainer example","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://github.com/ml-explore/mlx-lm"},"model":{"name":"Llama 3.2 3B Instruct","checkpoint":"mlx-community/Llama-3.2-3B-Instruct-4bit","revision":null,"precision":"4-bit · MLX"},"serving":{"runtime":"MLX-LM","environment":"macOS / pin mlx-lm and mlx versions","backend":"MLX / Metal","contextTokens":null,"contextEvidence":"No Mac Studio load-test context is published in this record.","settings":"Use the MLX-converted checkpoint; record prompt length, cache settings and generation limits.","instructions":"https://github.com/ml-explore/mlx-lm"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Ecosystem-level guidance; no M3 Ultra or M5 Ultra reproduction is attached.","Available GPU memory and runtime support must be checked on the exact Mac.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"mac-mistral-7b","selectedHardwareId":"apple-mac-studio","referencePlatform":"Apple Silicon / macOS","evidence":{"origin":"MLX-LM maintainers","status":"Maintainer example","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://github.com/ml-explore/mlx-lm"},"model":{"name":"Mistral 7B Instruct v0.3","checkpoint":"mlx-community/Mistral-7B-Instruct-v0.3-4bit","revision":null,"precision":"4-bit · MLX"},"serving":{"runtime":"MLX-LM","environment":"macOS / pin mlx-lm and mlx versions","backend":"MLX / Metal","contextTokens":null,"contextEvidence":"No Mac Studio load-test context is published in this record.","settings":"Use the MLX-converted checkpoint; record prompt length, cache settings and generation limits.","instructions":"https://github.com/ml-explore/mlx-lm"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Ecosystem-level guidance; no M3 Ultra or M5 Ultra reproduction is attached.","Available GPU memory and runtime support must be checked on the exact Mac.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"mac-qwen3-8b","selectedHardwareId":"apple-mac-studio","referencePlatform":"Apple Silicon / macOS","evidence":{"origin":"MLX Community","status":"Community model card","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://huggingface.co/mlx-community/Qwen3-8B-4bit"},"model":{"name":"Qwen3 8B","checkpoint":"mlx-community/Qwen3-8B-4bit","revision":null,"precision":"4-bit · MLX"},"serving":{"runtime":"MLX-LM","environment":"macOS / pin mlx-lm and mlx versions","backend":"MLX / Metal","contextTokens":null,"contextEvidence":"No Mac Studio load-test context is published in this record.","settings":"Use the MLX-converted checkpoint; record prompt length, cache settings and generation limits.","instructions":"https://github.com/ml-explore/mlx-lm"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Ecosystem-level guidance; no M3 Ultra or M5 Ultra reproduction is attached.","Available GPU memory and runtime support must be checked on the exact Mac.","No exact-OEM benchmark or cost per user is published in this record."]}]}]}