{"schemaVersion":1,"publisher":"OnPremBench by Understand Tech","scope":"Sourced catalogue records; not a live price feed or measured capacity ranking","reuse":"Original sources and third-party image rights remain applicable. No blanket license is granted for third-party material.","machines":[{"id":"nvidia-dgx-spark","brand":"NVIDIA","name":"DGX Spark","chip":"GB10 Grace Blackwell Superchip","memory":128,"memoryText":"128 GB unified","bandwidth":"273 GB/s","platform":"NVIDIA GB10","os":"NVIDIA DGX OS","storage":"4 TB NVMe","form":"Desktop AI system","image":"/images/nvidia-dgx-spark-pny.png","source":"https://docs.nvidia.com/dgx/dgx-spark/hardware.html","tag":"UT field story","description":"A compact NVIDIA development system for exploring local models, applications and workflows.","notes":["Memory capacity does not establish interactive speed or simultaneous request capacity.","UT has documented a Spark deployment. Confirm exact software versions and support scope."],"managed":"documented","offer":{"amount":4699,"currency":"USD","region":"United States","url":"https://marketplace.nvidia.com/en-us/enterprise/personal-ai-supercomputers/dgx-spark/","configuration":"128 GB unified / 4 TB NVMe","tax":"Tax treatment not stated","date":"2026-09-07"},"reviewed":"2026-09-13","availability":"Listed for order · recheck with supplier","dimensions":"150 × 150 × 50.5 mm","weight":"1.2 kg","power":"240 W adapter · 140 W SoC TDP","network":"10 GbE · ConnectX-7 · Wi-Fi 7","configuration":"128 GB unified / 4 TB NVMe","status":"Catalogued","architecture":"arm64","evidence":"Supplier-documented specifications; exact-machine benchmarks not attached","configurations":[{"schemaVersion":2,"recordId":"spark-gpt-oss-20b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-20b","checkpoint":"openai/gpt-oss-20b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-gpt-oss-120b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"gpt-oss-120b","checkpoint":"openai/gpt-oss-120b","revision":null,"precision":"MXFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-llama-70b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Llama 3.3 70B Instruct","checkpoint":"nvidia/Llama-3.3-70B-Instruct-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-8b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 8B","checkpoint":"nvidia/Qwen3-8B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-14b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 14B","checkpoint":"nvidia/Qwen3-14B-FP8","revision":null,"precision":"FP8"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"Follow the checkpoint requirements in the upstream guide.","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]},{"schemaVersion":2,"recordId":"spark-qwen-32b","selectedHardwareId":"nvidia-dgx-spark","referencePlatform":"NVIDIA DGX Spark","evidence":{"origin":"NVIDIA","status":"Vendor-listed validation","reviewed":"2026-09-10","exactOemReproduced":false,"utReproduced":false,"source":"https://build.nvidia.com/spark/sglang"},"model":{"name":"Qwen3 32B","checkpoint":"nvidia/Qwen3-32B-FP4","revision":null,"precision":"NVFP4"},"serving":{"runtime":"SGLang","environment":"lmsysorg/sglang:latest-cu130","backend":"CUDA 13 / FlashInfer","contextTokens":8192,"contextEvidence":"Shared guide starting value; not a measured limit for this model.","settings":"--quantization modelopt_fp4","instructions":"https://build.nvidia.com/spark/sglang/instructions"},"measurements":{"provisionedUsers":null,"activeUsers":null,"activeWindow":null,"requestRatePerSecond":null,"simultaneousRequests":null,"p95TtftMs":null,"perStreamOutputTokensPerSecond":null,"aggregateOutputTokensPerSecond":null,"errorRate":null,"taskQuality":null,"wallPowerWatts":null,"rawEvidenceUrl":null},"limitations":["Mutable container tag; pin its digest and model revision.","Reference-platform evidence does not establish exact-OEM performance.","No exact-OEM benchmark or cost per user is published in this record."]}]}]}