{"speedTests":[{"id":"cmrxraev700wyo401gml7d1ll","modelId":"cmrvb4rvu09ezlg017mfmbl0m","modelRevision":"b482b5d57fda6e4e562a652869bde24ba2a57c92","hardwareId":"cmpw3cr6h003rr101f9z41zen","engineId":"cmrxo73bt00njo401nynvrqoq","userId":"cmqj0ywfi00hhnv01blw13jp4","promptTokens":8192,"outputTokens":1024,"contextLength":262144,"batchSize":1,"prefillTokens":null,"ttftMs":3488.54,"tokSOut":21.6,"tokSPrefill":2348.3,"tokSTotal":180.7,"peakVramGb":null,"gpuPowerWatts":[],"totalPowerWatts":null,"hardwareCost":null,"promptSha256":null,"promptSample":null,"outputSha256":null,"outputSample":null,"engineTimingsRaw":null,"verifiedRun":false,"verificationIssues":[],"notes":"Objective mixed-workload result on one NVIDIA DGX Spark GB10. Seven distinct semantic prompts were measured once each with no benchmark-prefix warmup: benchmark guidance, software design, incident review, data analysis, code generation, policy drafting, and system comparison. Six prompts were 8,192 endpoint tokens and one was 8,190; every request generated 1,024 tokens at c1. Decode samples were 21.6, 29.2, 18.2, 20.1, 40.1, 22.6, and 20.0 tok/s; median 21.6, mean 24.54, range 18.2-40.1. No sample was discarded. Fresh TTFT samples were 3,686.77, 3,480.58, 3,567.17, 3,456.60, 3,488.54, 3,457.77, and 3,492.21 ms; median 3,488.54 ms. Client-observed effective prefill estimates were 2,222.0, 2,353.6, 2,296.5, 2,369.4, 2,348.3, 2,369.2, and 2,345.8 prompt tok/s; median 2,348.3. Prefix caching remained enabled for the production configuration, but every prompt had a different first cache block. vLLM 0.25.1, official NVFP4 target b482b5d57fda6e4e562a652869bde24ba2a57c92, matched NVFP4 DFlash draft 723794750422b3efbf3a7b3af76dffb4ba035943, K=15, FP8 KV, FLASHINFER_CUTLASS, max_model_len=262144, max_num_seqs=32, gpu_memory_utilization=0.76. The 0.76 setting replaces Poolside's nominal 0.85 to preserve the fixed 12 GiB RAM safety floor. Minimum available RAM was 14.50 GiB and no guard event occurred. Prefill is prompt tokens divided by client-observed TTFT, not kernel-only throughput. Fixed-harness evidence: https://huggingface.co/spaces/osolmaz/local-frontier/blob/main/benchmarks/laguna-s-2-1-c1-20260723/README.md","adminNotes":null,"status":"APPROVED","suspiciousReason":null,"lastEditedAt":null,"createdAt":"2026-07-23T16:59:12.452Z","updatedAt":"2026-07-23T16:59:12.452Z","model":{"hfId":"poolside/Laguna-S-2.1-NVFP4","displayName":"Laguna-S-2.1-NVFP4","family":null,"params":118,"baseModelId":"cmrvb4rvm09exlg01ttv0044x"},"hardware":{"hwClass":"UNIFIED","gpuName":null,"gpuCount":1,"vramGb":null,"isHeterogeneousGpu":false,"chipVendor":"NVIDIA","chipFamily":"GB10","chipVariant":"GB10 Grace Blackwell","unifiedMemoryGb":128,"npuTops":null,"cpu":null,"ramGb":null,"os":"Ubuntu 24.04","powerWatts":null,"gpuSlots":[]},"engine":{"engineName":"vllm","engineVersion":"0.25.1","engineRepository":null,"engineBuild":null,"engineCommit":null,"quantization":"NVFP4"},"engineFlags":{"commandSnippet":"vllm serve poolside/Laguna-S-2.1-NVFP4 --served-model-name poolside/Laguna-S-2.1-NVFP4 --speculative-config '{\"model\":\"poolside/Laguna-S-2.1-DFlash-NVFP4\",\"num_speculative_tokens\":15}' --enable-auto-tool-choice --tool-call-parser poolside_v1 --reasoning-parser poolside_v1 --override-generation-config '{\"temperature\":0.7,\"top_p\":0.95}' --max-num-seqs 32 --max-model-len 262144 --gpu-memory-utilization 0.76 --host 0.0.0.0 --port 8000","tensorParallel":1,"pipelineParallel":null,"gpuLayers":null,"splitMode":null,"kvCacheDtype":"fp8","gpuMemUtil":0.76,"kvCacheSizeMb":null,"prefixCaching":null,"attentionBackend":"FLASHINFER","flashAttn":null,"chunkedPrefill":null,"prefillChunkSize":null,"contBatching":null,"cpuOffloadGb":null,"cpuLayers":null,"ropeScaling":null,"ropeScale":null,"yarnExtFactor":null,"engineQuant":null,"sglangQuant":null,"maxRunningSeqs":32,"schedulerDelayFactor":null,"numParallel":null,"concurrency":null,"specDecoding":true,"specMethod":"dflash","specModel":null,"specDraftModel":"poolside/Laguna-S-2.1-DFlash-NVFP4","specNumTokens":15,"specNgramSize":null,"specDraftTp":null,"specDraftWindowSize":null,"mtpEnabled":false,"mtpDraftLayers":null,"specDraftTokens":null,"specAcceptedTokens":null,"specAcceptanceRate":null,"specMeanAcceptedLength":null,"temperature":null,"topP":0.95,"topK":null,"minP":null,"repeatPenalty":null,"mirostat":null,"extraFlags":"CUDA graphs enabled; FLASHINFER_CUTLASS NVFP4 MoE backend; poolside_v1 tool and reasoning parsers"},"user":{"id":"cmqj0ywfi00hhnv01blw13jp4","name":"Onur Solmaz","username":"osolmaz","verified":false,"verifiedAt":null}}],"total":1,"limit":20,"offset":0}