{ "run_id": "6f89eb6649bb4b839b3f8e2ff373608b", "timestamp": "2026-07-28T05:39:29+00:00", "target_id": "qwen3.6-35b-a3b-nvfp4@your-node", "target": { "id": "qwen3.6-35b-a3b-nvfp4@your-node", "display_name": "Qwen3.6-35B-A3B (NVFP4)", "serving": { "engine": "vllm", "model_name": "brain", "base_url": "http://your-node:8001/v1", "service": "spark", "max_model_len": 65536, "flags": { "enforce_eager": true, "gpu_memory_utilization": 0.55, "kv_cache_dtype": "fp8", "max_num_batched_tokens": 4096, "moe_backend": "flashinfer_b12x", "reasoning_parser": "qwen3", "tool_call_parser": "qwen3_coder", "enable_auto_tool_choice": true } }, "host": "your-node", "family": "qwen3.6", "params_b": 35, "active_params_b": 3, "quant": "nvfp4", "checkpoint": null, "tier": "production", "aliases": [ "brain", "spark" ], "notes": "Live behind the `brain` and `local-moe` aliases. This is the model the bench exists to interrogate: it was deployed without measurement. FLAGS OF CONCERN: --enforce-eager disables CUDA graphs (costs throughput) and --gpu-memory-utilization 0.55 leaves ~45% of unified memory unused, capping KV cache and therefore concurrency. Both are worth an A/B once the baseline is recorded.\n", "slug": "qwen3.6-35b-a3b-nvfp4__your-node", "inspect_model": "openai-api/spark/brain" }, "host": { "id": "your-node", "ssh": "your-node", "hardware": "NVIDIA GB10 Grace Blackwell (DGX Spark)", "memory_gb": 121, "memory_bandwidth_gbs": 273, "notes": "Unified LPDDR5X. Memory bandwidth is the binding constraint on decode throughput, not compute -- large dense models are bandwidth-starved here while MoE models with small active-param counts do comparatively well.\n" }, "quality": [], "perf": { "engine": "vllm", "points": [ { "concurrency": 1, "input_tokens": 1024, "output_tokens": 256, "n_requests": 1, "completed": 1, "failed": 0, "duration_s": 8.309, "output_tps_total": 30.81, "output_tps_per_stream": 30.81, "ttft_p50_ms": 314.74, "ttft_p95_ms": 314.74, "tpot_p50_ms": 31.35, "prefill_tps": 4530.7, "error": null }, { "concurrency": 8, "input_tokens": 1024, "output_tokens": 256, "n_requests": 8, "completed": 8, "failed": 0, "duration_s": 11.002, "output_tps_total": 186.16, "output_tps_per_stream": 23.32, "ttft_p50_ms": 1918.86, "ttft_p95_ms": 2048.13, "tpot_p50_ms": 35.58, "prefill_tps": 838.2, "error": null }, { "concurrency": 32, "input_tokens": 1024, "output_tokens": 256, "n_requests": 32, "completed": 32, "failed": 0, "duration_s": 32.638, "output_tps_total": 251.0, "output_tps_per_stream": 8.07, "ttft_p50_ms": 5044.49, "ttft_p95_ms": 8067.19, "tpot_p50_ms": 104.93, "prefill_tps": 289.0, "error": null } ], "peak_memory_gb": null, "idle_memory_gb": null, "notes": null }, "verdict": { "signal_score": null, "reference_score": null, "single_stream_tps": 30.8, "interactive_viable": true, "peak_throughput_tps": 251.0, "peak_throughput_concurrency": 32 }, "runner": { "host": "workstation", "python": "3.12.3", "platform": "Linux-7.0.0-28-generic-x86_64-with-glibc2.39" }, "schema_version": 1 }