{"data":{"accepted_at":null,"id":"deepseek-v4-flash-0731-fp8-fp4-rtxpro6000-sglang-tp2-sweep","measured_at":"2026-08-26","metrics":{"concurrency":4,"inference_engine_version":"lmsysorg/sglang:dev-cu13-inkling-dspark@sha256:fbea1a4e25b26660dbc2384a27ead8817e9b7670f257b5c3143e0450d14524d7 (stock; SGLang commit b7252cc6b0c78b25ecea7ee5efa91a6ae37d0f19); no patches","latest_point_at":"2026-08-26","max_context_tokens":50,"peak_generation_tps":155.2,"peak_prompt_tps":null,"point_count":5},"recipe_id":"deepseek-v4-flash-0731-fp8-fp4-rtxpro6000-sglang-tp2","rows":[{"concurrency":1,"context_tokens":30,"context_tokens_note":"estimated prompt length (bench JSON records no prompt_tokens)","decode_tok_s":57.85,"decode_tok_s_per_stream":57.85,"effective_tok_s_incl_ttft":56.15,"output_tokens":118,"peak_vram_gb":null,"prefill_tok_s":null,"prompt_class":"short","samples":2,"status":"measured","ttft_ms_p50":80},{"concurrency":1,"context_tokens":40,"context_tokens_note":"estimated prompt length (bench JSON records no prompt_tokens)","decode_tok_s":57.8,"decode_tok_s_per_stream":57.8,"effective_tok_s_incl_ttft":57.6,"output_tokens":987,"peak_vram_gb":null,"prefill_tok_s":null,"prompt_class":"reasoning","samples":2,"status":"measured","ttft_ms_p50":84},{"concurrency":1,"context_tokens":40,"context_tokens_note":"estimated prompt length (bench JSON records no prompt_tokens)","decode_tok_s":57.85,"decode_tok_s_per_stream":57.85,"effective_tok_s_incl_ttft":57.7,"output_tokens":1995,"peak_vram_gb":null,"prefill_tok_s":null,"prompt_class":"analytical","samples":2,"status":"measured","ttft_ms_p50":80},{"concurrency":1,"context_tokens":50,"context_tokens_note":"estimated prompt length (bench JSON records no prompt_tokens)","decode_tok_s":57.75,"decode_tok_s_per_stream":57.75,"effective_tok_s_incl_ttft":57.7,"output_tokens":2500,"peak_vram_gb":null,"prefill_tok_s":null,"prompt_class":"long_2500cap","samples":2,"status":"measured","ttft_ms_p50":80},{"concurrency":4,"context_tokens":40,"context_tokens_note":"estimated prompt length (bench JSON records no prompt_tokens)","decode_tok_s":155.2,"decode_tok_s_per_stream":47.2,"effective_tok_s_incl_ttft":155.2,"output_tokens":1611,"peak_vram_gb":null,"prefill_tok_s":null,"prompt_class":"reasoning_x4","samples":1,"status":"measured","ttft_ms_p50":364}],"schema_version":"local-ai-registry/v1","source":{"kind":"submitter","methodology":"OpenAI /v1/chat/completions, streaming (SSE) with usage in the final chunk; temperature 0.7 / top_p 0.8; four prompt classes x 2 runs (short 400 cap, reasoning, analytical, long 2500 cap); effective tok/s = completion_tokens (reasoning + content, from usage) / wall incl. TTFT; TTFT = first streamed delta of content or reasoning; decode tok/s excludes TTFT; 4-way concurrency on the reasoning prompt, aggregate = total completion tokens / batch wall. Idle server, GPUs at the 300 W Max-Q limit. Raw JSON + script in the source repo benchmarks/.","paths":["benchmarks/deepseek-v4-flash-0731-tp2-tps-20260826.json"],"url":"https://github.com/ppickle1989/sm120-sglang-recipes"},"relationships":{"recipe":{"api":"/api/v1/recipes/deepseek-v4-flash-0731-fp8-fp4-rtxpro6000-sglang-tp2","href":"/recipes/deepseek-v4-flash-0731-fp8-fp4-rtxpro6000-sglang-tp2","id":"deepseek-v4-flash-0731-fp8-fp4-rtxpro6000-sglang-tp2"}}},"meta":{"source":"registry"}}