{"data":{"accepted_at":null,"id":"glm53-flash-nvfp4-rtxpro6000-sglang-tp4-sweep","measured_at":"2026-08-26","metrics":{"concurrency":8,"inference_engine_version":"lmsysorg/sglang:glm-5.3-flash@sha256:3a97bd50034ca60c6e6c86b8e36a73675d261f6a5eb71197796aee5175409290","latest_point_at":"2026-08-26","max_context_tokens":213094,"peak_generation_tps":550,"peak_prompt_tps":6383,"point_count":10},"recipe_id":"glm53-flash-nvfp4-rtxpro6000-sglang-tp4","rows":[{"concurrency":1,"context_tokens":null,"decode_tok_s":142,"decode_tok_s_per_stream":141.8,"output_tokens":512,"peak_vram_gb":null,"prefill_tok_s":null,"samples":1,"status":"observed","ttft_ms_p50":124},{"concurrency":2,"context_tokens":null,"decode_tok_s":240.4,"decode_tok_s_per_stream":123,"output_tokens":1024,"peak_vram_gb":null,"prefill_tok_s":null,"samples":2,"status":"observed","ttft_ms_p50":252},{"concurrency":4,"context_tokens":null,"decode_tok_s":381.8,"decode_tok_s_per_stream":100.3,"output_tokens":2048,"peak_vram_gb":null,"prefill_tok_s":null,"samples":4,"status":"observed","ttft_ms_p50":256},{"concurrency":8,"context_tokens":null,"decode_tok_s":550,"decode_tok_s_per_stream":72.5,"output_tokens":4096,"peak_vram_gb":null,"prefill_tok_s":null,"samples":8,"status":"observed","ttft_ms_p50":332},{"concurrency":1,"context_tokens":2125,"decode_tok_s":null,"decode_tok_s_per_stream":null,"output_tokens":0,"peak_vram_gb":null,"prefill_tok_s":5344,"samples":1,"status":"observed","ttft_ms_p50":398},{"concurrency":1,"context_tokens":6325,"decode_tok_s":null,"decode_tok_s_per_stream":null,"output_tokens":0,"peak_vram_gb":null,"prefill_tok_s":6383,"samples":1,"status":"observed","ttft_ms_p50":991},{"concurrency":1,"context_tokens":21407,"decode_tok_s":null,"decode_tok_s_per_stream":null,"output_tokens":0,"peak_vram_gb":null,"prefill_tok_s":6275,"samples":1,"status":"observed","ttft_ms_p50":3411},{"concurrency":1,"context_tokens":53315,"decode_tok_s":null,"decode_tok_s_per_stream":null,"output_tokens":0,"peak_vram_gb":null,"prefill_tok_s":6180,"samples":1,"status":"observed","ttft_ms_p50":8627},{"concurrency":1,"context_tokens":106726,"decode_tok_s":null,"decode_tok_s_per_stream":null,"output_tokens":0,"peak_vram_gb":null,"prefill_tok_s":6029,"samples":1,"status":"observed","ttft_ms_p50":17702},{"concurrency":1,"context_tokens":213094,"decode_tok_s":null,"decode_tok_s_per_stream":null,"output_tokens":0,"peak_vram_gb":null,"prefill_tok_s":5783,"samples":1,"status":"observed","ttft_ms_p50":36850}],"schema_version":"local-ai-registry/v1","source":{"commit":"46ea34371842cfce939f4f76b5e60c1e99235f57","paths":["Dockerfile","manifest.json","compose.yaml","README.md","patches/"],"repository":"https://github.com/0xSero/glm53-flash-nvfp4-sm120-exact-docker"},"relationships":{"recipe":{"api":"/api/v1/recipes/glm53-flash-nvfp4-rtxpro6000-sglang-tp4","href":"/recipes/glm53-flash-nvfp4-rtxpro6000-sglang-tp4","id":"glm53-flash-nvfp4-rtxpro6000-sglang-tp4"}}},"meta":{"source":"registry"}}