{"data":{"capabilities":{"chat":true,"reasoning":true,"tools":null,"vision":null},"description":"Runtime-tested selective EXL3 Q4 deployment on four RTX PRO 6000 Blackwell GPUs. Routed experts in layers 3-44 use 4.0 bpw EXL3; the backbone remains BF16. The accepted TP4/EP1 path keeps full CUDA graphs and FP8 E4M3 KV. A sustained C256, ISL 131, OSL 1024 screen delivered 102,609.85 output tokens/minute. Tool and vision execution remain unvalidated, and the locally built runtime image is not yet published by digest, so this remains a candidate.","engine":{"graph_mode":"full decode capture planned to 256 concurrent requests (runtime-captured graph buckets through batch 62) and full prefill capture at batch size 1","name":"sglang","version":"0.0.0.dev1+gf609d677b with Sparkinfer 1.0.1 selective-EXL3 overlay"},"facts":{},"hardware_count":4,"hardware_id":"rtx-pro-6000-blackwell-96gb","id":"glm53-flash-exl3-q4-rtxpro6000-sglang-tp4","launch":{"accelerator_backend":"nvidia","arguments":["--model-path","/model","--served-model-name","glm-5.3-flash-exl3-q4","--tp-size","4","--ep-size","1","--quantization","exl3","--context-length","262144","--kv-cache-dtype","fp8_e4m3","--attention-backend","dsa","--dsa-prefill-backend","flashinfer_sparse_mla","--dsa-decode-backend","flashinfer_sparse_mla","--linear-attn-backend","triton","--disable-shared-experts-fusion","--chunked-prefill-size","256","--max-prefill-tokens","256","--max-running-requests","256","--mem-fraction-static","0.80","--cuda-graph-backend-decode","full","--cuda-graph-backend-prefill","full","--cuda-graph-max-bs-decode","256","--cuda-graph-max-bs-prefill","1","--reasoning-parser","glm45","--tool-call-parser","glm47","--host","127.0.0.1","--port","8000"],"container":{"captured_at":"2026-08-27T22:05:00Z","compose_file":null,"digest":null,"image":"glm53-flash-sglang-exl3:stage-20260827-r6","reason":"runtime-tested-local-image-not-yet-published-by-digest","runtime":"docker","source":[{"captured_at":"2026-08-27T22:05:00Z","kind":"runtime-acceptance","url":"https://github.com/0xSero/local-ai-registry"}],"state":"mutable"},"environment":{"SGLANG_EXL3_MAX_BATCH_TOKENS":"256"},"image":"glm53-flash-sglang-exl3:stage-20260827-r6","ipc":"host","kind":"docker","mounts":[{"read_only":true,"source":"${MODEL_ROOT}/GLM-5.3-Flash-EXL3-Q4","target":"/model"},{"read_only":true,"source":"${RUNTIME_ASSETS}/chat-template-mm.jinja","target":"/chat-template.jinja"}],"shm_size":"32g"},"metadata":{"acceptance":{"artifact_forward_kl_bf16_to_q4":0.06579,"artifact_perplexity_delta_percent":2.38,"artifact_top1_agreement_percent":91.7,"cuda_graph_capture":true,"endpoint_health":true,"generated_completion":true,"model_discovery":true,"tools":null,"vision":null},"benchmark":{"accepted_single_stream_output_tok_s":73.6191,"cold_1024_token_prefill_tok_s":827.7402,"matched_ep4_candidate_output_tok_s":563.981,"max_sustained_output_tok_s":1710.1642,"max_sustained_output_tokens_per_minute":102609.852,"max_sustained_profile":"C256, measured ISL 131, requested OSL 1024, warmed shared prefix","max_sustained_total_api_tokens_per_minute":115736.6983,"mtp5_candidate_output_tok_s":64.8608},"checkpoint_layout":"glm53-selective-exl3-tp4-v1","controller_recipe_id":"glm-5.3-flash-exl3-q4","quality_note":"Artifact KLD is held-out BF16-versus-Q4 evaluation. A new full-server aligned-logit KLD run has not been published.","runtime_state":"runtime-tested","selection":"TP4/EP1 accepted; sustained C256 is the max-output-TPM profile; C32/C64 remain the lower-latency profiles; virtual-slice EP4, MTP5, shared-expert fusion, single-batch overlap, and incomplete GLM-5 Next TBO candidates were rejected"},"model_instance_id":"0xsero-glm-5-3-flash-exl3-q4--selective-exl3-q4","provenance":{"captured_at":"2026-08-27T22:05:00Z","sources":[{"captured_at":"2026-08-27T22:05:00Z","kind":"artifact","url":"https://huggingface.co/0xSero/GLM-5.3-Flash-EXL3-Q4"},{"captured_at":"2026-08-27T22:05:00Z","kind":"runtime-acceptance","url":"https://github.com/0xSero/local-ai-registry"}]},"recipe_source":"0xsero","schema_version":"local-ai-registry/v1","serving":{"expert_parallel":1,"kv_cache_dtype":"fp8_e4m3","max_concurrency":256,"max_context_tokens":262144,"tensor_parallel":4},"speed_sweep_ids":["glm53-flash-exl3-q4-rtxpro6000-sglang-tp4-sweep"],"status":"candidate","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-27T22:05:00Z","sources":[{"captured_at":"2026-08-27T22:05:00Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/0xSero/GLM-5.3-Flash-EXL3-Q4"}]},"reason":"hf-api-confirmed-public","repository":"0xSero/GLM-5.3-Flash-EXL3-Q4","status":"known","url":"https://huggingface.co/0xSero/GLM-5.3-Flash-EXL3-Q4"},"registry":{"launchable":false,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["glm53-flash-exl3-q4-rtxpro6000-sglang-tp4-sweep"],"detail_urls":["/api/v1/speed-sweep/glm53-flash-exl3-q4-rtxpro6000-sglang-tp4-sweep"]},"runtime":"docker"},"relationships":{"hardware":{"api":"/api/v1/hardware/rtx-pro-6000-blackwell-96gb","href":"/hardware/rtx-pro-6000-blackwell-96gb","id":"rtx-pro-6000-blackwell-96gb","name":"RTX PRO 6000 Blackwell"},"model":{"api":"/api/v1/models/glm-5-3-flash","href":"/models/glm-5-3-flash","id":"glm-5-3-flash","name":"GLM-5.3-Flash"},"model_instance":{"api":"/api/v1/model-instances/0xsero-glm-5-3-flash-exl3-q4--selective-exl3-q4","href":"/model-instances/0xsero-glm-5-3-flash-exl3-q4--selective-exl3-q4","id":"0xsero-glm-5-3-flash-exl3-q4--selective-exl3-q4","name":"0xSero/GLM-5.3-Flash-EXL3-Q4"},"speed_sweep":[{"api":"/api/v1/speed-sweep/glm53-flash-exl3-q4-rtxpro6000-sglang-tp4-sweep","href":"/speed-sweep/glm53-flash-exl3-q4-rtxpro6000-sglang-tp4-sweep","id":"glm53-flash-exl3-q4-rtxpro6000-sglang-tp4-sweep"}]}},"meta":{"source":"registry"}}