{"data":{"capabilities":{"chat":null,"reasoning":null,"tools":null,"vision":null},"description":"Observed LocalMaxxing leaderboard run. Evidence for compatibility, not an executable launch contract.","engine":{"graph_mode":null,"name":"llama.cpp","version":"2e0e57f10"},"facts":{"capabilities.chat":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.reasoning":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.tools":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.vision":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"engine.graph_mode":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"runtime-detail-not-published","state":"unknown"},"serving.kv_cache_tokens":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"kv-cache-capacity-not-published","state":"unknown"},"serving.max_concurrency":{"provenance":{"captured_at":"2026-08-31T23:03:15Z","sources":[{"captured_at":"2026-08-31T23:03:15Z","kind":"registry-derived","url":"https://www.localmaxxing.com/en/runs/cmtcu7175000arw01u33ofyw2"}]},"reason":"server-capacity-derived-from-source-evidence","state":"known"}},"hardware_count":1,"hardware_id":"radeon-ai-pro-r9700-32gb","id":"glm-5-3-flash-ud-q2-k-xl-radeon-ai-pro-r9700-32gb-llama-cpp-tp1","launch":{"container":{"captured_at":"2026-08-30T09:10:02Z","compose_file":null,"digest":null,"image":null,"reason":"reference-only-launch","runtime":null,"source":[{"captured_at":"2026-08-30T09:10:02Z","kind":"recipe-launch","url":"https://www.localmaxxing.com/en/runs/cmtcu7175000arw01u33ofyw2"}],"state":"none"},"kind":"reference","run_id":"cmtcu7175000arw01u33ofyw2","source":"localmaxxing","url":"https://www.localmaxxing.com/en/runs/cmtcu7175000arw01u33ofyw2"},"metadata":{"localmaxxing":{"backend":"rocm","batch_size":1,"hardware_label":"Radeon AI Pro R9700","notes":"Tuned GLM-5.3-Flash UD-Q2_K_XL benchmark on Lucebox. Exact live configuration: --device ROCm0,ROCm1 --split-mode layer --tensor-split 35,82 --ctx-size 32768 --parallel 1 --reasoning off --reasoning-budget 0 --override-tensor blk.*.ffn_gate_shexp.weight=ROCm0,blk.*.ffn_up_shexp.weight=ROCm0,blk.*.ffn_down_shexp.weight=ROCm0 --metrics --jinja. The structured hardware record uses the AMD Radeon AI PRO R9700 (32 GB); the model was layer-split across the R9700 (ROCm0) and AMD Radeon 8060S unified-memory GPU (ROCm1) over M.2 OCuLink PCIe 4.0 x4. One slot, 32K context, temperature 0, reasoning disabled, cache_prompt=false. Three measured streamed requests after one warmup using the fixed 26-token count prompt and 64-token output cap; all three outputs matched. Peak workload allocation is measured from amdgpu DRM per-process VRAM+GTT counters across both GPUs and reported in GiB (101.919 GiB), not device capacity. Build/runtime: llama.cpp GLM-5-Next support PR commit 2e0e57f10. This is a short fixed-prompt comparison measurement; TTFT and total throughput are client-derived from monotonic stream timing, while prefill/decode rates are server-authoritative.","observed_command":"llama-server -hf unsloth/GLM-5.3-Flash-GGUF:UD-Q2_K_XL --device ROCm0,ROCm1 --split-mode layer --tensor-split 35,82 --ctx-size 32768 --parallel 1 --reasoning off --reasoning-budget 0 --override-tensor blk.*.ffn_gate_shexp.weight=ROCm0,blk.*.ffn_up_shexp.weight=ROCm0,blk.*.ffn_down_shexp.weight=ROCm0 --metrics --jinja","run_id":"cmtcu7175000arw01u33ofyw2","tokenized":{"arguments":["llama-server","-hf","unsloth/GLM-5.3-Flash-GGUF:UD-Q2_K_XL","--device","ROCm0,ROCm1","--split-mode","layer","--tensor-split","35,82","--ctx-size","32768","--parallel","1","--reasoning","off","--reasoning-budget","0","--override-tensor","blk.*.ffn_gate_shexp.weight=ROCm0,blk.*.ffn_up_shexp.weight=ROCm0,blk.*.ffn_down_shexp.weight=ROCm0","--metrics","--jinja"],"fidelity":"faithful"}}},"model_instance_id":"unsloth-glm-5-3-flash-gguf--ud-q2-k-xl","provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"normalized-recipe","url":"https://www.localmaxxing.com/en/runs/cmtcu7175000arw01u33ofyw2"}]},"recipe_source":"localmaxxing","schema_version":"local-ai-registry/v1","serving":{"kv_cache_tokens":null,"max_concurrency":1,"max_context_tokens":32768,"tensor_parallel":1},"speed_sweep_ids":["glm-5-3-flash-ud-q2-k-xl-radeon-ai-pro-r9700-32gb-llama-cpp-tp1-sweep"],"status":"candidate","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/unsloth/GLM-5.3-Flash-GGUF"}]},"reason":"hf-api-confirmed-public","repository":"unsloth/GLM-5.3-Flash-GGUF","status":"known","url":"https://huggingface.co/unsloth/GLM-5.3-Flash-GGUF"},"registry":{"launchable":false,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["glm-5-3-flash-ud-q2-k-xl-radeon-ai-pro-r9700-32gb-llama-cpp-tp1-sweep"],"detail_urls":["/api/v1/speed-sweep/glm-5-3-flash-ud-q2-k-xl-radeon-ai-pro-r9700-32gb-llama-cpp-tp1-sweep"]},"runtime":"reference"},"relationships":{"hardware":{"api":"/api/v1/hardware/radeon-ai-pro-r9700-32gb","href":"/hardware/radeon-ai-pro-r9700-32gb","id":"radeon-ai-pro-r9700-32gb","name":"Radeon AI PRO R9700"},"model":{"api":"/api/v1/models/glm-5-3-flash","href":"/models/glm-5-3-flash","id":"glm-5-3-flash","name":"GLM-5.3-Flash"},"model_instance":{"api":"/api/v1/model-instances/unsloth-glm-5-3-flash-gguf--ud-q2-k-xl","href":"/model-instances/unsloth-glm-5-3-flash-gguf--ud-q2-k-xl","id":"unsloth-glm-5-3-flash-gguf--ud-q2-k-xl","name":"unsloth/GLM-5.3-Flash-GGUF"},"speed_sweep":[{"api":"/api/v1/speed-sweep/glm-5-3-flash-ud-q2-k-xl-radeon-ai-pro-r9700-32gb-llama-cpp-tp1-sweep","href":"/speed-sweep/glm-5-3-flash-ud-q2-k-xl-radeon-ai-pro-r9700-32gb-llama-cpp-tp1-sweep","id":"glm-5-3-flash-ud-q2-k-xl-radeon-ai-pro-r9700-32gb-llama-cpp-tp1-sweep"}]}},"meta":{"source":"registry"}}