{"data":{"capabilities":{"chat":true,"reasoning":true,"tools":true,"vision":true},"description":"MiaAI-Lab GLM-5.2 hybrid NVFP4+AQLM vision profile across three DGX Sparks with TP3, DCP1, MTP-3, and full CUDA graphs","engine":{"graph_mode":"full","name":"vllm","version":"jarrelscy glm52-sm120 fork"},"facts":{"metadata":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"not-observed","state":"unknown"},"serving.kv_cache_tokens":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"kv-cache-capacity-not-published","state":"unknown"}},"hardware_count":3,"hardware_id":"dgx-spark-gb10-128gb","id":"glm52-nvfp4-aqlm-dgxspark-vllm-tp3","launch":{"accelerator_backend":"nvidia","container":{"captured_at":"2026-08-27T06:04:13.773Z","compose_file":null,"digest":null,"image":null,"reason":"non-container-launch-kind","runtime":null,"source":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"recipe-launch","url":"https://github.com/0xSero/local-ai-registry"}],"state":"none"},"container_port":8888,"environment":{"DCP_SIZE":"1","ENABLE_MTP":"1","GPU_MEM_UTIL":"0.895","HF_REVISION":"53e0082eedebd806b63e19779c47905937d768ca","KV_CACHE_DTYPE":"nvfp4_ds_mla","KV_CACHE_MEMORY_BYTES":"11811160064","MAX_MODEL_LEN":"348160","MAX_NUM_BATCHED_TOKENS":"4096","MAX_NUM_SEQS":"1","MTP_SPEC_TOKENS":"3","TP_SIZE":"3"},"host_port":8888,"ipc":"host","kind":"script","network_mode":"host","script":{"env_template":"sources/github/miaai-lab/glm52-nvfp4-aqlm-triple-dgx-spark/5c85163ccb8d98d395880e71e2dbd03976a3f4ad/.env.example","file":"sources/github/miaai-lab/glm52-nvfp4-aqlm-triple-dgx-spark/5c85163ccb8d98d395880e71e2dbd03976a3f4ad/start.sh"},"shm_size":"64g"},"metadata":{},"model_instance_id":"jarrelscy-glm-5-2-nvfp4-aqlm-hybrid--nvfp4-hot-experts-aqlm-2-bit-cold-experts","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"normalized-recipe","url":"https://github.com/0xSero/local-ai-registry"}]},"recipe_source":"mialabs","schema_version":"local-ai-registry/v1","serving":{"kv_cache_tokens":null,"max_concurrency":1,"max_context_tokens":40000,"tensor_parallel":3},"speed_sweep_ids":["glm52-nvfp4-aqlm-dgxspark-vllm-tp3-sweep"],"status":"candidate","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/jarrelscy/GLM-5.2-NVFP4-AQLM-hybrid"}]},"reason":"hf-api-confirmed-public","repository":"jarrelscy/GLM-5.2-NVFP4-AQLM-hybrid","status":"known","url":"https://huggingface.co/jarrelscy/GLM-5.2-NVFP4-AQLM-hybrid"},"registry":{"launchable":false,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["glm52-nvfp4-aqlm-dgxspark-vllm-tp3-sweep"],"detail_urls":["/api/v1/speed-sweep/glm52-nvfp4-aqlm-dgxspark-vllm-tp3-sweep"]},"runtime":"native"},"relationships":{"hardware":{"api":"/api/v1/hardware/dgx-spark-gb10-128gb","href":"/hardware/dgx-spark-gb10-128gb","id":"dgx-spark-gb10-128gb","name":"NVIDIA DGX Spark GB10"},"model":{"api":"/api/v1/models/glm-5-2","href":"/models/glm-5-2","id":"glm-5-2","name":"GLM-5.2"},"model_instance":{"api":"/api/v1/model-instances/jarrelscy-glm-5-2-nvfp4-aqlm-hybrid--nvfp4-hot-experts-aqlm-2-bit-cold-experts","href":"/model-instances/jarrelscy-glm-5-2-nvfp4-aqlm-hybrid--nvfp4-hot-experts-aqlm-2-bit-cold-experts","id":"jarrelscy-glm-5-2-nvfp4-aqlm-hybrid--nvfp4-hot-experts-aqlm-2-bit-cold-experts","name":"jarrelscy/GLM-5.2-NVFP4-AQLM-hybrid"},"speed_sweep":[{"api":"/api/v1/speed-sweep/glm52-nvfp4-aqlm-dgxspark-vllm-tp3-sweep","href":"/speed-sweep/glm52-nvfp4-aqlm-dgxspark-vllm-tp3-sweep","id":"glm52-nvfp4-aqlm-dgxspark-vllm-tp3-sweep"}]}},"meta":{"source":"registry"}}