{"data":{"capabilities":{"chat":true,"reasoning":false,"tools":true,"vision":false},"description":"Qwen3.6-35B-A3B GPTQ INT4 on one RTX 3090","engine":{"graph_mode":"full-and-piecewise","name":"vllm","version":"0.27.1"},"facts":{"metadata":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"not-observed","state":"unknown"},"serving.kv_cache_tokens":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"kv-cache-capacity-not-published","state":"unknown"},"serving.max_context_tokens":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"not-observed","state":"unknown"}},"hardware_count":1,"hardware_id":"rtx-3090-24gb","id":"qwen36-gptq-int4-rtx3090-vllm-tp1","launch":{"accelerator_backend":"nvidia","arguments":["--model","Twu31/Qwen3.6-35B-A3B-GPTQ-INT4-W4A16-LowLatency","--revision","d14fd7339b09364ea2aabc542ea20f7dccd97ca1","--served-model-name","Qwen3.6-35B","--tensor-parallel-size","1","--max-model-len","131072","--max-num-seqs","2","--kv-cache-dtype","fp8","--gpu-memory-utilization","0.97","--enable-auto-tool-choice","--tool-call-parser","qwen3_xml","--reasoning-parser","qwen3"],"container":{"captured_at":"2026-08-27T06:04:13.773Z","compose_file":null,"digest":"sha256:0a51ea5b4ae2dc5d81890e5173f54203d2a3ae0cfffe51b8fd2afd4391bfd967","image":"vllm/vllm-openai@sha256:0a51ea5b4ae2dc5d81890e5173f54203d2a3ae0cfffe51b8fd2afd4391bfd967","reason":"image-reference-in-launch","runtime":"docker","source":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"recipe-launch","url":"https://github.com/0xSero/local-ai-registry"}],"state":"digest-pinned"},"container_port":8000,"environment":{"VLLM_MEMORY_PROFILER_ESTIMATE_CUDAGRAPHS":"0"},"host_port":12434,"image":"vllm/vllm-openai@sha256:0a51ea5b4ae2dc5d81890e5173f54203d2a3ae0cfffe51b8fd2afd4391bfd967","ipc":"host","kind":"docker","mounts":[{"read_only":false,"source":"~/.cache/huggingface","target":"/root/.cache/huggingface"},{"read_only":false,"source":"~/.cache/inference-index/vllm","target":"/root/.cache/vllm"}],"network_mode":"bridge","shm_size":"16g"},"metadata":{},"model_instance_id":"twu31-qwen3-6-35b-a3b-gptq-int4-w4a16-lowlatency--int4-w4a16","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"normalized-recipe","url":"https://github.com/0xSero/local-ai-registry"}]},"recipe_source":"0xsero","schema_version":"local-ai-registry/v1","serving":{"kv_cache_tokens":null,"max_concurrency":4,"max_context_tokens":131072,"tensor_parallel":1},"speed_sweep_ids":["qwen36-gptq-int4-rtx3090-vllm-tp1-sweep"],"status":"validated","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/Twu31/Qwen3.6-35B-A3B-GPTQ-INT4-W4A16-LowLatency"}]},"reason":"hf-api-confirmed-public","repository":"Twu31/Qwen3.6-35B-A3B-GPTQ-INT4-W4A16-LowLatency","status":"known","url":"https://huggingface.co/Twu31/Qwen3.6-35B-A3B-GPTQ-INT4-W4A16-LowLatency"},"registry":{"launchable":true,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["qwen36-gptq-int4-rtx3090-vllm-tp1-sweep"],"detail_urls":["/api/v1/speed-sweep/qwen36-gptq-int4-rtx3090-vllm-tp1-sweep"]},"runtime":"docker"},"relationships":{"hardware":{"api":"/api/v1/hardware/rtx-3090-24gb","href":"/hardware/rtx-3090-24gb","id":"rtx-3090-24gb","name":"GeForce RTX 3090"},"model":{"api":"/api/v1/models/qwen3-6-35b-a3b","href":"/models/qwen3-6-35b-a3b","id":"qwen3-6-35b-a3b","name":"Qwen3.6-35B-A3B"},"model_instance":{"api":"/api/v1/model-instances/twu31-qwen3-6-35b-a3b-gptq-int4-w4a16-lowlatency--int4-w4a16","href":"/model-instances/twu31-qwen3-6-35b-a3b-gptq-int4-w4a16-lowlatency--int4-w4a16","id":"twu31-qwen3-6-35b-a3b-gptq-int4-w4a16-lowlatency--int4-w4a16","name":"Twu31/Qwen3.6-35B-A3B-GPTQ-INT4-W4A16-LowLatency"},"speed_sweep":[{"api":"/api/v1/speed-sweep/qwen36-gptq-int4-rtx3090-vllm-tp1-sweep","href":"/speed-sweep/qwen36-gptq-int4-rtx3090-vllm-tp1-sweep","id":"qwen36-gptq-int4-rtx3090-vllm-tp1-sweep"}]}},"meta":{"source":"registry"}}