{"data":{"capabilities":{"chat":true,"reasoning":true,"tools":true,"vision":true},"description":"Saved two-DGX-Spark Inkling Small NVFP4 vLLM profile with MTP1, BF16 KV, SM121 paged-KV overlays, full and piecewise CUDA graphs, and text/image/audio support; blocked from CLI until the host-specific two-node bundle and model revision are independently replayed","engine":{"graph_mode":"full-and-piecewise","name":"vllm","version":"65b7662d3fcb773afaf751ab29ac6960a0cf011d+sm121-overlays"},"facts":{"metadata":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"not-observed","state":"unknown"}},"hardware_count":2,"hardware_id":"dgx-spark-gb10-128gb","id":"inkling-small-nvfp4-dgxspark-vllm-tp2","launch":{"accelerator_backend":"nvidia","compose":{"file":"compose.yml","project_name":"inkling-small-tp2"},"container":{"captured_at":"2026-08-27T06:04:13.773Z","compose_file":"compose.yml","digest":null,"image":null,"reason":"compose-manifest-resolves-image-outside-normalized-record","runtime":"docker-compose","source":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"recipe-launch","url":"https://github.com/0xSero/local-ai-registry"}],"state":"indirect"},"container_port":8000,"environment":{"CUTE_DSL_ARCH":"sm_121a","GPU_MEMORY_UTILIZATION":"0.9","INKLING_MODEL":"/models/Inkling-Small-NVFP4","INKLING_VLLM_IMAGE":"local/inkling-vllm:20260812-65b7662d-sm121-paged-a80c4d7-audio1","KV_CACHE_DTYPE":"auto","KV_CACHE_MEMORY_BYTES":"29450000000","MAX_MODEL_LEN":"262144","MAX_NUM_BATCHED_TOKENS":"8192","MAX_NUM_SEQS":"4","MTP_NUM_TOKENS":"1","PIPELINE_PARALLEL_SIZE":"1","SERVED_MODEL_NAME":"inkling-small","TENSOR_PARALLEL_SIZE":"2"},"host_port":8000,"ipc":"host","kind":"docker-compose","network_mode":"host","shm_size":"64g"},"metadata":{},"model_instance_id":"thinkingmachines-inkling-small-nvfp4--nvfp4","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"normalized-recipe","url":"https://github.com/0xSero/local-ai-registry"}]},"recipe_source":"0xsero","schema_version":"local-ai-registry/v1","serving":{"kv_cache_tokens":420162,"max_concurrency":4,"max_context_tokens":262144,"tensor_parallel":2},"speed_sweep_ids":["inkling-small-nvfp4-dgxspark-vllm-tp2-sweep"],"status":"candidate","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/thinkingmachines/Inkling-Small-NVFP4"}]},"reason":"hf-api-confirmed-public","repository":"thinkingmachines/Inkling-Small-NVFP4","status":"known","url":"https://huggingface.co/thinkingmachines/Inkling-Small-NVFP4"},"registry":{"launchable":false,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["inkling-small-nvfp4-dgxspark-vllm-tp2-sweep"],"detail_urls":["/api/v1/speed-sweep/inkling-small-nvfp4-dgxspark-vllm-tp2-sweep"]},"runtime":"docker"},"relationships":{"hardware":{"api":"/api/v1/hardware/dgx-spark-gb10-128gb","href":"/hardware/dgx-spark-gb10-128gb","id":"dgx-spark-gb10-128gb","name":"NVIDIA DGX Spark GB10"},"model":{"api":"/api/v1/models/inkling-small","href":"/models/inkling-small","id":"inkling-small","name":"Inkling-Small"},"model_instance":{"api":"/api/v1/model-instances/thinkingmachines-inkling-small-nvfp4--nvfp4","href":"/model-instances/thinkingmachines-inkling-small-nvfp4--nvfp4","id":"thinkingmachines-inkling-small-nvfp4--nvfp4","name":"thinkingmachines/Inkling-Small-NVFP4"},"speed_sweep":[{"api":"/api/v1/speed-sweep/inkling-small-nvfp4-dgxspark-vllm-tp2-sweep","href":"/speed-sweep/inkling-small-nvfp4-dgxspark-vllm-tp2-sweep","id":"inkling-small-nvfp4-dgxspark-vllm-tp2-sweep"}]}},"meta":{"source":"registry"}}