{"data":{"capabilities":{"chat":true,"reasoning":false,"tools":false,"vision":false},"description":"Two-B70 Qwen3.8-27B Q4_K_M llama.cpp SYCL candidate. A fresh live screen found four 65,536-token slots, correct chat output, and C1/C2/C4 measurements through 8K, but the checked-in parallel-4 profile cannot expose the claimed 128K per-request context and becomes strongly asymmetric under concurrency.","engine":{"graph_mode":"not-applicable","name":"llama-cpp","version":"4302fb59969a5d8cf9f8e5f55fdd4506d0ed2126+b70-patches"},"facts":{"metadata":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"not-observed","state":"unknown"},"serving.kv_cache_tokens":{"provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"kv-cache-capacity-not-published","state":"unknown"}},"hardware_count":2,"hardware_id":"intel-arc-pro-b70-32gb","id":"qwen38-q4km-arcb70-llamacpp-tp2","launch":{"accelerator_backend":"intel-xpu","compose":{"file":"compose.yml","project_name":"qwen38-b70-tp2"},"container":{"captured_at":"2026-08-27T06:04:13.773Z","compose_file":"compose.yml","digest":null,"image":null,"reason":"compose-manifest-resolves-image-outside-normalized-record","runtime":"docker-compose","source":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"recipe-launch","url":"https://github.com/0xSero/local-ai-registry"}],"state":"indirect"},"container_port":8010,"environment":{"BATCH":"8192","DRAFT_NGL":"0","DRAFT_THREADS":"16","ENABLE_MTP":"1","ENABLE_VISION":"0","GPU_COUNT":"2","PARALLEL":"4","SPEC_DRAFT_N_MAX":"5","THREADS":"16","UBATCH":"8192"},"host_port":8010,"ipc":"host","kind":"docker-compose","network_mode":"bridge","shm_size":"8g"},"metadata":{},"model_instance_id":"ggml-org-qwen3-8-27b-gguf--q4-k-m","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"normalized-recipe","url":"https://github.com/0xSero/local-ai-registry"}]},"recipe_source":"0xsero","schema_version":"local-ai-registry/v1","serving":{"kv_cache_tokens":null,"max_concurrency":4,"max_context_tokens":8192,"tensor_parallel":2},"speed_sweep_ids":["qwen38-q4km-arcb70-llamacpp-tp2-sweep"],"status":"candidate","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-27T06:04:13.773Z","sources":[{"captured_at":"2026-08-27T06:04:13.773Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/ggml-org/Qwen3.8-27B-GGUF"}]},"reason":"hf-api-confirmed-public","repository":"ggml-org/Qwen3.8-27B-GGUF","status":"known","url":"https://huggingface.co/ggml-org/Qwen3.8-27B-GGUF"},"registry":{"launchable":false,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["qwen38-q4km-arcb70-llamacpp-tp2-sweep"],"detail_urls":["/api/v1/speed-sweep/qwen38-q4km-arcb70-llamacpp-tp2-sweep"]},"runtime":"docker"},"relationships":{"hardware":{"api":"/api/v1/hardware/intel-arc-pro-b70-32gb","href":"/hardware/intel-arc-pro-b70-32gb","id":"intel-arc-pro-b70-32gb","name":"Intel Arc Pro B70"},"model":{"api":"/api/v1/models/qwen3-8-27b","href":"/models/qwen3-8-27b","id":"qwen3-8-27b","name":"Qwen3.8-27B"},"model_instance":{"api":"/api/v1/model-instances/ggml-org-qwen3-8-27b-gguf--q4-k-m","href":"/model-instances/ggml-org-qwen3-8-27b-gguf--q4-k-m","id":"ggml-org-qwen3-8-27b-gguf--q4-k-m","name":"ggml-org/Qwen3.8-27B-GGUF"},"speed_sweep":[{"api":"/api/v1/speed-sweep/qwen38-q4km-arcb70-llamacpp-tp2-sweep","href":"/speed-sweep/qwen38-q4km-arcb70-llamacpp-tp2-sweep","id":"qwen38-q4km-arcb70-llamacpp-tp2-sweep"}]}},"meta":{"source":"registry"}}