{"data":{"capabilities":{"chat":null,"reasoning":null,"tools":null,"vision":null},"description":"Observed LocalMaxxing leaderboard run. Evidence for compatibility, not an executable launch contract.","engine":{"graph_mode":null,"name":"vllm","version":"0.20.1-local"},"facts":{"capabilities.chat":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.reasoning":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.tools":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.vision":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"engine.graph_mode":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"runtime-detail-not-published","state":"unknown"},"serving.kv_cache_tokens":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"kv-cache-capacity-not-published","state":"unknown"}},"hardware_count":4,"hardware_id":"intel-arc-pro-b70-32gb","id":"minimax-m2-7-int4-autoround-w4a16-intel-arc-pro-b70-32gb-vllm-tp4-uqa329e6","launch":{"container":{"captured_at":"2026-08-30T09:10:02Z","compose_file":null,"digest":null,"image":null,"reason":"reference-only-launch","runtime":null,"source":[{"captured_at":"2026-08-30T09:10:02Z","kind":"recipe-launch","url":"https://www.localmaxxing.com/en/runs/cmp4f31dh000amz01uqa329e6"}],"state":"none"},"kind":"reference","run_id":"cmp4f31dh000amz01uqa329e6","source":"localmaxxing","url":"https://www.localmaxxing.com/en/runs/cmp4f31dh000amz01uqa329e6"},"metadata":{"localmaxxing":{"backend":null,"hardware_label":"Intel Arc Pro B70","notes":"MiniMax M2.7 AutoRound W4A16 on 4x Intel Arc Pro B70 via vLLM/XPU TP4. Local weights are Lasimeri/MiniMax-M2.7-int4-AutoRound submitted under base MiniMaxAI/MiniMax-M2.7. Same recipe as prior 73.23 run, now repeated at 73.31 output tok/s and 97.74 total tok/s at p512/n1536. This is a variance-confirmed measured high, not a new quality-changing optimization. No model weight, quantization, router precision, expert routing, KV dtype, sampler, speculative decoding, or power-limit changes. Q/K RMS variance allreduces remain preserved. Total tok/s includes the 512-token prompt plus 1536 decoded tokens; no isolated prefill-only metric is emitted by this vLLM bench mode.","observed_command":"VLLM_MINIMAX_M2_ATTN_DELAY_ALLREDUCE=1 VLLM_XPU_ENABLE_XPU_GRAPH=1 VLLM_XPU_FORCE_GRAPH_WITH_COMM=1 VLLM_XPU_GRAPH_NOOP_COMM_CAPTURE=1 VLLM_XPU_USE_LLM_SCALER_MOE=1 CCL_TOPO_P2P_ACCESS=1 VLLM_CACHE_ROOT=/mnt/fast-ai/vllm-cache-exp/minimax-xpugraph-attndelay-block256-mbt512-noprefix-20260513T171301Z vllm bench throughput --backend vllm --async-engine --block-size 256 --no-enable-prefix-caching --model /mnt/fast-ai/llm-models/minimax-m2.7-int4-autoround --tokenizer /mnt/fast-ai/llm-models/minimax-m2.7-int4-autoround --trust-remote-code --dtype float16 --tensor-parallel-size 4 --distributed-executor-backend mp --max-model-len 2048 --max-num-batched-tokens 512 --max-num-seqs 1 --dataset-name random --random-input-len 512 --random-output-len 1536 --random-range-ratio 0 --num-prompts 1 --disable-log-stats --compilation-config '{\"use_inductor_graph_partition\":true,\"compile_sizes\":[1],\"cudagraph_mode\":\"PIECEWISE\"}'","run_id":"cmp4f31dh000amz01uqa329e6","tokenized":{"arguments":["vllm","bench","throughput","--backend","vllm","--async-engine","--block-size","256","--no-enable-prefix-caching","--model","/mnt/fast-ai/llm-models/minimax-m2.7-int4-autoround","--tokenizer","/mnt/fast-ai/llm-models/minimax-m2.7-int4-autoround","--trust-remote-code","--dtype","float16","--tensor-parallel-size","4","--distributed-executor-backend","mp","--max-model-len","2048","--max-num-batched-tokens","512","--max-num-seqs","1","--dataset-name","random","--random-input-len","512","--random-output-len","1536","--random-range-ratio","0","--num-prompts","1","--disable-log-stats","--compilation-config","{\"use_inductor_graph_partition\":true,\"compile_sizes\":[1],\"cudagraph_mode\":\"PIECEWISE\"}"],"environment":{"CCL_TOPO_P2P_ACCESS":"1","VLLM_CACHE_ROOT":"/mnt/fast-ai/vllm-cache-exp/minimax-xpugraph-attndelay-block256-mbt512-noprefix-20260513T171301Z","VLLM_MINIMAX_M2_ATTN_DELAY_ALLREDUCE":"1","VLLM_XPU_ENABLE_XPU_GRAPH":"1","VLLM_XPU_FORCE_GRAPH_WITH_COMM":"1","VLLM_XPU_GRAPH_NOOP_COMM_CAPTURE":"1","VLLM_XPU_USE_LLM_SCALER_MOE":"1"},"fidelity":"faithful"}}},"model_instance_id":"minimaxai-minimax-m2-7--int4-autoround-w4a16","provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"normalized-recipe","url":"https://www.localmaxxing.com/en/runs/cmp4f31dh000amz01uqa329e6"}]},"recipe_source":"localmaxxing","schema_version":"local-ai-registry/v1","serving":{"kv_cache_tokens":null,"max_concurrency":1,"max_context_tokens":2048,"tensor_parallel":4},"speed_sweep_ids":["minimax-m2-7-int4-autoround-w4a16-intel-arc-pro-b70-32gb-vllm-tp4-uqa329e6-sweep"],"status":"candidate","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/MiniMaxAI/MiniMax-M2.7"}]},"reason":"hf-api-confirmed-public","repository":"MiniMaxAI/MiniMax-M2.7","status":"known","url":"https://huggingface.co/MiniMaxAI/MiniMax-M2.7"},"registry":{"launchable":false,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["minimax-m2-7-int4-autoround-w4a16-intel-arc-pro-b70-32gb-vllm-tp4-uqa329e6-sweep"],"detail_urls":["/api/v1/speed-sweep/minimax-m2-7-int4-autoround-w4a16-intel-arc-pro-b70-32gb-vllm-tp4-uqa329e6-sweep"]},"runtime":"reference"},"relationships":{"hardware":{"api":"/api/v1/hardware/intel-arc-pro-b70-32gb","href":"/hardware/intel-arc-pro-b70-32gb","id":"intel-arc-pro-b70-32gb","name":"Intel Arc Pro B70"},"model":{"api":"/api/v1/models/minimax-m2-7","href":"/models/minimax-m2-7","id":"minimax-m2-7","name":"MiniMax-M2.7"},"model_instance":{"api":"/api/v1/model-instances/minimaxai-minimax-m2-7--int4-autoround-w4a16","href":"/model-instances/minimaxai-minimax-m2-7--int4-autoround-w4a16","id":"minimaxai-minimax-m2-7--int4-autoround-w4a16","name":"MiniMaxAI/MiniMax-M2.7"},"speed_sweep":[{"api":"/api/v1/speed-sweep/minimax-m2-7-int4-autoround-w4a16-intel-arc-pro-b70-32gb-vllm-tp4-uqa329e6-sweep","href":"/speed-sweep/minimax-m2-7-int4-autoround-w4a16-intel-arc-pro-b70-32gb-vllm-tp4-uqa329e6-sweep","id":"minimax-m2-7-int4-autoround-w4a16-intel-arc-pro-b70-32gb-vllm-tp4-uqa329e6-sweep"}]}},"meta":{"source":"registry"}}