{"data":{"capabilities":{"chat":null,"reasoning":null,"tools":null,"vision":null},"description":"Observed LocalMaxxing leaderboard run. Evidence for compatibility, not an executable launch contract.","draft_launch":{"accelerator_backend":"nvidia","arguments":["-hf","ornith-ai/Ornith-1.0-35B-GGUF:Q4_K_M","--n-gpu-layers","999","--host","0.0.0.0","--port","8080","-c","262144"],"container_port":8080,"entrypoint":null,"environment":{"LLAMA_CACHE":"/root/.cache/huggingface"},"host_port":8080,"image":"ghcr.io/ggml-org/llama.cpp:server-cuda12-b10481@sha256:b2497f8834f5ecb4e38530f6bf2734b8e0be107f0f0857e259672d1cb85b71c2","ipc":"host","kind":"docker","mounts":[{"read_only":false,"source":"~/.cache/huggingface","target":"/root/.cache/huggingface"}],"shm_size":"16g","synthesized":{"generated_at":"2026-08-31T22:12:17Z","image_provenance":"gemma-4-12b-q4-k-m-rtx-3060-12gb-llama-cpp-tp1","template":"llama-cpp-server-v1"}},"engine":{"graph_mode":null,"name":"llama.cpp","version":"8655 (277ff5fff)"},"facts":{"capabilities.chat":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.reasoning":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.tools":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"capabilities.vision":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"capability-not-verified","state":"unknown"},"engine.graph_mode":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"runtime-detail-not-published","state":"unknown"},"serving.kv_cache_tokens":{"provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"registry-enrichment","url":"https://github.com/0xSero/local-ai-registry"}]},"reason":"kv-cache-capacity-not-published","state":"unknown"}},"hardware_count":1,"hardware_id":"rtx-3090-24gb","id":"ornith-1-0-35b-gguf-q4-k-m-rtx-3090-24gb-llama-cpp-tp1","launch":{"container":{"captured_at":"2026-08-30T09:10:02Z","compose_file":null,"digest":null,"image":null,"reason":"reference-only-launch","runtime":null,"source":[{"captured_at":"2026-08-30T09:10:02Z","kind":"recipe-launch","url":"https://www.localmaxxing.com/en/runs/cmqw5k2wu039dqr01j2yh9kg5"}],"state":"none"},"kind":"reference","run_id":"cmqw5k2wu039dqr01j2yh9kg5","source":"localmaxxing","url":"https://www.localmaxxing.com/en/runs/cmqw5k2wu039dqr01j2yh9kg5"},"metadata":{"localmaxxing":{"backend":"cuda","hardware_label":"RTX 3090","notes":"Ornith-1.0 35B GGUF Q4_K_M on 1x RTX 3090 via llama.cpp CUDA, full 262144 context, q4_0 KV cache, flash-attn, no prompt cache, single streaming request, 5 measured 1000-token runs after 1 warmup, no-thinking template forced with chat_template_kwargs enable_thinking=false. single GPU. Performance-window run with GPU power autotune paused and test GPUs capped at 350W. GPU reached 89C and software thermal slowdown was observed in 22 telemetry samples, so later runs tapered. Host is PCIe 3.0/no NVLink; this row uses one card.","observed_command":"llama-server --model ornith-1.0-35b-Q4_K_M.gguf --ctx-size 262144 --cache-type-k q4_0 --cache-type-v q4_0 --parallel 1 --batch-size 1024 --ubatch-size 256 --flash-attn on --n-gpu-layers 999 --jinja --reasoning-format deepseek --reasoning auto --no-cache-prompt --cache-ram 0","run_id":"cmqw5k2wu039dqr01j2yh9kg5","tokenized":{"arguments":["llama-server","--model","ornith-1.0-35b-Q4_K_M.gguf","--ctx-size","262144","--cache-type-k","q4_0","--cache-type-v","q4_0","--parallel","1","--batch-size","1024","--ubatch-size","256","--flash-attn","on","--n-gpu-layers","999","--jinja","--reasoning-format","deepseek","--reasoning","auto","--no-cache-prompt","--cache-ram","0"],"fidelity":"faithful"}}},"model_instance_id":"ornith-ai-ornith-1-0-35b-gguf--q4-k-m","provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"normalized-recipe","url":"https://www.localmaxxing.com/en/runs/cmqw5k2wu039dqr01j2yh9kg5"}]},"recipe_source":"localmaxxing","schema_version":"local-ai-registry/v1","serving":{"kv_cache_tokens":null,"max_concurrency":1,"max_context_tokens":262144,"tensor_parallel":1},"speed_sweep_ids":["ornith-1-0-35b-gguf-q4-k-m-rtx-3090-24gb-llama-cpp-tp1-sweep"],"status":"candidate","huggingface":{"link_type":"repository","provenance":{"captured_at":"2026-08-30T09:10:02Z","sources":[{"captured_at":"2026-08-30T09:10:02Z","kind":"huggingface-api","url":"https://huggingface.co/api/models/ornith-ai/Ornith-1.0-35B-GGUF"}]},"reason":"hf-api-confirmed-public","repository":"ornith-ai/Ornith-1.0-35B-GGUF","status":"known","url":"https://huggingface.co/ornith-ai/Ornith-1.0-35B-GGUF"},"registry":{"launchable":false,"speed_evidence":{"available":true,"count":1,"speed_sweep_ids":["ornith-1-0-35b-gguf-q4-k-m-rtx-3090-24gb-llama-cpp-tp1-sweep"],"detail_urls":["/api/v1/speed-sweep/ornith-1-0-35b-gguf-q4-k-m-rtx-3090-24gb-llama-cpp-tp1-sweep"]},"runtime":"reference"},"relationships":{"hardware":{"api":"/api/v1/hardware/rtx-3090-24gb","href":"/hardware/rtx-3090-24gb","id":"rtx-3090-24gb","name":"GeForce RTX 3090"},"model":{"api":"/api/v1/models/ornith-1-0-35b-gguf","href":"/models/ornith-1-0-35b-gguf","id":"ornith-1-0-35b-gguf","name":"Ornith-1.0-35B-GGUF"},"model_instance":{"api":"/api/v1/model-instances/ornith-ai-ornith-1-0-35b-gguf--q4-k-m","href":"/model-instances/ornith-ai-ornith-1-0-35b-gguf--q4-k-m","id":"ornith-ai-ornith-1-0-35b-gguf--q4-k-m","name":"ornith-ai/Ornith-1.0-35B-GGUF"},"speed_sweep":[{"api":"/api/v1/speed-sweep/ornith-1-0-35b-gguf-q4-k-m-rtx-3090-24gb-llama-cpp-tp1-sweep","href":"/speed-sweep/ornith-1-0-35b-gguf-q4-k-m-rtx-3090-24gb-llama-cpp-tp1-sweep","id":"ornith-1-0-35b-gguf-q4-k-m-rtx-3090-24gb-llama-cpp-tp1-sweep"}]}},"meta":{"source":"registry"}}