This is my OpenCode setup for local models, mainly using a customized llama.cpp configuration. A developer detailed their OpenCode setup for running local AI models, centered on a customized llama.cpp configuration. The setup leverages reasoning-effort levels for models like Qwen 3.8, DeepSeek V4, and Glimmer, and applies token budgets for models without explicit levels. The configuration includes a llama-server command with flags for reasoning budgets and context preservation, plus an OpenCode configuration file for integrating local llama.cpp and vLLM providers. This is my OpenCode setup for local models, mainly using a customized llama.cpp configuration. For Qwen 3.8, DeepSeek V4, and Glimmer, the models are already trained to support reasoning effort levels. Depending on the model, these may be exposed as low , medium , high , xhigh , or as low , high , and max . For models that support reasoning but were not trained with explicit reasoning-effort levels, such as the Qwen 3.5 and 3.6 variants, I use a token budget to limit the amount of reasoning. Although Qwen 3.8 has built-in reasoning-effort levels, I still apply a maximum reasoning-token cap for each effort level. Although the llama.cpp CLI flags specify preserve thinking and a default reasoning budget, these can still be overridden through the API, so this works fine for my setup. Yes, there is also an xhigh-no-preserve variant. In this mode, the model uses its reasoning as a scratchpad without preserving it in the conversation history. I use this when I do not want the reasoning output to unnecessarily consume the context window. Most of the time, I use low reasoning. /home/USER/llama.cpp/llama-server \ --model /mnt/d/MODEL STORE/LLM SETUP/Qwen3.8-27B/Qwen3.8-27B-UD-Q5 K XL.gguf \ --alias Qwen3.8-27B \ --host 0.0.0.0 \ --chat-template-file /mnt/d/MODEL STORE/LLM SETUP/Qwen3.8-27B/chat template.jinja \ --no-context-shift \ --metrics \ --kv-unified \ --cache-ram 16384 \ --ctx-size 81920 \ --port 8001 \ --cache-type-k q8 0 \ --cache-type-v q8 0 \ --flash-attn on \ --temp 1.0 \ --top-p 0.95 \ --top-k 20 \ --min-p 0.0 \ --presence-penalty 0.0 \ --repeat-penalty 1.0 \ --jinja \ --chat-template-kwargs '{"preserve thinking": true}' \ --spec-type draft-mtp \ --spec-draft-n-max 2 \ --reasoning-budget 8192 \ -np 1 \ -ub 256 https://huggingface.co/froggeric/Qwen-Fixed-Chat-Templates https://huggingface.co/froggeric/Qwen-Fixed-Chat-Templates nvim ~/.config/opencode/opencode.json { "$schema": "https://opencode.ai/config.json", "plugin": "@tarquinen/opencode-dcp@latest" , "provider": { "openai": { "options": { "headerTimeout": 60000, "timeout": 600000, "chunkTimeout": 60000 } }, "local-vllm": { "npm": "@ai-sdk/openai-compatible", "name": "vLLM Local ", "options": { "baseURL": "http://localhost:8000/v1" }, "models": { "cyankiwi/Nemotron-Orchestrator-8B-AWQ-4bit": { "name": "cyankiwi/Nemotron-Orchestrator-8B-AWQ-4bit", "max tokens": 40960 } } }, "local-llamacpp": { "npm": "@ai-sdk/openai-compatible", "name": "llama.cpp Local ", "options": { "baseURL": "http://localhost:8001/v1" }, "models": { "Qwen3.6-35B": { "name": "Qwen3.6-35B", "max tokens": 131072, "modalities": { "input": "image", "text" , "output": "text" }, "variants": { "none": { "chat template kwargs": { "enable thinking": false } }, "low": { "reasoning budget tokens": 512 }, "medium": { "reasoning budget tokens": 2048 }, "xhigh": { "reasoning budget tokens": 8192 }, "xhigh-no-preserve": { "reasoning budget tokens": 8192, "chat template kwargs": { "preserve thinking": false } } } }, "Qwen3.8-27B": { "name": "Qwen3.8-27B", "max tokens": 81920, "modalities": { "input": "image", "text" , "output": "text" }, "variants": { "none": { "reasoningEffort": "none" }, "low": { "reasoningEffort": "low", "reasoning budget tokens": 512 }, "medium": { "reasoningEffort": "medium", "reasoning budget tokens": 2048 }, "xhigh": { "reasoningEffort": "xhigh", "reasoning budget tokens": 8192 }, "xhigh-no-preserve": { "reasoningEffort": "xhigh", "reasoning budget tokens": 8192, "chat template kwargs": { "preserve thinking": false } } } }, "Muse-Glimmer-30B": { "name": "Muse-Glimmer-30B", "max tokens": 81920, "modalities": { "input": "image", "text" , "output": "text" }, "variants": { "low": { "reasoningEffort": "low", "reasoning budget tokens": 512 }, "medium": { "reasoningEffort": "medium", "reasoning budget tokens": 2048 }, "high": { "reasoningEffort": "high", "reasoning budget tokens": 4096 }, "xhigh": { "reasoningEffort": "xhigh", "reasoning budget tokens": 8192 } } }, "DeepSeek-V4-Flash-0731": { "name": "DeepSeek-V4-Flash-0731", "max tokens": 81920, "variants": { "low": { "reasoningEffort": "low", "reasoning budget tokens": 512 }, "high": { "reasoningEffort": "high", "reasoning budget tokens": 4096 }, "max": { "reasoningEffort": "max", "reasoning budget tokens": 8192 } } }, "Omnicoder-2-9B": { "name": "Omnicoder-2-9B", "max tokens": 131072, "modalities": { "input": "image", "text" , "output": "text" }, "variants": { "none": { "chat template kwargs": { "enable thinking": false } }, "low": { "reasoning budget tokens": 512 }, "medium": { "reasoning budget tokens": 2048 }, "xhigh": { "reasoning budget tokens": 8192 }, "xhigh-no-preserve": { "reasoning budget tokens": 8192, "chat template kwargs": { "preserve thinking": false } } } }, "Ornith-9B": { "name": "Ornith-9B", "max tokens": 131072, "modalities": { "input": "image", "text" , "output": "text" }, "variants": { "none": { "chat template kwargs": { "enable thinking": false } }, "low": { "reasoning budget tokens": 512 }, "medium": { "reasoning budget tokens": 2048 }, "xhigh": { "reasoning budget tokens": 8192 }, "xhigh-no-preserve": { "reasoning budget tokens": 8192, "chat template kwargs": { "preserve thinking": false } } } } } }, "local-ninfer": { "npm": "@ai-sdk/openai-compatible", "name": "Ninfer Windows ", "options": { "baseURL": "http://localhost:10909/v1" }, "models": { "qwen3.8-27b": { "name": "Qwen3.8-27B", "max tokens": 81920, "variants": { "none": { "reasoningEffort": "none" }, "low": { "reasoningEffort": "low" }, "medium": { "reasoningEffort": "medium" }, "xhigh": { "reasoningEffort": "xhigh" } } } } } } } - 3090 - WSL2 Ubuntu 24.04 - 6.18.33.2-microsoft-standard-WSL2 - E5 2690v4 - 96G RAM For WSL2 , Total 128G, DDR4 2400 ECC - Qwen 3.8 27B KV Q8:Q8 81920 : 30 tok/s - Qwen 3.6 35B A3B KV Q8:Q8 180224: 75 tok/s - DSv4 Flash KV Q8:Q8 65536: 6 tok/s