This is my OpenCode setup for local models, mainly using a customized llama.cpp configuration.
For Qwen 3.8, DeepSeek V4, and Glimmer, the models are already trained to support reasoning effort levels. Depending on the model, these may be exposed as low
, medium
, high
, xhigh
, or as low
, high
, and max
.
For models that support reasoning but were not trained with explicit reasoning-effort levels, such as the Qwen 3.5 and 3.6 variants, I use a token budget to limit the amount of reasoning.
Although Qwen 3.8 has built-in reasoning-effort levels, I still apply a maximum reasoning-token cap for each effort level.
Although the llama.cpp CLI flags specify preserve_thinking
and a default reasoning budget, these can still be overridden through the API, so this works fine for my setup.
Yes, there is also an xhigh-no-preserve
variant. In this mode, the model uses its reasoning as a scratchpad without preserving it in the conversation history. I use this when I do not want the reasoning output to unnecessarily consume the context window.
Most of the time, I use low
reasoning.
/home/USER/llama.cpp/llama-server \
--model /mnt/d/MODEL_STORE/LLM_SETUP/Qwen3.8-27B/Qwen3.8-27B-UD-Q5_K_XL.gguf \
--alias Qwen3.8-27B \
--host 0.0.0.0 \
--chat-template-file /mnt/d/MODEL_STORE/LLM_SETUP/Qwen3.8-27B/chat_template.jinja \
--no-context-shift \
--metrics \
--kv-unified \
--cache-ram 16384 \
--ctx-size 81920 \
--port 8001 \
--cache-type-k q8_0 \
--cache-type-v q8_0 \
--flash-attn on \
--temp 1.0 \
--top-p 0.95 \
--top-k 20 \
--min-p 0.0 \
--presence-penalty 0.0 \
--repeat-penalty 1.0 \
--jinja \
--chat-template-kwargs '{"preserve_thinking": true}' \
--spec-type draft-mtp \
--spec-draft-n-max 2 \
--reasoning-budget 8192 \
-np 1 \
-ub 256
https://huggingface.co/froggeric/Qwen-Fixed-Chat-Templates
nvim ~/.config/opencode/opencode.json
{
"$schema": "https://opencode.ai/config.json",
"plugin": [
"@tarquinen/opencode-dcp@latest"
],
"provider": {
"openai": {
"options": {
"headerTimeout": 60000,
"timeout": 600000,
"chunkTimeout": 60000
}
},
"local-vllm": {
"npm": "@ai-sdk/openai-compatible",
"name": "vLLM (Local)",
"options": {
"baseURL": "http://localhost:8000/v1"
},
"models": {
"cyankiwi/Nemotron-Orchestrator-8B-AWQ-4bit": {
"name": "cyankiwi/Nemotron-Orchestrator-8B-AWQ-4bit",
"max_tokens": 40960
}
}
},
"local-llamacpp": {
"npm": "@ai-sdk/openai-compatible",
"name": "llama.cpp (Local)",
"options": {
"baseURL": "http://localhost:8001/v1"
},
"models": {
"Qwen3.6-35B": {
"name": "Qwen3.6-35B",
"max_tokens": 131072,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"variants": {
"none": {
"chat_template_kwargs": {
"enable_thinking": false
}
},
"low": {
"reasoning_budget_tokens": 512
},
"medium": {
"reasoning_budget_tokens": 2048
},
"xhigh": {
"reasoning_budget_tokens": 8192
},
"xhigh-no-preserve": {
"reasoning_budget_tokens": 8192,
"chat_template_kwargs": {
"preserve_thinking": false
}
}
}
},
"Qwen3.8-27B": {
"name": "Qwen3.8-27B",
"max_tokens": 81920,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"variants": {
"none": {
"reasoningEffort": "none"
},
"low": {
"reasoningEffort": "low",
"reasoning_budget_tokens": 512
},
"medium": {
"reasoningEffort": "medium",
"reasoning_budget_tokens": 2048
},
"xhigh": {
"reasoningEffort": "xhigh",
"reasoning_budget_tokens": 8192
},
"xhigh-no-preserve": {
"reasoningEffort": "xhigh",
"reasoning_budget_tokens": 8192,
"chat_template_kwargs": {
"preserve_thinking": false
}
}
}
},
"Muse-Glimmer-30B": {
"name": "Muse-Glimmer-30B",
"max_tokens": 81920,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"variants": {
"low": {
"reasoningEffort": "low",
"reasoning_budget_tokens": 512
},
"medium": {
"reasoningEffort": "medium",
"reasoning_budget_tokens": 2048
},
"high": {
"reasoningEffort": "high",
"reasoning_budget_tokens": 4096
},
"xhigh": {
"reasoningEffort": "xhigh",
"reasoning_budget_tokens": 8192
}
}
},
"DeepSeek-V4-Flash-0731": {
"name": "DeepSeek-V4-Flash-0731",
"max_tokens": 81920,
"variants": {
"low": {
"reasoningEffort": "low",
"reasoning_budget_tokens": 512
},
"high": {
"reasoningEffort": "high",
"reasoning_budget_tokens": 4096
},
"max": {
"reasoningEffort": "max",
"reasoning_budget_tokens": 8192
}
}
},
"Omnicoder-2-9B": {
"name": "Omnicoder-2-9B",
"max_tokens": 131072,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"variants": {
"none": {
"chat_template_kwargs": {
"enable_thinking": false
}
},
"low": {
"reasoning_budget_tokens": 512
},
"medium": {
"reasoning_budget_tokens": 2048
},
"xhigh": {
"reasoning_budget_tokens": 8192
},
"xhigh-no-preserve": {
"reasoning_budget_tokens": 8192,
"chat_template_kwargs": {
"preserve_thinking": false
}
}
}
},
"Ornith-9B": {
"name": "Ornith-9B",
"max_tokens": 131072,
"modalities": {
"input": [
"image",
"text"
],
"output": [
"text"
]
},
"variants": {
"none": {
"chat_template_kwargs": {
"enable_thinking": false
}
},
"low": {
"reasoning_budget_tokens": 512
},
"medium": {
"reasoning_budget_tokens": 2048
},
"xhigh": {
"reasoning_budget_tokens": 8192
},
"xhigh-no-preserve": {
"reasoning_budget_tokens": 8192,
"chat_template_kwargs": {
"preserve_thinking": false
}
}
}
}
}
},
"local-ninfer": {
"npm": "@ai-sdk/openai-compatible",
"name": "Ninfer (Windows)",
"options": {
"baseURL": "http://localhost:10909/v1"
},
"models": {
"qwen3.8-27b": {
"name": "Qwen3.8-27B",
"max_tokens": 81920,
"variants": {
"none": {
"reasoningEffort": "none"
},
"low": {
"reasoningEffort": "low"
},
"medium": {
"reasoningEffort": "medium"
},
"xhigh": {
"reasoningEffort": "xhigh"
}
}
}
}
}
}
}
-
3090
-
WSL2 Ubuntu 24.04
-
6.18.33.2-microsoft-standard-WSL2
-
E5 2690v4
-
96G RAM (For WSL2), Total 128G, DDR4 2400 ECC
-
Qwen 3.8 27B KV Q8:Q8 81920 : 30 tok/s
-
Qwen 3.6 35B A3B KV Q8:Q8 180224: 75 tok/s
-
DSv4 Flash KV Q8:Q8 65536: 6 tok/s