{"slug": "claude-mtplx-wrapper-to-streamline-running-claude-code-against-mtplx", "title": "claude-mtplx wrapper to streamline running Claude Code against mtplx", "summary": "A developer released claude-mtplx, a bash wrapper that starts a local MTPLX model server if one isn't already listening and then execs the Claude Code CLI against the Anthropic-compatible endpoint. The script exposes environment overrides for host, port, runtime profile, paged KV quantization (off/q8/q4), SSD session cache, permission mode, and readiness timeout, and warns when an already-running server's profile differs from the requested one. The author notes the SSD session cache measured zero cached tokens per request against a long-lived server, so it defaults to off.", "body_md": "|  | #!/usr/bin/env bash | \n|  | # claude-mtplx — run the Claude Code CLI against a locally served MTPLX model. | \n|  | # | \n|  | # Starts the MTPLX server if it isn't already listening, then execs `claude` | \n|  | # with the Anthropic-compatible endpoint pointed at it. Any arguments are | \n|  | # passed straight through to claude, e.g.: | \n|  | # | \n|  | # claude-mtplx # interactive session | \n|  | # claude-mtplx -p \"explain this\" # one-shot | \n|  | # claude-mtplx --mtplx-stop # stop the server and exit | \n|  | # | \n|  | # Note: --profile and the KV-quant setting only apply when this script STARTS | \n|  | # the server. If one is already listening, it is reused exactly as it was | \n|  | # launched; the script warns when its profile differs from the requested one. | \n|  | # | \n|  | # Environment overrides: | \n|  | # MTPLX_HOST, MTPLX_PORT where the server listens (127.0.0.1:8000) | \n|  | # MTPLX_PROFILE runtime profile (turbo) | \n|  | # MTPLX_KV_QUANT paged KV quantization: off\\|q8\\|q4 (q8) | \n|  | # MTPLX_SSD_CACHE SSD session cache: off\\|on\\|write-only (off) | \n|  | # MTPLX_PERMISSION_MODE claude --permission-mode value (manual) | \n|  | # MTPLX_EXCLUDE_DYNAMIC 1 to keep the prompt prefix stable, 0 to disable (1) | \n|  | # MTPLX_API_KEY auth token, if the server requires one | \n|  | # MTPLX_READY_TIMEOUT seconds to wait for model load (300) | \n|  | set -euo pipefail | \n|  | HOST=\"${MTPLX_HOST:-127.0.0.1}\" | \n|  | PORT=\"${MTPLX_PORT:-8000}\" | \n|  | PROFILE=\"${MTPLX_PROFILE:-turbo}\" | \n|  | KV_QUANT=\"${MTPLX_KV_QUANT:-q8}\" # off\\|q8\\|q4 - q8 roughly halves KV bytes/token | \n|  | # Explicit: MTPLX's app setting is \"target-default\", which resolves to ON, so | \n|  | # simply omitting the flag does NOT disable the SSD cache. Measured 0 cached | \n|  | # tokens on every request against a long-lived server; its only plausible value | \n|  | # is across server restarts, which was not tested. | \n|  | SSD_CACHE=\"${MTPLX_SSD_CACHE:-off}\" # off\\|on\\|write-only | \n|  | READY_TIMEOUT=\"${MTPLX_READY_TIMEOUT:-300}\" | \n|  | BASE=\"http://${HOST}:${PORT}\" | \n|  | LOG_DIR=\"${HOME}/.mtplx/logs\" | \n|  | LOG=\"${LOG_DIR}/claude-mtplx.log\" | \n|  | die() { printf 'claude-mtplx: %s\\n' \"$*\" >&2; exit 1; } | \n|  | note() { printf 'claude-mtplx: %s\\n' \"$*\" >&2; } | \n|  | health() { curl -sf -m 3 \"${BASE}/health\" 2>/dev/null; } | \n|  | command -v mtplx >/dev/null 2>&1 \\|\\| die \"mtplx not found on PATH\" | \n|  | command -v claude >/dev/null 2>&1 \\|\\| die \"claude not found on PATH\" | \n|  | command -v jq >/dev/null 2>&1 \\|\\| die \"jq not found on PATH\" | \n|  | if [ \"${1:-}\" = \"--mtplx-stop\" ]; then | \n|  | mtplx stop --port \"$PORT\" | \n|  | exit $? | \n|  | fi | \n|  | # --- start the server if it isn't already answering ------------------------- | \n|  | if ! health >/dev/null; then | \n|  | note \"starting MTPLX on ${BASE} (first model load takes ~1 min)\" | \n|  | mkdir -p \"$LOG_DIR\" | \n|  | nohup mtplx quickstart \\ | \n|  | --profile \"$PROFILE\" \\ | \n|  | --host \"$HOST\" \\ | \n|  | --port \"$PORT\" \\ | \n|  | --paged-kv-quantization \"$KV_QUANT\" \\ | \n|  | --ssd-session-cache \"$SSD_CACHE\" \\ | \n|  | --no-stats-footer >>\"$LOG\" 2>&1 & | \n|  | server_pid=$! | \n|  | waited=0 | \n|  | until health >/dev/null; do | \n|  | if ! kill -0 \"$server_pid\" 2>/dev/null; then | \n|  | note \"server exited during startup; last lines of ${LOG}:\" | \n|  | tail -20 \"$LOG\" >&2 | \n|  | exit 1 | \n|  | fi | \n|  | if [ \"$waited\" -ge \"$READY_TIMEOUT\" ]; then | \n|  | die \"server not ready after ${READY_TIMEOUT}s (see ${LOG})\" | \n|  | fi | \n|  | sleep 2 | \n|  | waited=$((waited + 2)) | \n|  | done | \n|  | note \"server ready after ${waited}s (log: ${LOG})\" | \n|  | fi | \n|  | # --- read what the server is actually serving ------------------------------- | \n|  | info=\"$(health)\" \\|\\| die \"health check failed at ${BASE}\" | \n|  | MODEL=\"$(printf '%s' \"$info\" \\| jq -r '.model // empty')\" | \n|  | [ -n \"$MODEL\" ] \\|\\| die \"could not determine served model id from ${BASE}/health\" | \n|  | CTX=\"$(printf '%s' \"$info\" \\| jq -r '.context_window // empty')\" | \n|  | # The profile and KV-quant flags above only shape a server this script launches. | \n|  | # If one was already up (started by hand, or by MTPLX.app), it keeps whatever | \n|  | # settings it was launched with. | \n|  | RUNNING_PROFILE=\"$(printf '%s' \"$info\" \\| jq -r '.profile.name // empty')\" | \n|  | if [ -n \"$RUNNING_PROFILE\" ] && [ \"$RUNNING_PROFILE\" != \"$PROFILE\" ]; then | \n|  | note \"reusing a running server on profile '${RUNNING_PROFILE}' (wanted '${PROFILE}')\" | \n|  | note \"restart to apply: claude-mtplx --mtplx-stop && claude-mtplx\" | \n|  | fi | \n|  | # --- point Claude Code at it ------------------------------------------------ | \n|  | # ANTHROPIC_API_KEY must be absent: if both it and AUTH_TOKEN are set, the key | \n|  | # wins and Claude Code tries to authenticate against api.anthropic.com. | \n|  | unset ANTHROPIC_API_KEY | \n|  | export ANTHROPIC_BASE_URL=\"$BASE\" | \n|  | export ANTHROPIC_AUTH_TOKEN=\"${MTPLX_API_KEY:-mtplx-local}\" | \n|  | # All four aliases matter: Claude Code routes background work (conversation | \n|  | # titles, subagents) to the Haiku/Sonnet aliases, which would otherwise | \n|  | # resolve to real Anthropic model ids this server does not serve. | \n|  | export ANTHROPIC_MODEL=\"$MODEL\" | \n|  | export ANTHROPIC_DEFAULT_OPUS_MODEL=\"$MODEL\" | \n|  | export ANTHROPIC_DEFAULT_SONNET_MODEL=\"$MODEL\" | \n|  | export ANTHROPIC_DEFAULT_HAIKU_MODEL=\"$MODEL\" | \n|  | export CLAUDE_CODE_SUBAGENT_MODEL=\"$MODEL\" | \n|  | # The model id is not in Claude Code's catalog, so it would otherwise assume a | \n|  | # 200k window for auto-compact. Use the window the server reports. | \n|  | [ -n \"$CTX\" ] && export CLAUDE_CODE_MAX_CONTEXT_TOKENS=\"$CTX\" | \n|  | # Local decode is slow enough that default timeouts abort mid-response. | \n|  | export API_TIMEOUT_MS=\"${API_TIMEOUT_MS:-3000000}\" | \n|  | export CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1 | \n|  | note \"model ${MODEL}${CTX:+ (context ${CTX})}\" | \n|  | # Auto mode calls a classifier model before each non-read-only tool use. Against | \n|  | # a slow local model that call reliably exceeds its own wall-clock timeout -- | \n|  | # which API_TIMEOUT_MS does NOT govern -- and every Bash call then fails with | \n|  | # \"temporarily unavailable (timed out)\". Read-only tools bypass the classifier, | \n|  | # which is why reads keep working while commands do not. So default to manual | \n|  | # approval; pass --permission-mode explicitly, or set MTPLX_PERMISSION_MODE, to | \n|  | # override. | \n|  | mode_given=no | \n|  | for arg in \"$@\"; do | \n|  | case \"$arg\" in | \n|  | --permission-mode\\|--permission-mode=*) mode_given=yes; break ;; | \n|  | esac | \n|  | done | \n|  | if [ \"$mode_given\" = no ]; then | \n|  | set -- --permission-mode \"${MTPLX_PERMISSION_MODE:-manual}\" \"$@\" | \n|  | fi | \n|  | # Measured: when git status changes between runs (the normal case), moving the | \n|  | # per-machine sections out of the system prompt keeps more of the cached prefix | \n|  | # valid and roughly halves prefill -- ~44s -> ~24s over 4 runs each. It does not | \n|  | # restore the full reuse you get from a byte-identical prompt (~0.4s), but it is | \n|  | # a consistent win. Set MTPLX_EXCLUDE_DYNAMIC=0 to turn it off. | \n|  | excl_given=no | \n|  | for arg in \"$@\"; do | \n|  | case \"$arg\" in | \n|  | --exclude-dynamic-system-prompt-sections) excl_given=yes; break ;; | \n|  | esac | \n|  | done | \n|  | if [ \"$excl_given\" = no ] && [ \"${MTPLX_EXCLUDE_DYNAMIC:-1}\" = 1 ]; then | \n|  | set -- --exclude-dynamic-system-prompt-sections \"$@\" | \n|  | fi | \n|  | exec claude \"$@\" |", "url": "https://wpnews.pro/news/claude-mtplx-wrapper-to-streamline-running-claude-code-against-mtplx", "canonical_source": "https://gist.github.com/tommythorn/129f5c93db491d5f4bff6d6202f429d4", "published_at": "2026-09-03 07:03:21+00:00", "updated_at": "2026-09-25 18:31:31.081126+00:00", "lang": "en", "topics": ["ai-tools", "developer-tools", "large-language-models", "ai-infrastructure"], "entities": ["Claude Code", "MTPLX", "Anthropic"], "also_reported_by": [], "alternates": {"html": "https://wpnews.pro/news/claude-mtplx-wrapper-to-streamline-running-claude-code-against-mtplx", "markdown": "https://wpnews.pro/news/claude-mtplx-wrapper-to-streamline-running-claude-code-against-mtplx.md", "text": "https://wpnews.pro/news/claude-mtplx-wrapper-to-streamline-running-claude-code-against-mtplx.txt", "jsonld": "https://wpnews.pro/news/claude-mtplx-wrapper-to-streamline-running-claude-code-against-mtplx.jsonld"}}