# claude-mtplx wrapper to streamline running Claude Code against mtplx

> Source: <https://gist.github.com/tommythorn/129f5c93db491d5f4bff6d6202f429d4>
> Published: 2026-09-03 07:03:21+00:00

|  | #!/usr/bin/env bash | 
|  | # claude-mtplx — run the Claude Code CLI against a locally served MTPLX model. | 
|  | # | 
|  | # Starts the MTPLX server if it isn't already listening, then execs `claude` | 
|  | # with the Anthropic-compatible endpoint pointed at it. Any arguments are | 
|  | # passed straight through to claude, e.g.: | 
|  | # | 
|  | # claude-mtplx # interactive session | 
|  | # claude-mtplx -p "explain this" # one-shot | 
|  | # claude-mtplx --mtplx-stop # stop the server and exit | 
|  | # | 
|  | # Note: --profile and the KV-quant setting only apply when this script STARTS | 
|  | # the server. If one is already listening, it is reused exactly as it was | 
|  | # launched; the script warns when its profile differs from the requested one. | 
|  | # | 
|  | # Environment overrides: | 
|  | # MTPLX_HOST, MTPLX_PORT where the server listens (127.0.0.1:8000) | 
|  | # MTPLX_PROFILE runtime profile (turbo) | 
|  | # MTPLX_KV_QUANT paged KV quantization: off\|q8\|q4 (q8) | 
|  | # MTPLX_SSD_CACHE SSD session cache: off\|on\|write-only (off) | 
|  | # MTPLX_PERMISSION_MODE claude --permission-mode value (manual) | 
|  | # MTPLX_EXCLUDE_DYNAMIC 1 to keep the prompt prefix stable, 0 to disable (1) | 
|  | # MTPLX_API_KEY auth token, if the server requires one | 
|  | # MTPLX_READY_TIMEOUT seconds to wait for model load (300) | 
|  | set -euo pipefail | 
|  | HOST="${MTPLX_HOST:-127.0.0.1}" | 
|  | PORT="${MTPLX_PORT:-8000}" | 
|  | PROFILE="${MTPLX_PROFILE:-turbo}" | 
|  | KV_QUANT="${MTPLX_KV_QUANT:-q8}" # off\|q8\|q4 - q8 roughly halves KV bytes/token | 
|  | # Explicit: MTPLX's app setting is "target-default", which resolves to ON, so | 
|  | # simply omitting the flag does NOT disable the SSD cache. Measured 0 cached | 
|  | # tokens on every request against a long-lived server; its only plausible value | 
|  | # is across server restarts, which was not tested. | 
|  | SSD_CACHE="${MTPLX_SSD_CACHE:-off}" # off\|on\|write-only | 
|  | READY_TIMEOUT="${MTPLX_READY_TIMEOUT:-300}" | 
|  | BASE="http://${HOST}:${PORT}" | 
|  | LOG_DIR="${HOME}/.mtplx/logs" | 
|  | LOG="${LOG_DIR}/claude-mtplx.log" | 
|  | die() { printf 'claude-mtplx: %s\n' "$*" >&2; exit 1; } | 
|  | note() { printf 'claude-mtplx: %s\n' "$*" >&2; } | 
|  | health() { curl -sf -m 3 "${BASE}/health" 2>/dev/null; } | 
|  | command -v mtplx >/dev/null 2>&1 \|\| die "mtplx not found on PATH" | 
|  | command -v claude >/dev/null 2>&1 \|\| die "claude not found on PATH" | 
|  | command -v jq >/dev/null 2>&1 \|\| die "jq not found on PATH" | 
|  | if [ "${1:-}" = "--mtplx-stop" ]; then | 
|  | mtplx stop --port "$PORT" | 
|  | exit $? | 
|  | fi | 
|  | # --- start the server if it isn't already answering ------------------------- | 
|  | if ! health >/dev/null; then | 
|  | note "starting MTPLX on ${BASE} (first model load takes ~1 min)" | 
|  | mkdir -p "$LOG_DIR" | 
|  | nohup mtplx quickstart \ | 
|  | --profile "$PROFILE" \ | 
|  | --host "$HOST" \ | 
|  | --port "$PORT" \ | 
|  | --paged-kv-quantization "$KV_QUANT" \ | 
|  | --ssd-session-cache "$SSD_CACHE" \ | 
|  | --no-stats-footer >>"$LOG" 2>&1 & | 
|  | server_pid=$! | 
|  | waited=0 | 
|  | until health >/dev/null; do | 
|  | if ! kill -0 "$server_pid" 2>/dev/null; then | 
|  | note "server exited during startup; last lines of ${LOG}:" | 
|  | tail -20 "$LOG" >&2 | 
|  | exit 1 | 
|  | fi | 
|  | if [ "$waited" -ge "$READY_TIMEOUT" ]; then | 
|  | die "server not ready after ${READY_TIMEOUT}s (see ${LOG})" | 
|  | fi | 
|  | sleep 2 | 
|  | waited=$((waited + 2)) | 
|  | done | 
|  | note "server ready after ${waited}s (log: ${LOG})" | 
|  | fi | 
|  | # --- read what the server is actually serving ------------------------------- | 
|  | info="$(health)" \|\| die "health check failed at ${BASE}" | 
|  | MODEL="$(printf '%s' "$info" \| jq -r '.model // empty')" | 
|  | [ -n "$MODEL" ] \|\| die "could not determine served model id from ${BASE}/health" | 
|  | CTX="$(printf '%s' "$info" \| jq -r '.context_window // empty')" | 
|  | # The profile and KV-quant flags above only shape a server this script launches. | 
|  | # If one was already up (started by hand, or by MTPLX.app), it keeps whatever | 
|  | # settings it was launched with. | 
|  | RUNNING_PROFILE="$(printf '%s' "$info" \| jq -r '.profile.name // empty')" | 
|  | if [ -n "$RUNNING_PROFILE" ] && [ "$RUNNING_PROFILE" != "$PROFILE" ]; then | 
|  | note "reusing a running server on profile '${RUNNING_PROFILE}' (wanted '${PROFILE}')" | 
|  | note "restart to apply: claude-mtplx --mtplx-stop && claude-mtplx" | 
|  | fi | 
|  | # --- point Claude Code at it ------------------------------------------------ | 
|  | # ANTHROPIC_API_KEY must be absent: if both it and AUTH_TOKEN are set, the key | 
|  | # wins and Claude Code tries to authenticate against api.anthropic.com. | 
|  | unset ANTHROPIC_API_KEY | 
|  | export ANTHROPIC_BASE_URL="$BASE" | 
|  | export ANTHROPIC_AUTH_TOKEN="${MTPLX_API_KEY:-mtplx-local}" | 
|  | # All four aliases matter: Claude Code routes background work (conversation | 
|  | # titles, subagents) to the Haiku/Sonnet aliases, which would otherwise | 
|  | # resolve to real Anthropic model ids this server does not serve. | 
|  | export ANTHROPIC_MODEL="$MODEL" | 
|  | export ANTHROPIC_DEFAULT_OPUS_MODEL="$MODEL" | 
|  | export ANTHROPIC_DEFAULT_SONNET_MODEL="$MODEL" | 
|  | export ANTHROPIC_DEFAULT_HAIKU_MODEL="$MODEL" | 
|  | export CLAUDE_CODE_SUBAGENT_MODEL="$MODEL" | 
|  | # The model id is not in Claude Code's catalog, so it would otherwise assume a | 
|  | # 200k window for auto-compact. Use the window the server reports. | 
|  | [ -n "$CTX" ] && export CLAUDE_CODE_MAX_CONTEXT_TOKENS="$CTX" | 
|  | # Local decode is slow enough that default timeouts abort mid-response. | 
|  | export API_TIMEOUT_MS="${API_TIMEOUT_MS:-3000000}" | 
|  | export CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1 | 
|  | note "model ${MODEL}${CTX:+ (context ${CTX})}" | 
|  | # Auto mode calls a classifier model before each non-read-only tool use. Against | 
|  | # a slow local model that call reliably exceeds its own wall-clock timeout -- | 
|  | # which API_TIMEOUT_MS does NOT govern -- and every Bash call then fails with | 
|  | # "temporarily unavailable (timed out)". Read-only tools bypass the classifier, | 
|  | # which is why reads keep working while commands do not. So default to manual | 
|  | # approval; pass --permission-mode explicitly, or set MTPLX_PERMISSION_MODE, to | 
|  | # override. | 
|  | mode_given=no | 
|  | for arg in "$@"; do | 
|  | case "$arg" in | 
|  | --permission-mode\|--permission-mode=*) mode_given=yes; break ;; | 
|  | esac | 
|  | done | 
|  | if [ "$mode_given" = no ]; then | 
|  | set -- --permission-mode "${MTPLX_PERMISSION_MODE:-manual}" "$@" | 
|  | fi | 
|  | # Measured: when git status changes between runs (the normal case), moving the | 
|  | # per-machine sections out of the system prompt keeps more of the cached prefix | 
|  | # valid and roughly halves prefill -- ~44s -> ~24s over 4 runs each. It does not | 
|  | # restore the full reuse you get from a byte-identical prompt (~0.4s), but it is | 
|  | # a consistent win. Set MTPLX_EXCLUDE_DYNAMIC=0 to turn it off. | 
|  | excl_given=no | 
|  | for arg in "$@"; do | 
|  | case "$arg" in | 
|  | --exclude-dynamic-system-prompt-sections) excl_given=yes; break ;; | 
|  | esac | 
|  | done | 
|  | if [ "$excl_given" = no ] && [ "${MTPLX_EXCLUDE_DYNAMIC:-1}" = 1 ]; then | 
|  | set -- --exclude-dynamic-system-prompt-sections "$@" | 
|  | fi | 
|  | exec claude "$@" |
