| | #!/usr/bin/env bash |
| | # claude-mtplx — run the Claude Code CLI against a locally served MTPLX model. |
| | # |
| | # Starts the MTPLX server if it isn't already listening, then execs claude |
| | # with the Anthropic-compatible endpoint pointed at it. Any arguments are |
| | # passed straight through to claude, e.g.: |
| | # |
| | # claude-mtplx # interactive session |
| | # claude-mtplx -p "explain this" # one-shot |
| | # claude-mtplx --mtplx-stop # stop the server and exit |
| | # | | | # Note: --profile and the KV-quant setting only apply when this script STARTS | | | # the server. If one is already listening, it is reused exactly as it was | | | # launched; the script warns when its profile differs from the requested one. | | | # | | | # Environment overrides: | | | # MTPLX_HOST, MTPLX_PORT where the server listens (127.0.0.1:8000) | | | # MTPLX_PROFILE runtime profile (turbo) | | | # MTPLX_KV_QUANT paged KV quantization: off|q8|q4 (q8) |
| | # MTPLX_SSD_CACHE SSD session cache: off\|on\|write-only (off) |
| | # MTPLX_PERMISSION_MODE claude --permission-mode value (manual) |
| | # MTPLX_EXCLUDE_DYNAMIC 1 to keep the prompt prefix stable, 0 to disable (1) | | | # MTPLX_API_KEY auth token, if the server requires one | | | # MTPLX_READY_TIMEOUT seconds to wait for model load (300) | | | set -euo pipefail |
| | HOST="${MTPLX_HOST:-127.0.0.1}" |
| | PORT="${MTPLX_PORT:-8000}" |
| | PROFILE="${MTPLX_PROFILE:-turbo}" |
| | KV_QUANT="${MTPLX_KV_QUANT:-q8}" # off\|q8\|q4 - q8 roughly halves KV bytes/token |
| | # Explicit: MTPLX's app setting is "target-default", which resolves to ON, so | | | # simply omitting the flag does NOT disable the SSD cache. Measured 0 cached | | | # tokens on every request against a long-lived server; its only plausible value | | | # is across server restarts, which was not tested. |
| | SSD_CACHE="${MTPLX_SSD_CACHE:-off}" # off\|on\|write-only |
| | READY_TIMEOUT="${MTPLX_READY_TIMEOUT:-300}" |
| | BASE="http://${HOST}:${PORT}" |
| | LOG_DIR="${HOME}/.mtplx/logs" |
| | LOG="${LOG_DIR}/claude-mtplx.log" |
| | die() { printf 'claude-mtplx: %s\n' "$*" >&2; exit 1; } |
| | note() { printf 'claude-mtplx: %s\n' "$*" >&2; } |
| | health() { curl -sf -m 3 "${BASE}/health" 2>/dev/null; } |
| | command -v mtplx >/dev/null 2>&1 || die "mtplx not found on PATH" | | | command -v claude >/dev/null 2>&1 || die "claude not found on PATH" | | | command -v jq >/dev/null 2>&1 || die "jq not found on PATH" |
| | if [ "${1:-}" = "--mtplx-stop" ]; then |
| | mtplx stop --port "$PORT" |
| | exit $? | | | fi |
| | # --- start the server if it isn't already answering ------------------------- |
| | if ! health >/dev/null; then |
| | note "starting MTPLX on ${BASE} (first model load takes ~1 min)" |
| | mkdir -p "$LOG_DIR" | | | nohup mtplx quickstart \ |
| | --profile "$PROFILE" \ |
| | --host "$HOST" \ |
| | --port "$PORT" \ |
| | --paged-kv-quantization "$KV_QUANT" \ |
| | --ssd-session-cache "$SSD_CACHE" \ |
| | --no-stats-footer >>"$LOG" 2>&1 & |
| | server_pid=$! | | | waited=0 |
| | until health >/dev/null; do |
| | if ! kill -0 "$server_pid" 2>/dev/null; then |
| | note "server exited during startup; last lines of ${LOG}:" |
| | tail -20 "$LOG" >&2 |
| | exit 1 | | | fi |
| | if [ "$waited" -ge "$READY_TIMEOUT" ]; then |
| | die "server not ready after ${READY_TIMEOUT}s (see ${LOG})" |
| | fi | | | sleep 2 | | | waited=$((waited + 2)) | | | done | | | note "server ready after ${waited}s (log: ${LOG})" | | | fi |
| | # --- read what the server is actually serving ------------------------------- |
| | info="$(health)" \|\| die "health check failed at ${BASE}" |
| | MODEL="$(printf '%s' "$info" \| jq -r '.model // empty')" |
| | [ -n "$MODEL" ] \|\| die "could not determine served model id from ${BASE}/health" |
| | CTX="$(printf '%s' "$info" \| jq -r '.context_window // empty')" |
| | # The profile and KV-quant flags above only shape a server this script launches. | | | # If one was already up (started by hand, or by MTPLX.app), it keeps whatever | | | # settings it was launched with. | | | RUNNING_PROFILE="$(printf '%s' "$info" | jq -r '.profile.name // empty')" |
| | if [ -n "$RUNNING_PROFILE" ] && [ "$RUNNING_PROFILE" != "$PROFILE" ]; then |
| | note "reusing a running server on profile '${RUNNING_PROFILE}' (wanted '${PROFILE}')" |
| | note "restart to apply: claude-mtplx --mtplx-stop && claude-mtplx" |
| | fi | | | # --- point Claude Code at it ------------------------------------------------ | | | # ANTHROPIC_API_KEY must be absent: if both it and AUTH_TOKEN are set, the key | | | # wins and Claude Code tries to authenticate against api.anthropic.com. | | | unset ANTHROPIC_API_KEY | | | export ANTHROPIC_BASE_URL="$BASE" | | | export ANTHROPIC_AUTH_TOKEN="${MTPLX_API_KEY:-mtplx-local}" | | | # All four aliases matter: Claude Code routes background work (conversation | | | # titles, subagents) to the Haiku/Sonnet aliases, which would otherwise | | | # resolve to real Anthropic model ids this server does not serve. | | | export ANTHROPIC_MODEL="$MODEL" | | | export ANTHROPIC_DEFAULT_OPUS_MODEL="$MODEL" | | | export ANTHROPIC_DEFAULT_SONNET_MODEL="$MODEL" | | | export ANTHROPIC_DEFAULT_HAIKU_MODEL="$MODEL" | | | export CLAUDE_CODE_SUBAGENT_MODEL="$MODEL" | | | # The model id is not in Claude Code's catalog, so it would otherwise assume a | | | # 200k window for auto-compact. Use the window the server reports. | | | [ -n "$CTX" ] && export CLAUDE_CODE_MAX_CONTEXT_TOKENS="$CTX" | | | # Local decode is slow enough that default timeouts abort mid-response. | | | export API_TIMEOUT_MS="${API_TIMEOUT_MS:-3000000}" | | | export CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1 | | | note "model ${MODEL}${CTX:+ (context ${CTX})}" | | | # Auto mode calls a classifier model before each non-read-only tool use. Against | | | # a slow local model that call reliably exceeds its own wall-clock timeout -- | | | # which API_TIMEOUT_MS does NOT govern -- and every Bash call then fails with | | | # "temporarily unavailable (timed out)". Read-only tools bypass the classifier, | | | # which is why reads keep working while commands do not. So default to manual | | | # approval; pass --permission-mode explicitly, or set MTPLX_PERMISSION_MODE, to | | | # override. | | | mode_given=no | | | for arg in "$@"; do | | | case "$arg" in | | | --permission-mode|--permission-mode=*) mode_given=yes; break ;; | | | esac | | | done |
| | if [ "$mode_given" = no ]; then |
| | set -- --permission-mode "${MTPLX_PERMISSION_MODE:-manual}" "$@" |
| | fi | | | # Measured: when git status changes between runs (the normal case), moving the | | | # per-machine sections out of the system prompt keeps more of the cached prefix | | | # valid and roughly halves prefill -- ~44s -> ~24s over 4 runs each. It does not | | | # restore the full reuse you get from a byte-identical prompt (~0.4s), but it is | | | # a consistent win. Set MTPLX_EXCLUDE_DYNAMIC=0 to turn it off. | | | excl_given=no | | | for arg in "$@"; do | | | case "$arg" in | | | --exclude-dynamic-system-prompt-sections) excl_given=yes; break ;; | | | esac | | | done |
| | if [ "$excl_given" = no ] && [ "${MTPLX_EXCLUDE_DYNAMIC:-1}" = 1 ]; then |
| | set -- --exclude-dynamic-system-prompt-sections "$@" |
| | fi | | | exec claude "$@" |