# Measure how much local LLM speed your Mac loses to heat over 30 minutes. Declares its definitions for burst, retention and onset so your numbers are comparable to someone else's.

> Source: <https://gist.github.com/anupmaster/8f7d47667a76309971fb4fffd978487e>
> Published: 2026-09-15 21:15:46+00:00

|  | #!/usr/bin/env python3 | 
|  | """ | 
|  | sustained_bench.py - measure sustained local-inference throughput on Apple silicon. | 
|  |  | 
|  | Writes one JSON record per generation window to a JSONL file, so the decay curve | 
|  | survives intact instead of being collapsed into a summary. | 
|  |  | 
|  | Design commitments (these are the reason the output is comparable to someone else's): | 
|  | * A warmup generation runs first and is DISCARDED. It contains lazy Metal kernel | 
|  | compilation and model page-in, which otherwise manufactures a fake decay curve. | 
|  | * Windows are fixed at N TOKENS, never N seconds. Time-boxed windows silently | 
|  | change what throughput means once the machine starts throttling. | 
|  | * Four fixed prompts rotate. This holds work-per-window constant while preventing | 
|  | unbounded KV-cache growth, so you measure heat rather than cache pressure. | 
|  | * Environment is captured at run start: chip, macOS build, Low Power Mode, AC state. | 
|  | Ambient room temperature is the one thing it cannot capture. Write it down. | 
|  |  | 
|  | Do NOT run this under sudo. thermal_log.sh needs root; this does not, and running | 
|  | the measured workload as root changes scheduling behaviour. | 
|  |  | 
|  | Usage: | 
|  | ./sustained_bench.py --backend mlx --model mlx-community/Qwen2.5-7B-Instruct-4bit \\ | 
|  | --duration 1800 --out run-plugged-lidopen.jsonl --label "plugged/lid-open/hard" | 
|  |  | 
|  | ./sustained_bench.py --backend ollama --model qwen2.5:7b --duration 1800 \\ | 
|  | --out run-unplugged.jsonl --label "unplugged/lid-open/hard" | 
|  |  | 
|  | ./sustained_bench.py --backend mock --duration 1800 --out synthetic.jsonl | 
|  | """ | 
|  |  | 
|  | from __future__ import annotations | 
|  |  | 
|  | import argparse | 
|  | import json | 
|  | import math | 
|  | import platform | 
|  | import subprocess | 
|  | import sys | 
|  | import time | 
|  |  | 
|  | # Four fixed prompts. Rotated, not extended. Roughly matched in length so that | 
|  | # prefill cost per window is near-constant. | 
|  | PROMPTS = [ | 
|  | "Explain how a B-tree index speeds up a range query, step by step.", | 
|  | "Describe the trade-offs between mutexes and channels for shared state.", | 
|  | "Walk through how a TLS handshake establishes a shared secret.", | 
|  | "Explain why floating point addition is not associative, with an example.", | 
|  | ] | 
|  |  | 
|  |  | 
|  | def sh(cmd: list[str]) -> str: | 
|  | """Run a command, return stdout stripped, empty string on any failure.""" | 
|  | try: | 
|  | return subprocess.run( | 
|  | cmd, capture_output=True, text=True, timeout=10, check=False | 
|  | ).stdout.strip() | 
|  | except Exception: | 
|  | return "" | 
|  |  | 
|  |  | 
|  | def environment() -> dict: | 
|  | """Capture everything about the machine that changes the answer.""" | 
|  | env = { | 
|  | "platform": platform.platform(), | 
|  | "python": platform.python_version(), | 
|  | "chip": sh(["sysctl", "-n", "machdep.cpu.brand_string"]), | 
|  | "cores": sh(["sysctl", "-n", "hw.ncpu"]), | 
|  | "memory_bytes": sh(["sysctl", "-n", "hw.memsize"]), | 
|  | "os_product": sh(["sw_vers", "-productVersion"]), | 
|  | "os_build": sh(["sw_vers", "-buildVersion"]), | 
|  | } | 
|  | pmset = sh(["pmset", "-g"]) | 
|  | env["lowpowermode"] = "unknown" | 
|  | for line in pmset.splitlines(): | 
|  | if "lowpowermode" in line: | 
|  | env["lowpowermode"] = line.split()[-1] | 
|  | ps = sh(["pmset", "-g", "ps"]) | 
|  | if "AC Power" in ps: | 
|  | env["power_source"] = "ac" | 
|  | elif "Battery Power" in ps: | 
|  | env["power_source"] = "battery" | 
|  | else: | 
|  | env["power_source"] = "unknown" | 
|  | return env | 
|  |  | 
|  |  | 
|  | class MLXBackend: | 
|  | """mlx_lm. The usual choice on Apple silicon.""" | 
|  |  | 
|  | virtual = False | 
|  |  | 
|  | def __init__(self, model: str): | 
|  | from mlx_lm import load  # imported late so other backends need no MLX | 
|  |  | 
|  | self.model, self.tokenizer = load(model) | 
|  | self.name = f"mlx:{model}" | 
|  |  | 
|  | def generate(self, prompt: str, max_tokens: int) -> tuple[int, float]: | 
|  | from mlx_lm import generate | 
|  |  | 
|  | messages = [{"role": "user", "content": prompt}] | 
|  | text = self.tokenizer.apply_chat_template( | 
|  | messages, add_generation_prompt=True, tokenize=False | 
|  | ) | 
|  | t0 = time.monotonic() | 
|  | out = generate( | 
|  | self.model, self.tokenizer, prompt=text, | 
|  | max_tokens=max_tokens, verbose=False, | 
|  | ) | 
|  | elapsed = time.monotonic() - t0 | 
|  | return len(self.tokenizer.encode(out)), elapsed | 
|  |  | 
|  |  | 
|  | class OllamaBackend: | 
|  | """Ollama HTTP API. Wall-clock timed here, not trusted from eval_duration.""" | 
|  |  | 
|  | virtual = False | 
|  |  | 
|  | def __init__(self, model: str, host: str = "http://127.0.0.1:11434"): | 
|  | self.model, self.host = model, host | 
|  | self.name = f"ollama:{model}" | 
|  |  | 
|  | def generate(self, prompt: str, max_tokens: int) -> tuple[int, float]: | 
|  | import urllib.request | 
|  |  | 
|  | payload = json.dumps({ | 
|  | "model": self.model, | 
|  | "prompt": prompt, | 
|  | "stream": False, | 
|  | "options": {"num_predict": max_tokens}, | 
|  | }).encode() | 
|  | req = urllib.request.Request( | 
|  | f"{self.host}/api/generate", data=payload, | 
|  | headers={"Content-Type": "application/json"}, | 
|  | ) | 
|  | t0 = time.monotonic() | 
|  | with urllib.request.urlopen(req, timeout=600) as r: | 
|  | body = json.loads(r.read()) | 
|  | elapsed = time.monotonic() - t0 | 
|  | return int(body.get("eval_count", 0)), elapsed | 
|  |  | 
|  |  | 
|  | class MockBackend: | 
|  | """ | 
|  | Synthetic thermal curve. No model, no sleeping, virtual clock. | 
|  |  | 
|  | Exists so the harness and the analyzer can be validated against a decay whose | 
|  | true onset is known exactly. That is the only way to measure how far the | 
|  | detector lags the physics, which is the central claim of the article this | 
|  | harness accompanies. | 
|  | """ | 
|  |  | 
|  | virtual = True | 
|  |  | 
|  | def __init__(self, burst: float = 40.0, floor_frac: float = 0.60, | 
|  | onset_s: float = 90.0, tau_s: float = 120.0): | 
|  | self.burst, self.floor_frac = burst, floor_frac | 
|  | self.onset_s, self.tau_s = onset_s, tau_s | 
|  | self.t = 0.0 | 
|  | self.name = (f"mock:burst={burst}tok/s,onset={onset_s}s," | 
|  | f"tau={tau_s}s,floor={floor_frac}") | 
|  |  | 
|  | def rate_at(self, t: float) -> float: | 
|  | if t < self.onset_s: | 
|  | return self.burst | 
|  | decayed = self.floor_frac + (1 - self.floor_frac) * math.exp( | 
|  | -(t - self.onset_s) / self.tau_s | 
|  | ) | 
|  | return self.burst * decayed | 
|  |  | 
|  | def generate(self, prompt: str, max_tokens: int) -> tuple[int, float]: | 
|  | # Integrate the rate across the window rather than sampling it once, so the | 
|  | # window's reported rate is the true average over its span. | 
|  | # | 
|  | # Fixed step, not an adaptive one. An earlier version used | 
|  | # step = min(0.05, remaining / rate), which collapses the step size as | 
|  | # remaining approaches zero and then grinds on floating-point residue, | 
|  | # exiting only via the guard below. Bounded iteration counts only. | 
|  | STEP = 0.01 | 
|  | produced, dt, t = 0.0, 0.0, self.t | 
|  | while produced < max_tokens and dt < 3600: | 
|  | produced += self.rate_at(t) * STEP | 
|  | t += STEP | 
|  | dt += STEP | 
|  | self.t += dt | 
|  | return max_tokens, dt | 
|  |  | 
|  |  | 
|  | def build_backend(args) -> object: | 
|  | if args.backend == "mlx": | 
|  | return MLXBackend(args.model) | 
|  | if args.backend == "ollama": | 
|  | return OllamaBackend(args.model) | 
|  | return MockBackend( | 
|  | burst=args.mock_burst, floor_frac=args.mock_floor, | 
|  | onset_s=args.mock_onset, tau_s=args.mock_tau, | 
|  | ) | 
|  |  | 
|  |  | 
|  | def main() -> int: | 
|  | p = argparse.ArgumentParser(description=__doc__, | 
|  | formatter_class=argparse.RawDescriptionHelpFormatter) | 
|  | p.add_argument("--backend", choices=["mlx", "ollama", "mock"], default="mock") | 
|  | p.add_argument("--model", default="") | 
|  | p.add_argument("--duration", type=float, default=1800, | 
|  | help="run length in seconds, excluding warmup (default 1800)") | 
|  | p.add_argument("--tokens-per-window", type=int, default=256, | 
|  | help="FIXED tokens per window. Never time-box a window.") | 
|  | p.add_argument("--out", default="run.jsonl") | 
|  | p.add_argument("--label", default="", | 
|  | help='condition, e.g. "plugged/lid-open/hard-surface"') | 
|  | p.add_argument("--ambient-c", type=float, default=None, | 
|  | help="room temperature in C. The harness cannot measure this.") | 
|  | p.add_argument("--mock-burst", type=float, default=40.0) | 
|  | p.add_argument("--mock-floor", type=float, default=0.60) | 
|  | p.add_argument("--mock-onset", type=float, default=90.0) | 
|  | p.add_argument("--mock-tau", type=float, default=120.0) | 
|  | args = p.parse_args() | 
|  |  | 
|  | if args.backend in ("mlx", "ollama") and not args.model: | 
|  | p.error(f"--model is required for backend {args.backend}") | 
|  |  | 
|  | backend = build_backend(args) | 
|  |  | 
|  | header = { | 
|  | "record": "header", | 
|  | "schema": 1, | 
|  | "started_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), | 
|  | "backend": backend.name, | 
|  | "virtual_clock": backend.virtual, | 
|  | "tokens_per_window": args.tokens_per_window, | 
|  | "target_duration_s": args.duration, | 
|  | "label": args.label, | 
|  | "ambient_c": args.ambient_c, | 
|  | "prompts": PROMPTS, | 
|  | "env": environment(), | 
|  | } | 
|  |  | 
|  | with open(args.out, "w") as fh: | 
|  | fh.write(json.dumps(header) + "\n") | 
|  | fh.flush() | 
|  |  | 
|  | # Warmup. Discarded on purpose. Recorded only so the file proves it happened. | 
|  | w_tokens, w_elapsed = backend.generate(PROMPTS[0], args.tokens_per_window) | 
|  | fh.write(json.dumps({ | 
|  | "record": "warmup", "tokens": w_tokens, "wall_s": round(w_elapsed, 4), | 
|  | "tok_s": round(w_tokens / w_elapsed, 3) if w_elapsed else None, | 
|  | "note": "discarded from all statistics", | 
|  | }) + "\n") | 
|  | fh.flush() | 
|  | print(f"warmup discarded: {w_tokens} tok in {w_elapsed:.1f}s", file=sys.stderr) | 
|  |  | 
|  | clock = (lambda: backend.t) if backend.virtual else None | 
|  | t0 = time.monotonic() | 
|  | elapsed_total = 0.0 | 
|  | i = 0 | 
|  | if backend.virtual: | 
|  | backend.t = 0.0 | 
|  |  | 
|  | while elapsed_total < args.duration: | 
|  | prompt = PROMPTS[i % len(PROMPTS)] | 
|  | t_start = elapsed_total | 
|  | tokens, wall = backend.generate(prompt, args.tokens_per_window) | 
|  | elapsed_total = clock() if clock else time.monotonic() - t0 | 
|  | rec = { | 
|  | "record": "window", | 
|  | "i": i, | 
|  | "prompt_idx": i % len(PROMPTS), | 
|  | "t_start_s": round(t_start, 3), | 
|  | "t_end_s": round(elapsed_total, 3), | 
|  | "wall_s": round(wall, 4), | 
|  | "tokens": tokens, | 
|  | "tok_s": round(tokens / wall, 3) if wall else None, | 
|  | } | 
|  | fh.write(json.dumps(rec) + "\n") | 
|  | fh.flush() | 
|  | print(f"  window {i:>3}  t={t_start:7.1f}s  {rec['tok_s']:>7.2f} tok/s", | 
|  | file=sys.stderr) | 
|  | i += 1 | 
|  |  | 
|  | fh.write(json.dumps({ | 
|  | "record": "footer", "windows": i, | 
|  | "total_s": round(elapsed_total, 3), | 
|  | }) + "\n") | 
|  |  | 
|  | print(f"\n{i} windows over {elapsed_total:.0f}s written to {args.out}", | 
|  | file=sys.stderr) | 
|  | print(f"next: ./analyze.py {args.out}", file=sys.stderr) | 
|  | return 0 | 
|  |  | 
|  |  | 
|  | if __name__ == "__main__": | 
|  | sys.exit(main()) |
