{"slug": "cache-gap-check-py", "title": "cache_gap_check.py", "summary": "A developer published cache_gap_check.py, a read-only Python 3 stdlib script that scans Claude Code transcripts (~/.claude/projects/**/*.jsonl) for subagent sidechain requests, measures idle gaps between each subagent's consecutive requests, and estimates prompt-cache cost under the current 5-minute TTL versus a hypothetical 1-hour TTL in multiples of base input price. The tool flags gaps between 5 and 60 minutes as the window where the longer TTL could help and reports a verdict when estimated savings exceed 5%. It makes no network calls and never writes files.", "body_md": "|  | #!/usr/bin/env python3 | \n|  | \"\"\"cache_gap_check.py - READ-ONLY. Is raising subagentPromptCacheTtl 5m -> 1h likely worth it? | \n|  | Scans Claude Code transcripts (~/.claude/projects/**/*.jsonl) for subagent (sidechain) | \n|  | requests, measures idle gaps between each subagent's consecutive requests, and estimates | \n|  | cache cost under the current 5m TTL vs a hypothetical 1h TTL, in multiples of base input | \n|  | price. Python 3 stdlib only. Never writes files and makes no network calls. | \n|  | \"\"\" | \n|  | import argparse, json, os, sys | \n|  | from collections import defaultdict | \n|  | from datetime import datetime, timedelta, timezone | \n|  | W5, W1H = 1.25, 2.0 # cache write multipliers (Anthropic prompt-caching docs) | \n|  | GAP_LO, GAP_HI = 300.0, 3600.0 # 5 min <= gap < 60 min is the window where 1h can help | \n|  | VERDICT_PCT = 5.0 # >5% estimated saving => \"likely worth it\"; within +-5% => marginal | \n|  | def parse_ts(s): | \n|  | try: | \n|  | return datetime.fromisoformat(str(s).replace(\"Z\", \"+00:00\")) | \n|  | except Exception: | \n|  | return None | \n|  | def find_files(path): | \n|  | if os.path.isfile(path): | \n|  | return [path] | \n|  | out = [] | \n|  | for root, _, files in os.walk(path): | \n|  | out += [os.path.join(root, f) for f in files if f.endswith(\".jsonl\")] | \n|  | return sorted(out) | \n|  | def load_requests(files, since): | \n|  | \"\"\"Return {agent_key: [request dicts]} for sidechain/subagent assistant messages with usage.\"\"\" | \n|  | agents = defaultdict(dict) # key -> {request_id: req} (dedupes streamed duplicate lines) | \n|  | for f in files: | \n|  | in_sub_dir = \"subagents\" in f.replace(\"\\\\\", \"/\").split(\"/\") | \n|  | try: | \n|  | fh = open(f, \"r\", encoding=\"utf-8\", errors=\"replace\") | \n|  | except OSError: | \n|  | continue | \n|  | with fh: | \n|  | for n, line in enumerate(fh): | \n|  | try: | \n|  | o = json.loads(line) | \n|  | except Exception: | \n|  | continue | \n|  | if not isinstance(o, dict): | \n|  | continue | \n|  | msg = o.get(\"message\") | \n|  | if not isinstance(msg, dict) or not isinstance(msg.get(\"usage\"), dict): | \n|  | continue | \n|  | if not (o.get(\"isSidechain\") is True or o.get(\"agentId\") or in_sub_dir): | \n|  | continue # main-thread request: not a subagent | \n|  | ts = parse_ts(o.get(\"timestamp\")) | \n|  | if ts is None: | \n|  | continue | \n|  | if ts.tzinfo is None: | \n|  | ts = ts.replace(tzinfo=timezone.utc) | \n|  | if ts < since: | \n|  | continue | \n|  | u = msg[\"usage\"] | \n|  | cc = u.get(\"cache_creation\") if isinstance(u.get(\"cache_creation\"), dict) else {} | \n|  | create = int(u.get(\"cache_creation_input_tokens\") or 0) | \n|  | w1h = int(cc.get(\"ephemeral_1h_input_tokens\") or 0) | \n|  | w5 = int(cc.get(\"ephemeral_5m_input_tokens\") or 0) | \n|  | if w1h + w5 < create: # no/partial breakdown: treat remainder as 5m | \n|  | w5 = create - w1h | \n|  | req = dict(ts=ts, create=create, read=int(u.get(\"cache_read_input_tokens\") or 0), | \n|  | w5=w5, w1h=w1h) | \n|  | key = o.get(\"agentId\") or f | \n|  | rid = msg.get(\"id\") or o.get(\"requestId\") or o.get(\"uuid\") or \"%s:%d\" % (f, n) | \n|  | prev = agents[key].get(rid) | \n|  | if prev is None or ts < prev[\"ts\"]: | \n|  | keep_ts = ts | \n|  | else: | \n|  | keep_ts = prev[\"ts\"] | \n|  | req[\"ts\"] = keep_ts # earliest line of a streamed message; usage from last line | \n|  | agents[key][rid] = req | \n|  | return {k: sorted(v.values(), key=lambda r: r[\"ts\"]) for k, v in agents.items()} | \n|  | def analyze(agents, read_mult): | \n|  | s = dict(agents=len(agents), requests=0, with_prev=0, lt5m=0, win=0, ge60m=0, | \n|  | win_rewritten=0, already_1h_tokens=0, cost_5m=0.0, cost_1h=0.0) | \n|  | for reqs in agents.values(): | \n|  | prev = None | \n|  | for r in reqs: | \n|  | total = r[\"create\"] + r[\"read\"] # cached-prefix tokens this request used | \n|  | s[\"requests\"] += 1 | \n|  | s[\"already_1h_tokens\"] += r[\"w1h\"] | \n|  | base = W5 * r[\"w5\"] + W1H * r[\"w1h\"] + read_mult * r[\"read\"] | \n|  | hyp_read = r[\"read\"] | \n|  | gap = None | \n|  | if prev is not None: | \n|  | gap = (r[\"ts\"] - prev[\"ts\"]).total_seconds() | \n|  | s[\"with_prev\"] += 1 | \n|  | if gap < GAP_LO: | \n|  | s[\"lt5m\"] += 1 | \n|  | elif gap < GAP_HI: | \n|  | s[\"win\"] += 1 | \n|  | s[\"win_rewritten\"] += r[\"create\"] | \n|  | # with 1h, the previous request's cached prefix would still be alive | \n|  | hyp_read = max(r[\"read\"], min(prev[\"create\"] + prev[\"read\"], total)) | \n|  | else: | \n|  | s[\"ge60m\"] += 1 | \n|  | hyp_write = total - hyp_read | \n|  | hyp = W1H * hyp_write + read_mult * hyp_read | \n|  | s[\"cost_5m\"] += base | \n|  | s[\"cost_1h\"] += hyp | \n|  | prev = r | \n|  | return s | \n|  | def verdict(s): | \n|  | if s[\"requests\"] == 0: | \n|  | return \"no-data\", \"No subagent requests found in range - nothing to conclude.\" | \n|  | if s[\"cost_5m\"] <= 0: | \n|  | return \"no-cache\", \"Subagents show no cache activity - TTL change would not matter.\" | \n|  | d = (s[\"cost_1h\"] - s[\"cost_5m\"]) / s[\"cost_5m\"] * 100.0 | \n|  | if d < -VERDICT_PCT: | \n|  | return \"worth-it\", \"LIKELY WORTH IT: est. %.1f%% lower cache cost with 1h.\" % -d | \n|  | if d > VERDICT_PCT: | \n|  | return \"not-worth-it\", \"NOT WORTH IT: est. %.1f%% higher cache cost with 1h.\" % d | \n|  | return \"marginal\", \"MARGINAL (%+.1f%%): within +-%d%%, leave at 5m unless latency matters.\" % (d, VERDICT_PCT) | \n|  | def main(): | \n|  | ap = argparse.ArgumentParser(description=\"Estimate whether subagentPromptCacheTtl 5m->1h is worth it (read-only).\") | \n|  | ap.add_argument(\"--days\", type=float, default=14, help=\"look-back window in days (default 14)\") | \n|  | ap.add_argument(\"--path\", default=os.path.expanduser(\"~/.claude/projects\"), | \n|  | help=\"transcript dir or file (default ~/.claude/projects)\") | \n|  | ap.add_argument(\"--json\", action=\"store_true\", help=\"machine-readable output\") | \n|  | ap.add_argument(\"--read-mult\", type=float, default=0.1, | \n|  | help=\"cache-read multiplier (default 0.1; docs list 0.05/0.025 for some newer models)\") | \n|  | a = ap.parse_args() | \n|  | since = datetime.now(timezone.utc) - timedelta(days=a.days) | \n|  | files = find_files(a.path) | \n|  | agents = load_requests(files, since) | \n|  | s = analyze(agents, a.read_mult) | \n|  | code, text = verdict(s) | \n|  | req = s[\"requests\"] | \n|  | pct = lambda n: (100.0 * n / req) if req else 0.0 | \n|  | delta = ((s[\"cost_1h\"] - s[\"cost_5m\"]) / s[\"cost_5m\"] * 100.0) if s[\"cost_5m\"] else 0.0 | \n|  | if a.json: | \n|  | print(json.dumps(dict(path=a.path, days=a.days, files_scanned=len(files), subagents=s[\"agents\"], | \n|  | requests=req, requests_with_prev=s[\"with_prev\"], gap_lt_5m=s[\"lt5m\"], | \n|  | gap_5m_to_60m=s[\"win\"], gap_5m_to_60m_pct=round(pct(s[\"win\"]), 1), gap_ge_60m=s[\"ge60m\"], | \n|  | tokens_rewritten_after_5m_60m_gaps=s[\"win_rewritten\"], | \n|  | tokens_already_written_as_1h=s[\"already_1h_tokens\"], | \n|  | cost_now_5m_base_units=round(s[\"cost_5m\"], 1), cost_1h_base_units=round(s[\"cost_1h\"], 1), | \n|  | est_change_pct=round(delta, 1), verdict=code, verdict_text=text, | \n|  | multipliers=dict(write_5m=W5, write_1h=W1H, read=a.read_mult)), indent=2)) | \n|  | return 0 | \n|  | print(\"Subagent cache-gap check \\| last %g d \\| %s\" % (a.days, a.path)) | \n|  | print(\"Scanned %d files -> %d subagents, %d subagent requests\" % (len(files), s[\"agents\"], req)) | \n|  | if req: | \n|  | print(\"Gap 5m-60m (1h could help): %d (%.1f%% of requests) \\| <5m: %d \\| >=60m: %d \\| first-in-agent: %d\" | \n|  | % (s[\"win\"], pct(s[\"win\"]), s[\"lt5m\"], s[\"ge60m\"], req - s[\"with_prev\"])) | \n|  | print(\"Cache tokens re-written after those gaps: %s\" % format(s[\"win_rewritten\"], \",\")) | \n|  | print(\"Est. cost now (5m) : %s base-input-token units\" % format(round(s[\"cost_5m\"]), \",\")) | \n|  | print(\"Est. cost with 1h : %s base-input-token units (%+.1f%%)\" % (format(round(s[\"cost_1h\"]), \",\"), delta)) | \n|  | if s[\"already_1h_tokens\"]: | \n|  | print(\"Note: %s written tokens were already 1h in the data (priced 2x in 'now').\" % format(s[\"already_1h_tokens\"], \",\")) | \n|  | print(\"Verdict: \" + text) | \n|  | print(\"Rule: 1h write=2x vs 5m=1.25x (+0.75x every write); each avoided expired re-write saves 1.25-%.2g=%.3gx.\" % (a.read_mult, W5 - a.read_mult)) | \n|  | return 0 | \n|  | if __name__ == \"__main__\": | \n|  | sys.exit(main()) |", "url": "https://wpnews.pro/news/cache-gap-check-py", "canonical_source": "https://gist.github.com/DannyMac180/9b9196d2434709f3ac04bc394d02a562", "published_at": "2026-10-10 17:04:50+00:00", "updated_at": "2026-10-10 20:18:26.799030+00:00", "lang": "en", "topics": ["ai-agents", "ai-tools", "developer-tools", "mlops"], "entities": ["Claude Code", "Anthropic"], "also_reported_by": [], "alternates": {"html": "https://wpnews.pro/news/cache-gap-check-py", "markdown": "https://wpnews.pro/news/cache-gap-check-py.md", "text": "https://wpnews.pro/news/cache-gap-check-py.txt", "jsonld": "https://wpnews.pro/news/cache-gap-check-py.jsonld"}}