cd /news/ai-agents/cache-gap-check-py · home › topics › ai-agents › article
[ARTICLE · art-148897] src=gist.github.com ↗ pub= topic=ai-agents verified=true sentiment=· neutral

cache_gap_check.py

A developer published cache_gap_check.py, a read-only Python 3 stdlib script that scans Claude Code transcripts (~/.claude/projects/**/*.jsonl) for subagent sidechain requests, measures idle gaps between each subagent's consecutive requests, and estimates prompt-cache cost under the current 5-minute TTL versus a hypothetical 1-hour TTL in multiples of base input price. The tool flags gaps between 5 and 60 minutes as the window where the longer TTL could help and reports a verdict when estimated savings exceed 5%. It makes no network calls and never writes files.

by read7 min views1 publishedOct 10, 2026

| | #!/usr/bin/env python3 | | | """cache_gap_check.py - READ-ONLY. Is raising subagentPromptCacheTtl 5m -> 1h likely worth it? | | | Scans Claude Code transcripts (~/.claude/projects/**/*.jsonl) for subagent (sidechain) | | | requests, measures idle gaps between each subagent's consecutive requests, and estimates | | | cache cost under the current 5m TTL vs a hypothetical 1h TTL, in multiples of base input | | | price. Python 3 stdlib only. Never writes files and makes no network calls. | | | """ | | | import argparse, json, os, sys | | | from collections import defaultdict | | | from datetime import datetime, timedelta, timezone | | | W5, W1H = 1.25, 2.0 # cache write multipliers (Anthropic prompt-caching docs) | | | GAP_LO, GAP_HI = 300.0, 3600.0 # 5 min <= gap < 60 min is the window where 1h can help |

|  | VERDICT_PCT = 5.0 # >5% estimated saving => "likely worth it"; within +-5% => marginal | 
|  | def parse_ts(s): | 

| | try: | | | return datetime.fromisoformat(str(s).replace("Z", "+00:00")) | | | except Exception: | | | return None |

|  | def find_files(path): | 
|  | if os.path.isfile(path): | 
|  | return [path] | 
|  | out = [] | 
|  | for root, _, files in os.walk(path): | 
|  | out += [os.path.join(root, f) for f in files if f.endswith(".jsonl")] | 
|  | return sorted(out) | 
|  | def load_requests(files, since): | 

| | """Return {agent_key: [request dicts]} for sidechain/subagent assistant messages with usage.""" | | | agents = defaultdict(dict) # key -> {request_id: req} (dedupes streamed duplicate lines) | | | for f in files: | | | in_sub_dir = "subagents" in f.replace("\", "/").split("/") | | | try: | | | fh = open(f, "r", encoding="utf-8", errors="replace") | | | except OSError: | | | continue | | | with fh: | | | for n, line in enumerate(fh): | | | try: | | | o = json.loads(line) | | | except Exception: | | | continue | | | if not isinstance(o, dict): | | | continue |

|  | msg = o.get("message") | 
|  | if not isinstance(msg, dict) or not isinstance(msg.get("usage"), dict): | 

| | continue | | | if not (o.get("isSidechain") is True or o.get("agentId") or in_sub_dir): | | | continue # main-thread request: not a subagent | | | ts = parse_ts(o.get("timestamp")) | | | if ts is None: | | | continue | | | if ts.tzinfo is None: | | | ts = ts.replace(tzinfo=timezone.utc) | | | if ts < since: | | | continue |

|  | u = msg["usage"] | 
|  | cc = u.get("cache_creation") if isinstance(u.get("cache_creation"), dict) else {} | 
|  | create = int(u.get("cache_creation_input_tokens") or 0) | 
|  | w1h = int(cc.get("ephemeral_1h_input_tokens") or 0) | 
|  | w5 = int(cc.get("ephemeral_5m_input_tokens") or 0) | 

| | if w1h + w5 < create: # no/partial breakdown: treat remainder as 5m |

|  | w5 = create - w1h | 
|  | req = dict(ts=ts, create=create, read=int(u.get("cache_read_input_tokens") or 0), | 
|  | w5=w5, w1h=w1h) | 
|  | key = o.get("agentId") or f | 
|  | rid = msg.get("id") or o.get("requestId") or o.get("uuid") or "%s:%d" % (f, n) | 
|  | prev = agents[key].get(rid) | 
|  | if prev is None or ts < prev["ts"]: | 

| | keep_ts = ts | | | else: | | | keep_ts = prev["ts"] | | | req["ts"] = keep_ts # earliest line of a streamed message; usage from last line |

|  | agents[key][rid] = req | 
|  | return {k: sorted(v.values(), key=lambda r: r["ts"]) for k, v in agents.items()} | 
|  | def analyze(agents, read_mult): | 
|  | s = dict(agents=len(agents), requests=0, with_prev=0, lt5m=0, win=0, ge60m=0, | 
|  | win_rewritten=0, already_1h_tokens=0, cost_5m=0.0, cost_1h=0.0) | 
|  | for reqs in agents.values(): | 

| | prev = None | | | for r in reqs: |

|  | total = r["create"] + r["read"] # cached-prefix tokens this request used | 
|  | s["requests"] += 1 | 
|  | s["already_1h_tokens"] += r["w1h"] | 
|  | base = W5 * r["w5"] + W1H * r["w1h"] + read_mult * r["read"] | 
|  | hyp_read = r["read"] | 

| | gap = None | | | if prev is not None: |

|  | gap = (r["ts"] - prev["ts"]).total_seconds() | 
|  | s["with_prev"] += 1 | 

| | if gap < GAP_LO: | | | s["lt5m"] += 1 | | | elif gap < GAP_HI: |

|  | s["win"] += 1 | 
|  | s["win_rewritten"] += r["create"] | 

| | # with 1h, the previous request's cached prefix would still be alive | | | hyp_read = max(r["read"], min(prev["create"] + prev["read"], total)) | | | else: |

|  | s["ge60m"] += 1 | 
|  | hyp_write = total - hyp_read | 

| | hyp = W1H * hyp_write + read_mult * hyp_read |

|  | s["cost_5m"] += base | 
|  | s["cost_1h"] += hyp | 

| | prev = r | | | return s |

|  | def verdict(s): | 
|  | if s["requests"] == 0: | 

| | return "no-data", "No subagent requests found in range - nothing to conclude." | | | if s["cost_5m"] <= 0: | | | return "no-cache", "Subagents show no cache activity - TTL change would not matter." |

|  | d = (s["cost_1h"] - s["cost_5m"]) / s["cost_5m"] * 100.0 | 
|  | if d < -VERDICT_PCT: | 

| | return "worth-it", "LIKELY WORTH IT: est. %.1f%% lower cache cost with 1h." % -d | | | if d > VERDICT_PCT: | | | return "not-worth-it", "NOT WORTH IT: est. %.1f%% higher cache cost with 1h." % d |

|  | return "marginal", "MARGINAL (%+.1f%%): within +-%d%%, leave at 5m unless latency matters." % (d, VERDICT_PCT) | 
|  | def main(): | 
|  | ap = argparse.ArgumentParser(description="Estimate whether subagentPromptCacheTtl 5m->1h is worth it (read-only).") | 
|  | ap.add_argument("--days", type=float, default=14, help="look-back window in days (default 14)") | 
|  | ap.add_argument("--path", default=os.path.expanduser("~/.claude/projects"), | 
|  | help="transcript dir or file (default ~/.claude/projects)") | 
|  | ap.add_argument("--json", action="store_true", help="machine-readable output") | 
|  | ap.add_argument("--read-mult", type=float, default=0.1, | 
|  | help="cache-read multiplier (default 0.1; docs list 0.05/0.025 for some newer models)") | 
|  | a = ap.parse_args() | 
|  | since = datetime.now(timezone.utc) - timedelta(days=a.days) | 
|  | files = find_files(a.path) | 
|  | agents = load_requests(files, since) | 
|  | s = analyze(agents, a.read_mult) | 
|  | code, text = verdict(s) | 
|  | req = s["requests"] | 
|  | pct = lambda n: (100.0 * n / req) if req else 0.0 | 
|  | delta = ((s["cost_1h"] - s["cost_5m"]) / s["cost_5m"] * 100.0) if s["cost_5m"] else 0.0 | 

| | if a.json: |

|  | print(json.dumps(dict(path=a.path, days=a.days, files_scanned=len(files), subagents=s["agents"], | 
|  | requests=req, requests_with_prev=s["with_prev"], gap_lt_5m=s["lt5m"], | 
|  | gap_5m_to_60m=s["win"], gap_5m_to_60m_pct=round(pct(s["win"]), 1), gap_ge_60m=s["ge60m"], | 

| | tokens_rewritten_after_5m_60m_gaps=s["win_rewritten"], | | | tokens_already_written_as_1h=s["already_1h_tokens"], |

|  | cost_now_5m_base_units=round(s["cost_5m"], 1), cost_1h_base_units=round(s["cost_1h"], 1), | 
|  | est_change_pct=round(delta, 1), verdict=code, verdict_text=text, | 
|  | multipliers=dict(write_5m=W5, write_1h=W1H, read=a.read_mult)), indent=2)) | 

| | return 0 |

|  | print("Subagent cache-gap check \| last %g d \| %s" % (a.days, a.path)) | 
|  | print("Scanned %d files -> %d subagents, %d subagent requests" % (len(files), s["agents"], req)) | 

| | if req: |

|  | print("Gap 5m-60m (1h could help): %d (%.1f%% of requests) \| <5m: %d \| >=60m: %d \| first-in-agent: %d" | 
|  | % (s["win"], pct(s["win"]), s["lt5m"], s["ge60m"], req - s["with_prev"])) | 
|  | print("Cache tokens re-written after those gaps: %s" % format(s["win_rewritten"], ",")) | 
|  | print("Est. cost now (5m) : %s base-input-token units" % format(round(s["cost_5m"]), ",")) | 
|  | print("Est. cost with 1h : %s base-input-token units (%+.1f%%)" % (format(round(s["cost_1h"]), ","), delta)) | 
|  | if s["already_1h_tokens"]: | 
|  | print("Note: %s written tokens were already 1h in the data (priced 2x in 'now')." % format(s["already_1h_tokens"], ",")) | 
|  | print("Verdict: " + text) | 
|  | print("Rule: 1h write=2x vs 5m=1.25x (+0.75x every write); each avoided expired re-write saves 1.25-%.2g=%.3gx." % (a.read_mult, W5 - a.read_mult)) | 

| | return 0 |

|  | if __name__ == "__main__": | 
|  | sys.exit(main()) |
── more in #ai-agents 4 stories · sorted by recency
── more on @claude code 3 stories trending now
sponsored brought to you by zahid.host 4,200+ EU-deployed projects
reading about agents? ship yours in a single git push.

Run your AI side-project on zahid.host

EU-based hosting, git-push deploys, automatic HTTPS, no cold starts. Free tier with a custom domain — perfect for shipping the agent you just read about.

$git push zahid main
→ Live at https://your-agent.zahid.host ✓
Get free account → Pricing
from €0/mo · no card required
LIVE [news/cache-gap-check-py] indexed:0 read:7min 2026-10-10 · —