Keeps the prompt cache warm while background tasks run

Claude Code mods, published as a plugin marketplace.
/plugin marketplace add RichardJSun/claude-code-mods
/plugin install cache-keepalive@richardjsun-mods
When Claude ends a turn while background tasks are still running, the prompt cache can expire before the task's notification arrives. The next request then re-writes the whole context into the cache at 2× the input price.
This mod pings the cache with a one-line fork of the conversation every 50 minutes while the session waits. A cache read costs 0.025–0.1× input, so a ping costs a few percent of a re-write. Pings stop when:
/keepalive pings now and reports the hit rate. The status line shows armed, warm or stopped.
The mod assumes the 1h cache TTL. It cannot read the TTL, so it infers the 5m TTL from overage, which switches a session to it. It does not arm while the 5-hour or 7-day rate-limit window is at 100% or more, and it stops if one crosses 100% during the wait. Mods cannot see model-scoped limits such as Fable's weekly one, so the mod ignores them. When such a limit runs out, the first ping misses and the mod stops. Rate-limit readings arrive with this session's own responses, so the mod may not see overage that other sessions cause during a wait. When a 5m TTL slips past the check, the first ping misses, pays for a full re-cache itself and saves nothing, and the mod stops.
The compactOnCap setting (off by default) compacts the conversation when the ping budget runs out, while the cache is still warm. It compacts only when the compaction's cost plus re-caching the summary is less than re-caching the whole context. The mod records the cost and summary size of each compaction it runs and uses them for the next decision. Compaction loses context detail, which is why it is off by default.
A background notification that arrives during compaction waits until compaction finishes, then starts its turn as usual.
hooks/register.ts 157 lines1import type { Hook, Register } from 'claude-code'
2
3type Engine = Parameters<Hook<'turn.start'>>[0]
4
5// Assumes the 1h cache TTL. Overage switches to the 5m TTL, so the mod stays off while any
6// rate-limit window is at 100%. A 5m TTL it fails to infer makes the first ping miss and stop the pings.
7const PING_EVERY_MS = 50 * 60_000
8const WRITE_MULT = 2
9const OUTPUT_MULT = 5
10
11// Compaction cost per context token and summary size over context, used until a compaction is measured
12const COMPACT_COST_GUESS = 0.6
13const SUMMARY_RATIO_GUESS = 0.15
14
15let timer: { cancel(): void } | undefined
16let generation = 0
17let compactOnCap = false
18let compacting = false
19let spent = 0
20let budget = 0
21let context = 0
22
23// Cache read price over base input price
24function readMult(model: string) {
25 const id = model.toLowerCase()
26 return /fable|mythos/.test(id) ? 0.025 : /opus[- ]?5[-. ]5/.test(id) ? 0.05 : 0.1
27}
28
29// Bumping the generation also stops a ping already in flight from rearming
30function stop() {
31 generation++
32 timer?.cancel()
33 timer = undefined
34}
35
36// Mods see only the account-wide windows, not model-scoped limits such as Fable's weekly one
37function inOverage(limits: { percentUsed: number }[]) {
38 return limits.some(l => l.percentUsed >= 100)
39}
40
41// Input-token equivalents a call cost, reads priced for the session's model
42function cost(u: { input_tokens: number, output_tokens: number, cache_read_input_tokens: number, cache_creation_input_tokens: number }, r: number) {
43 return r * u.cache_read_input_tokens + u.input_tokens
44 + WRITE_MULT * u.cache_creation_input_tokens + OUTPUT_MULT * u.output_tokens
45}
46
47async function ping($: Engine) {
48 const r = await $.model.fork({ prompt: 'Keep-alive ping. Reply with only: ok' })
49 if (!('usage' in r)) return { ok: false, line: `no ping: ${r.reason}` }
50
51 const u = r.usage
52 context = u.cache_read_input_tokens + u.cache_creation_input_tokens + u.input_tokens
53 const hit = u.cache_read_input_tokens / Math.max(context, 1)
54 spent += cost(u, readMult(await $.session.model()))
55
56 return {
57 ok: hit > 0.9,
58 line: `cache read ${u.cache_read_input_tokens}/${context} (${Math.round(hit * 100)}%), spent ${Math.round(spent)} of ${Math.round(budget)} token-equivalents`,
59 }
60}
61
62// Compacting pays when its own cost plus re-caching the summary is below re-caching the whole context
63async function compactPays($: Engine) {
64 const perToken = (await $.store.get('compactCost') as number | undefined) ?? COMPACT_COST_GUESS
65 const ratio = (await $.store.get('summaryRatio') as number | undefined) ?? SUMMARY_RATIO_GUESS
66 return perToken + WRITE_MULT * ratio < WRITE_MULT
67}
68
69async function compact($: Engine) {
70 compacting = true
71 try {
72 const r = await $.session.compact({ instructions: 'Also keep the background tasks\' IDs, output paths, and what to do when they finish.' })
73 if (r.skip !== undefined) return `compaction skipped: ${r.skip}`
74
75 const before = r.tokensBefore ?? context
76 if (r.usage && before) {
77 await $.store.set('compactCost', cost(r.usage, readMult(await $.session.model())) / before)
78 if (r.tokensAfter) await $.store.set('summaryRatio', r.tokensAfter / before)
79 }
80 const read = r.usage ? `${r.usage.cache_read_input_tokens} read from cache, ${r.usage.output_tokens} output` : 'no usage reported'
81 return `compacted ${before} → ${r.tokensAfter ?? '?'} tokens (${read})`
82 } catch (err) {
83 return `compaction refused: ${err instanceof Error ? err.message : String(err)}`
84 } finally {
85 compacting = false
86 }
87}
88
89function arm($: Engine) {
90 stop()
91 const armedAt = generation
92 timer = $.clock.after(PING_EVERY_MS, async () => {
93 timer = undefined
94 const { ok, line } = await ping($)
95 if (generation !== armedAt) return
96 $.ui.log(`keepalive: ${line}`, { to: 'debug' })
97 if (!ok) return $.ui.status('keepalive: missed, stopped')
98 if (spent < budget) {
99 $.ui.status('keepalive: warm')
100 return arm($)
101 }
102
103 if (!compactOnCap || !(await compactPays($))) return $.ui.status('keepalive: budget spent, stopped')
104 $.ui.status('keepalive: compacting')
105 const result = await compact($)
106 $.ui.log(`keepalive: ${result}`, { to: 'debug' })
107 $.ui.status(`keepalive: ${result.split(' (')[0]}`)
108 })
109}
110
111export const register: Register = (on, options) => {
112 compactOnCap = options.compactOnCap === true
113
114 on('session.start', async ($, e, next) => {
115 await $.command.register({ name: 'keepalive', description: 'Ping the prompt cache now and report the hit rate' })
116 return next(e)
117 })
118
119 on('turn.start', async ($, e, next) => {
120 stop()
121 if (!compacting) $.ui.status(undefined)
122 return next(e)
123 })
124
125 // Arm only when background work will wake the session later
126 on('classic.Stop', async ($, e, next) => {
127 if (e.background_tasks?.length && !compacting) {
128 const { context: ctx, rateLimits } = await $.session.usage()
129 if (inOverage(rateLimits)) {
130 $.ui.status('keepalive: off in overage')
131 return next(e)
132 }
133 spent = 0
134 budget = WRITE_MULT * (ctx.tokens ?? 0)
135 $.ui.status(`keepalive: armed, ${e.background_tasks.length} task(s)`)
136 arm($)
137 }
138 return next(e)
139 })
140
141 // Other sessions can push the account into overage during the wait
142 on('session.measure', async ($, e, next) => {
143 if (timer && e.changed.includes('rateLimits') && inOverage(e.rateLimits)) {
144 stop()
145 $.ui.status('keepalive: overage, stopped')
146 }
147 return next(e)
148 })
149
150 on('command.run', { command: 'keepalive' }, async $ => {
151 const { context: ctx } = await $.session.usage()
152 if (!budget) budget = WRITE_MULT * (ctx.tokens ?? 0)
153 const { ok, line } = await ping($)
154 return { text: `${ok ? 'HIT' : 'MISS'}: ${line}` }
155 })
156}
157