Keeps an idle session's prompt cache warm with a cache-served fork just before the TTL lapses, within limits you set.

A Claude Code mod (function-hook plugin) that keeps an idle session's prompt cache warm. Just before the cache TTL lapses (default: 55 minutes after the last request), it sends a $.model.fork ping. The ping is served from the cache and adds nothing to the conversation.
On Opus 5.5, one cache read costs about 1/35–1/40 as much as re-caching the same prefix, measured against subscription usage. Pinging through a workday therefore costs much less than one cold restart.
$.model.fork) always replays the main thread's last request, and subagent turns don't arm or reset the timer, so an idle subagent's cache is never pinged and just expires.ttlMinutes − leadMinutes (60 − 5).maxIdleHours (8), unless /keepwarm for|until set a window. The window restarts on any main-loop turn, including ones you didn't type: a background agent's result arriving, a /loop wakeup. That's intended, because those turns use the main cache too;minContextTokens (60k);maxLimitPercent (85);/keepwarm now hit lifts the bench.kind: verify).-p, SDK).~/.claude/cache-keeper/<session-id>.jsonl: token usage, plan-usage percentage before and after, and API-equivalent cost before and after./keepwarm [status | on | off | auto | now | for 3h | until 18:00]
claude plugin marketplace add chrisvaillancourt/claude-cache-keeper
claude plugin install cache-keeper@cache-keeper --scope user
Or from inside Claude Code: /plugin install cache-keeper --marketplace chrisvaillancourt/claude-cache-keeper.
To work on it locally, add your clone as the marketplace instead (claude plugin marketplace add <path to clone>). The plugin is read from that folder; after editing, run /reload-plugins.
claude plugin update cache-keeper@cache-keeper
Then restart Claude Code. If it says the plugin is already at the latest version but a newer release exists, refresh the marketplace first and run the update again:
claude plugin marketplace update cache-keeper
claude plugin update cache-keeper@cache-keeper
claude plugin validate .
claude plugin test .
Commit messages follow Conventional Commits (feat:, fix:, docs:, ...).
MIT. See LICENSE.
hooks/register.ts 317 lines1import { atom, read, update } from 'claude-code'
2import type { EngineInterface, ModelUsage, PluginOptions, Register, SessionUsage, Timer } from 'claude-code'
3
4import { classifyPing, decide, parseKeepwarmArgs } from './policy'
5import type { Config, PingRecord, Session } from '../types'
6
7const MIN = 60_000
8const HOUR = 60 * MIN
9
10const PING_PROMPT =
11 '[cache-keeper keep-alive] This is an automated ping to keep the prompt cache warm. Reply with exactly: ok'
12
13const INITIAL: Session = {
14 mode: 'auto',
15 untilMs: null,
16 lastRequestAt: null,
17 lastTurnAt: null,
18 isTurnRunning: false,
19 pings: 0,
20 lastPing: null,
21 stopReason: null,
22 probe: null,
23}
24
25const sessionAtom = atom({ plugin: 'cache-keeper', key: 'session' } as const, INITIAL)
26
27// Module variables reset on a hot reload; session.start fires again then and
28// reschedules from $.state, which survives.
29let config: Config = configFrom({})
30let isActive = false
31let isPinging = false
32let timer: Timer | null = null
33
34function num(v: unknown, fallback: number) {
35 return typeof v === 'number' && Number.isFinite(v) ? v : fallback
36}
37
38function configFrom(options: PluginOptions): Config {
39 return {
40 enabled: options.enabled !== false,
41 ttlMs: num(options.ttlMinutes, 60) * MIN,
42 leadMs: num(options.leadMinutes, 5) * MIN,
43 maxIdleMs: num(options.maxIdleHours, 8) * HOUR,
44 minContextTokens: num(options.minContextTokens, 60_000),
45 maxLimitPercent: num(options.maxLimitPercent, 85),
46 }
47}
48
49function hhmm(ms: number) {
50 const d = new Date(ms)
51 return `${String(d.getHours()).padStart(2, '0')}:${String(d.getMinutes()).padStart(2, '0')}`
52}
53
54function kTokens(n: number) {
55 return `${Math.round(n / 1000)}k`
56}
57
58function limitsOf(u: SessionUsage) {
59 return u.rateLimits.map(l => ({ kind: l.kind, percentUsed: l.percentUsed }))
60}
61
62function cancelTimer() {
63 timer?.cancel()
64 timer = null
65}
66
67/** Appends one JSON line to ~/.claude/cache-keeper/<session id>.jsonl; best effort. */
68async function appendLog($: EngineInterface, record: Record<string, unknown>) {
69 try {
70 const home = await $.env.get('HOME')
71 if (!home) return
72 const path = `${home}/.claude/cache-keeper/${await $.session.id()}.jsonl`
73 const before = (await $.fs.exists(path)) ? await $.fs.read(path) : ''
74 const line = JSON.stringify({ ts: new Date(await $.clock.now()).toISOString(), ...record })
75 await $.fs.write(path, `${before}${line}\n`)
76 } catch {
77 // Logging must never stop the keeper.
78 }
79}
80
81async function stop($: EngineInterface, reason: string, extra: Record<string, unknown> = {}) {
82 const s = await read($, sessionAtom)
83 if (s.stopReason !== reason) {
84 await update($, sessionAtom, cur => ({ ...cur, stopReason: reason }))
85 await appendLog($, { kind: 'stop', reason, pings: s.pings, ...extra })
86 }
87 $.ui.status(reason === 'off' ? undefined : `cache-keeper: stopped (${reason})`)
88}
89
90/**
91 * A fork that missed the cache (anthropics/claude-code#100083) costs about a
92 * full re-cache, and a fork that hit but didn't extend the main entry's TTL
93 * buys nothing. Either benches automatic pings, in every session, until
94 * Claude Code's version changes. A manual `/keepwarm now` hit lifts a miss.
95 */
96async function isBenched($: EngineInterface) {
97 const version = (await $.session.version()).version
98 const miss = (await $.store.get('forkMiss')) as { version?: string } | undefined
99 const refreshFail = (await $.store.get('refreshFail')) as { version?: string } | undefined
100 return miss?.version === version || refreshFail?.version === version
101}
102
103/**
104 * Judges the first real turn after a pinged break longer than the TTL: had the
105 * pings not refreshed the main entry, that turn re-writes most of the context.
106 */
107async function verifyRefresh($: EngineInterface, probe: NonNullable<Session['probe']>, usage: ModelUsage) {
108 const verdict = usage.cache_creation_input_tokens < 0.5 * probe.contextTokens ? 'warm' : 'cold'
109 const version = (await $.session.version()).version
110 await appendLog($, {
111 kind: 'verify',
112 verdict,
113 ...probe,
114 cacheRead: usage.cache_read_input_tokens,
115 cacheWrite: usage.cache_creation_input_tokens,
116 version,
117 })
118 const at = await $.clock.now()
119 if (verdict === 'cold') {
120 await $.store.set('refreshFail', { version, at })
121 $.ui.toast('cache-keeper: pings did not keep the cache warm on this version; automatic pings paused')
122 } else {
123 await $.store.set('refreshVerified', { version, at })
124 }
125}
126
127async function ping($: EngineInterface): Promise<PingRecord> {
128 isPinging = true
129 try {
130 const before = await $.session.usage()
131 const result = await $.model.fork({ prompt: PING_PROMPT })
132 const now = await $.clock.now()
133 const after = await $.session.usage()
134 const usage: ModelUsage | undefined = 'usage' in result ? result.usage : undefined
135 const outcome = usage ? classifyPing(usage) : 'error'
136 const record: PingRecord = {
137 at: now,
138 outcome,
139 cacheRead: usage?.cache_read_input_tokens ?? 0,
140 cacheWrite: usage?.cache_creation_input_tokens ?? 0,
141 input: usage?.input_tokens ?? 0,
142 output: usage?.output_tokens ?? 0,
143 ...(result.isAnswered ? {} : { detail: result.reason }),
144 }
145 const s = await update($, sessionAtom, cur => ({
146 ...cur,
147 lastPing: record,
148 ...(outcome === 'hit' ? { lastRequestAt: now, pings: cur.pings + 1, stopReason: null } : {}),
149 }))
150 await appendLog($, {
151 kind: 'ping',
152 ...record,
153 ping: s.pings,
154 idleMinutes: s.lastTurnAt === null ? null : Math.round((now - s.lastTurnAt) / MIN),
155 contextTokens: before.context.tokens ?? null,
156 limitsBefore: limitsOf(before),
157 limitsAfter: limitsOf(after),
158 costBefore: before.cost?.usd ?? null,
159 costAfter: after.cost?.usd ?? null,
160 })
161 if (outcome === 'miss') {
162 await $.store.set('forkMiss', {
163 version: (await $.session.version()).version,
164 at: now,
165 cacheRead: record.cacheRead,
166 cacheWrite: record.cacheWrite,
167 })
168 } else if (outcome === 'hit') {
169 await $.store.delete('forkMiss')
170 }
171 if (outcome === 'hit') {
172 $.ui.toast(`cache-keeper: kept ${kTokens(record.cacheRead)} cached (ping ${s.pings})`)
173 } else {
174 await stop($, outcome === 'miss' ? 'cache-miss' : 'error', { detail: record.detail ?? null })
175 }
176 return record
177 } finally {
178 isPinging = false
179 }
180}
181
182/** Decides what the session's cache needs now and acts: wait, ping, or stop. */
183async function schedule($: EngineInterface): Promise<void> {
184 cancelTimer()
185 if (!isActive || isPinging) return
186
187 const s = await read($, sessionAtom)
188 const now = await $.clock.now()
189 const usage = await $.session.usage()
190 const d = decide(s, now, { contextTokens: usage.context.tokens ?? 0, limits: limitsOf(usage) }, config)
191
192 switch (d.action) {
193 case 'idle':
194 $.ui.status(undefined)
195 return
196 case 'wait':
197 $.ui.status(`cache warm until ${hhmm(d.at + config.leadMs)} · keep-alive ${hhmm(d.at)}`)
198 timer = $.clock.after(d.at - now, () => void schedule($))
199 return
200 case 'stop':
201 await stop($, d.reason)
202 return
203 case 'ping':
204 if (await isBenched($)) {
205 await stop($, 'fork-miss-on-this-version')
206 return
207 }
208 if ((await ping($)).outcome === 'hit') await schedule($)
209 return
210 }
211}
212
213async function statusText($: EngineInterface) {
214 const s = await read($, sessionAtom)
215 const now = await $.clock.now()
216 const parts = [`keep-warm ${s.mode}${s.mode === 'auto' ? (config.enabled ? ' (on)' : ' (off)') : ''}`]
217 if (s.lastRequestAt !== null) {
218 const expiresAt = s.lastRequestAt + config.ttlMs
219 parts.push(now < expiresAt ? `cache warm until ${hhmm(expiresAt)}` : `cache likely cold since ${hhmm(expiresAt)}`)
220 if (timer !== null) parts.push(`next ping ${hhmm(expiresAt - config.leadMs)}`)
221 }
222 if (s.untilMs !== null) parts.push(`keeping warm until ${hhmm(s.untilMs)}`)
223 parts.push(`${s.pings} ping${s.pings === 1 ? '' : 's'} since last turn`)
224 if (s.lastPing) {
225 parts.push(`last ping ${hhmm(s.lastPing.at)} ${s.lastPing.outcome} (${kTokens(s.lastPing.cacheRead)} cached)`)
226 }
227 if (s.stopReason) parts.push(`stopped: ${s.stopReason}`)
228 return parts.join(' · ')
229}
230
231export const register: Register = (on, options) => {
232 config = configFrom(options)
233 isActive = false
234 isPinging = false
235 timer = null
236
237 on('session.start', async ($, e, next) => {
238 const result = await next(e)
239 isActive = e.isInteractive
240 if (!isActive) return result
241 await $.command.register({
242 name: 'keepwarm',
243 description: 'Prompt-cache keep-alive: status, on, off, auto, now, for <n>h, until HH:MM',
244 argumentHint: '[status|on|off|auto|now|for 3h|until 18:00]',
245 })
246 await update($, sessionAtom, s => ({ ...s, isTurnRunning: false }))
247 await schedule($)
248 return result
249 })
250
251 on('turn.start', async ($, e, next) => {
252 cancelTimer()
253 const now = await $.clock.now()
254 const s = await read($, sessionAtom)
255 const isProbe =
256 s.pings > 0 &&
257 s.lastTurnAt !== null &&
258 s.lastRequestAt !== null &&
259 now - s.lastTurnAt > config.ttlMs &&
260 now - s.lastRequestAt < config.ttlMs
261 const probe = isProbe
262 ? {
263 contextTokens: (await $.session.usage()).context.tokens ?? 0,
264 idleMinutes: Math.round((now - (s.lastTurnAt ?? now)) / MIN),
265 pings: s.pings,
266 }
267 : null
268 await update($, sessionAtom, cur => ({ ...cur, isTurnRunning: true, probe }))
269 return next(e)
270 })
271
272 on('turn.complete', async ($, e, next) => {
273 const result = await next(e)
274 if (e.agentId !== undefined || isPinging) return result
275 const now = await $.clock.now()
276 const { probe } = await read($, sessionAtom)
277 if (probe && e.usage !== undefined) await verifyRefresh($, probe, e.usage)
278 await update($, sessionAtom, s => ({
279 ...s,
280 probe: null,
281 isTurnRunning: false,
282 lastRequestAt: now,
283 lastTurnAt: now,
284 pings: 0,
285 stopReason: null,
286 }))
287 await schedule($)
288 return result
289 })
290
291 on('command.run', { command: 'keepwarm' }, async ($, e) => {
292 const cmd = parseKeepwarmArgs(e.args, await $.clock.now(), new Date().getTimezoneOffset())
293 switch (cmd.kind) {
294 case 'error':
295 return { text: cmd.message }
296 case 'status':
297 return { text: await statusText($) }
298 case 'mode':
299 await update($, sessionAtom, s => ({ ...s, mode: cmd.mode, stopReason: null }))
300 await schedule($)
301 return { text: `Keep-warm ${cmd.mode}. ${await statusText($)}` }
302 case 'until':
303 await update($, sessionAtom, s => ({ ...s, mode: 'on' as const, untilMs: cmd.untilMs, stopReason: null }))
304 await schedule($)
305 return { text: `Keeping warm until ${hhmm(cmd.untilMs)}. ${await statusText($)}` }
306 case 'now': {
307 cancelTimer()
308 const r = await ping($)
309 if (r.outcome === 'hit') await schedule($)
310 return {
311 text: `Ping ${r.outcome}: ${kTokens(r.cacheRead)} read from cache, ${kTokens(r.cacheWrite)} written, ${r.output} output tokens.`,
312 }
313 }
314 }
315 })
316}
317hooks/policy.ts 93 lines1import type { ModelUsage, SessionRateLimit } from 'claude-code'
2
3import type { Config, Mode, Session } from '../types'
4
5const MIN = 60_000
6const HOUR = 60 * MIN
7const DAY = 24 * HOUR
8
9export type Facts = {
10 contextTokens: number
11 limits: readonly Pick<SessionRateLimit, 'kind' | 'percentUsed'>[]
12}
13
14export type StopReason = 'off' | 'expired' | 'idle-limit' | 'small-context' | 'near-limit'
15
16export type Decision =
17 | { action: 'idle' }
18 | { action: 'wait'; at: number }
19 | { action: 'ping' }
20 | { action: 'stop'; reason: StopReason }
21
22/**
23 * What to do about the session's cache at `now`. Pure: the caller supplies
24 * the session's state, the context size and plan usage, and the config.
25 */
26export const decide = (s: Session, now: number, facts: Facts, config: Config): Decision => {
27 const isOn = s.mode === 'on' || (s.mode === 'auto' && config.enabled)
28 if (!isOn) return { action: 'stop', reason: 'off' }
29 if (s.isTurnRunning || s.lastRequestAt === null) return { action: 'idle' }
30
31 const expiresAt = s.lastRequestAt + config.ttlMs
32 if (now >= expiresAt) return { action: 'stop', reason: 'expired' }
33
34 const pingAt = expiresAt - config.leadMs
35 if (now < pingAt) return { action: 'wait', at: pingAt }
36
37 // lastTurnAt moves on every main-loop turn, typed or not (a background
38 // agent's result, a /loop wakeup): those extend the idle window by design.
39 const keepUntil = s.untilMs ?? (s.lastTurnAt ?? s.lastRequestAt) + config.maxIdleMs
40 if (now > keepUntil) return { action: 'stop', reason: 'idle-limit' }
41 if (facts.contextTokens < config.minContextTokens) return { action: 'stop', reason: 'small-context' }
42 if (facts.limits.some(l => l.percentUsed >= config.maxLimitPercent)) {
43 return { action: 'stop', reason: 'near-limit' }
44 }
45
46 return { action: 'ping' }
47}
48
49/** A fork the cache served at least 80% of is a hit; otherwise it paid for the prefix. */
50export const classifyPing = (u: ModelUsage): 'hit' | 'miss' => {
51 const prompt = u.cache_read_input_tokens + u.cache_creation_input_tokens + u.input_tokens
52 return prompt > 0 && u.cache_read_input_tokens / prompt >= 0.8 ? 'hit' : 'miss'
53}
54
55export type KeepwarmCommand =
56 | { kind: 'status' }
57 | { kind: 'now' }
58 | { kind: 'mode'; mode: Mode }
59 | { kind: 'until'; untilMs: number }
60 | { kind: 'error'; message: string }
61
62export const USAGE = 'Usage: /keepwarm [status | on | off | auto | now | for <n>h|<n>m | until HH:MM]'
63
64/**
65 * Parses /keepwarm's arguments. `tzOffsetMinutes` is Date#getTimezoneOffset():
66 * minutes to add to local time to get UTC.
67 */
68export const parseKeepwarmArgs = (args: string, now: number, tzOffsetMinutes: number): KeepwarmCommand => {
69 const [head, arg] = args.trim().toLowerCase().split(/\s+/).filter(Boolean)
70
71 if (head === undefined || head === 'status') return { kind: 'status' }
72 if (head === 'now') return { kind: 'now' }
73 if (head === 'on' || head === 'off' || head === 'auto') return { kind: 'mode', mode: head }
74
75 if (head === 'for' && arg !== undefined) {
76 const m = /^(\d+(?:\.\d+)?)(h|m)$/.exec(arg)
77 if (m) return { kind: 'until', untilMs: now + Number(m[1]) * (m[2] === 'h' ? HOUR : MIN) }
78 }
79
80 if (head === 'until' && arg !== undefined) {
81 const m = /^(\d{1,2}):(\d{2})$/.exec(arg)
82 if (m && Number(m[1]) < 24 && Number(m[2]) < 60) {
83 const offset = tzOffsetMinutes * MIN
84 const localMidnight = Math.floor((now - offset) / DAY) * DAY
85 let untilMs = localMidnight + Number(m[1]) * HOUR + Number(m[2]) * MIN + offset
86 if (untilMs <= now) untilMs += DAY
87 return { kind: 'until', untilMs }
88 }
89 }
90
91 return { kind: 'error', message: USAGE }
92}
93types/index.d.ts 43 lines1export type Mode = 'auto' | 'on' | 'off'
2
3export type Config = {
4 enabled: boolean
5 ttlMs: number
6 leadMs: number
7 maxIdleMs: number
8 minContextTokens: number
9 maxLimitPercent: number
10}
11
12export type PingRecord = {
13 at: number
14 outcome: 'hit' | 'miss' | 'error'
15 cacheRead: number
16 cacheWrite: number
17 input: number
18 output: number
19 detail?: string
20}
21
22export type Session = {
23 mode: Mode
24 untilMs: number | null
25 /** When the cache was last refreshed: the end of a main turn or a ping. */
26 lastRequestAt: number | null
27 /** When the last real (non-ping) main turn ended. */
28 lastTurnAt: number | null
29 isTurnRunning: boolean
30 /** Pings since the last real turn. */
31 pings: number
32 lastPing: PingRecord | null
33 stopReason: string | null
34 /** Set at the start of a real turn that follows a pinged break longer than the TTL. */
35 probe: { contextTokens: number; idleMinutes: number; pings: number } | null
36}
37
38declare module 'claude-code' {
39 interface PluginState {
40 'cache-keeper': { session: Session }
41 }
42}
43