Shows above the prompt how long the cache stays warm, what each message costs on your Claude plan, and one habit that saves usage

Claude Code mods by Henry Chua. A mod is a small add-on that runs inside Claude Code and draws on the screen.
Stop wasting Claude credits. Claude keeps your chat warm for an hour on a Pro or Max plan (5 minutes on the API). Let it go cold and your next message re-sends the whole chat at up to 40x the warm price. This mod shows you, above where you type:
/clear when you start something new on a cold chatNeeds Claude Code 2.1.287 or newer.
claude update
claude plugin marketplace add EasyAIHenry/henry-mods
claude plugin install cache-warm@henry-mods
Then start claude. The lines show after Claude's first reply. They also show in the Code tab of the Claude desktop app.
| Command | What it does |
|---|---|
/cache-warm | Show the band and the current reading |
/cache-warm hide | Hide the band |
/cache-warm plan pro (or max5, max20, api, auto) | Adds what each chat costs as a share of your plan |
/cache-warm ttl 5m (or 1h, auto) | Set the cache lifetime yourself |
To remove it: claude plugin uninstall cache-warm@henry-mods.
A mod runs with your permissions, so check any mod before you install it. You can list what this one does without running it:
git clone https://github.com/EasyAIHenry/henry-mods
claude plugin validate henry-mods/cache-warm
It reads the clock and the session's token usage, saves two settings (your plan and the cache lifetime), and draws. It makes no network requests and reads no files.
MIT licence.
hooks/register.tsx 435 lines1import { atom, read, update } from 'claude-code'
2import type { Register } from 'claude-code'
3
4import type { DraftSize, LastMessage, Plan, Ttl, TtlMode } from '../types'
5
6const lastAt = atom({ plugin: 'cache-warm', key: 'lastAt' } as const, null)
7const tokens = atom({ plugin: 'cache-warm', key: 'tokens' } as const, 0)
8const autoTtl = atom({ plugin: 'cache-warm', key: 'autoTtl' } as const, '5m')
9const mode = atom({ plugin: 'cache-warm', key: 'mode' } as const, 'auto')
10const isHidden = atom({ plugin: 'cache-warm', key: 'isHidden' } as const, false)
11const model = atom({ plugin: 'cache-warm', key: 'model' } as const, null)
12const effort = atom({ plugin: 'cache-warm', key: 'effort' } as const, null)
13const draft = atom({ plugin: 'cache-warm', key: 'draft' } as const, 'none')
14const hasLimits = atom({ plugin: 'cache-warm', key: 'hasLimits' } as const, false)
15const lastMessage = atom({ plugin: 'cache-warm', key: 'lastMessage' } as const, null)
16const weekStart = atom({ plugin: 'cache-warm', key: 'weekStart' } as const, null)
17const weekNow = atom({ plugin: 'cache-warm', key: 'weekNow' } as const, null)
18const plan = atom({ plugin: 'cache-warm', key: 'plan' } as const, 'auto')
19
20const MINUTE = 60_000
21const TTL_MS: Record<Ttl, number> = { '5m': 5 * MINUTE, '1h': 60 * MINUTE }
22// The band turns amber and a toast shows when this much time is left
23const WARN_MS = 5 * MINUTE
24// Below this many tokens a cold send is cheap, so the tips stay quiet
25const BIG_CONTEXT = 20_000
26
27type Price = { input: number; read: number }
28// Dollars per million tokens at standard speed: input, and a cache read (Anthropic
29// list prices, 25 Sep 2026). Checked in order, so the longer ids come first
30const PRICES: Array<[string, Price]> = [
31 ['fable-5-1', { input: 10, read: 0.25 }],
32 ['mythos-5-1', { input: 10, read: 0.25 }],
33 ['fable-5', { input: 10, read: 1 }],
34 ['opus-5-5', { input: 4, read: 0.2 }],
35 ['opus', { input: 5, read: 0.5 }],
36 ['sonnet-5', { input: 2, read: 0.2 }],
37 ['sonnet-4-6', { input: 3, read: 0.3 }],
38 ['haiku-4-5', { input: 1, read: 0.1 }],
39]
40// A cache write costs this many times the input price
41const WRITE: Record<Ttl, number> = { '5m': 1.25, '1h': 2 }
42// Models where /effort keeps the cache, on an API key or a Claude subscription
43const EFFORT_KEEPS_CACHE = ['opus-5-5', 'sonnet-5-5', 'fable-5-1']
44const PLANS: Record<'pro' | 'max5' | 'max20', { label: string; usd: number }> = {
45 pro: { label: 'Pro', usd: 20 },
46 max5: { label: 'Max 5x', usd: 100 },
47 max20: { label: 'Max 20x', usd: 200 },
48}
49const PLAN_CHOICES = ['auto', 'pro', 'max5', 'max20', 'api']
50
51// Everything the band draws from, read once per drawing
52type Snapshot = {
53 at: number | null
54 tokens: number
55 ttl: Ttl
56 now: number
57 isWorking: boolean
58 model: string | null
59 effort: string | number | null
60 draft: DraftSize
61 isSubscription: boolean
62 last: LastMessage | null
63 plan: Plan
64 weekShare: number | null
65}
66
67type Line = { text: string; color?: string; isDim?: boolean }
68
69function priceOf(id: string | null): Price | null {
70 if (id === null) return null
71 const hit = PRICES.find(([key]) => id.includes(key))
72 return hit ? hit[1] : null
73}
74
75// How many warm sends one cold send costs: the write rate over the read rate
76function coldMultiple(id: string | null, ttl: Ttl): string {
77 const price = priceOf(id)
78 const x = price ? (price.input * WRITE[ttl]) / price.read : WRITE[ttl] / 0.1
79 return (Number.isInteger(x) ? String(x) : x.toFixed(1)) + 'x'
80}
81
82function formatTokens(n: number): string {
83 if (n >= 1_000_000) return (n / 1_000_000).toFixed(1) + 'M'
84 if (n >= 1_000) return Math.round(n / 1_000) + 'k'
85 return String(n)
86}
87
88function formatLeft(ms: number): string {
89 if (ms >= 2 * MINUTE) return Math.ceil(ms / MINUTE) + ' min'
90 const seconds = Math.ceil(ms / 1000)
91 return Math.floor(seconds / 60) + ':' + String(seconds % 60).padStart(2, '0')
92}
93
94function money(usd: number): string {
95 if (usd < 0.01) return 'under $0.01'
96 return '$' + (usd < 100 ? usd.toFixed(2) : String(Math.round(usd)))
97}
98
99function percent(points: number): string {
100 return points < 0.1 ? 'under 0.1%' : points.toFixed(1) + '%'
101}
102
103function draftSize(length: number): DraftSize {
104 if (length === 0) return 'none'
105 if (length < 25) return 'short'
106 return length < 400 ? 'normal' : 'long'
107}
108
109function msLeft(s: Snapshot): number | null {
110 return s.at === null ? null : TTL_MS[s.ttl] - (s.now - s.at)
111}
112
113// What re-sending the chat costs at API prices, warm and cold; null for an unknown model
114function sendCost(s: Snapshot): { warm: number; cold: number } | null {
115 const price = priceOf(s.model)
116 if (price === null || s.tokens === 0) return null
117 return {
118 warm: (s.tokens * price.read) / 1_000_000,
119 cold: (s.tokens * price.input * WRITE[s.ttl]) / 1_000_000,
120 }
121}
122
123// Line 1: is the cache warm, and for how long
124function cacheLine(s: Snapshot): Line {
125 const size = s.tokens > 0 ? formatTokens(s.tokens) + ' tokens' : 'no context yet'
126 const multiple = coldMultiple(s.model, s.ttl)
127 if (s.isWorking) return { text: '● cache warm · Claude is working · ' + size, color: 'success' }
128 const left = msLeft(s)
129 if (left === null) return { text: '○ cache: no reading yet · it starts with Claude’s first reply', isDim: true }
130 if (left > WARN_MS) {
131 return {
132 text: '● cache warm · ' + formatLeft(left) + ' left of ' + TTL_MS[s.ttl] / MINUTE + ' · ' + size + ' · going cold costs ' + multiple,
133 color: 'success',
134 }
135 }
136 if (left > 0) {
137 return { text: '◐ cache cools in ' + formatLeft(left) + ' · reply now, or /compact at a break · ' + size, color: 'warning' }
138 }
139 return { text: '○ cache cold · your next message re-sends ' + size + ' at ' + multiple + ' the warm cost', color: 'error' }
140}
141
142// Line 2: what the next send costs, and what the last one did
143function moneyLine(s: Snapshot): Line | null {
144 const parts: string[] = []
145 const cost = sendCost(s)
146 if (cost) {
147 parts.push('next send ≈ ' + money(cost.warm) + (s.isSubscription ? ' at API prices' : '') + ', ' + money(cost.cold) + ' if cold')
148 }
149 if (s.last) {
150 if (s.isSubscription && s.last.fivePct !== null) parts.push('last message used ' + percent(s.last.fivePct) + ' of your 5-hour limit')
151 else if (s.last.usd !== null) parts.push('last message ' + money(s.last.usd))
152 }
153 if (s.plan !== 'auto' && s.plan !== 'api' && s.weekShare !== null && s.weekShare > 0) {
154 const { label, usd } = PLANS[s.plan]
155 // The plan's price for one week, times the share of the weekly limit this chat used
156 parts.push('this chat ≈ ' + money((s.weekShare / 100) * ((usd * 12) / 52)) + ' of your ' + label + ' plan')
157 }
158 return parts.length > 0 ? { text: parts.join(' · '), isDim: true } : null
159}
160
161// Line 3: one habit that saves usage, when it applies right now
162function tipLine(s: Snapshot): Line | null {
163 const size = formatTokens(s.tokens) + ' tokens'
164 const isBig = s.tokens >= BIG_CONTEXT
165 if (s.isWorking) {
166 return { text: 'Let Claude think. Stopping it to ask again re-sends all ' + size + '.', color: 'warning' }
167 }
168 if (s.draft === 'short' && isBig) {
169 return { text: 'Short follow-up? One full message beats three short ones: each send re-reads all ' + size + '.', color: 'warning' }
170 }
171 const isLowEffort = s.effort === 'low' || s.effort === 'medium'
172 const keepsCache = s.model !== null && EFFORT_KEEPS_CACHE.some(key => s.model.includes(key))
173 if (s.draft === 'long' && isLowEffort && keepsCache) {
174 return { text: 'Big ask at ' + s.effort + ' effort? /effort high lets Claude think longer, and on this model it keeps your cache.', color: 'warning' }
175 }
176 const left = msLeft(s)
177 if (left !== null && left <= 0 && isBig) {
178 const cost = sendCost(s)
179 const rewrite = cost ? 'the ' + money(cost.cold) + ' rewrite' : 'rewriting ' + size
180 return { text: 'Starting something new? /clear first, so the next send skips ' + rewrite + '.', color: 'warning' }
181 }
182 return null
183}
184
185// The lifetime in use: the person's choice, or the one the plan gives
186async function currentTtl($): Promise<Ttl> {
187 const chosen = await read($, mode)
188 return chosen === 'auto' ? await read($, autoTtl) : chosen
189}
190
191async function snapshot($, isWorking: boolean): Promise<Snapshot> {
192 const chosenPlan = await read($, plan)
193 const start = await read($, weekStart)
194 const now = await read($, weekNow)
195 return {
196 at: await read($, lastAt),
197 tokens: await read($, tokens),
198 ttl: await currentTtl($),
199 now: await $.clock.now(),
200 isWorking,
201 model: await read($, model),
202 effort: await read($, effort),
203 draft: await read($, draft),
204 isSubscription: chosenPlan === 'api' ? false : chosenPlan !== 'auto' || (await read($, hasLimits)),
205 last: await read($, lastMessage),
206 plan: chosenPlan,
207 weekShare: start !== null && now !== null ? Math.max(0, now - start) : null,
208 }
209}
210
211// Copy the saved choices from $.store into $.state
212async function loadChoices($) {
213 const savedMode = await $.store.get('mode')
214 if (savedMode === 'auto' || savedMode === '5m' || savedMode === '1h') await update($, mode, () => savedMode)
215 const savedPlan = await $.store.get('plan')
216 if (typeof savedPlan === 'string' && PLAN_CHOICES.includes(savedPlan)) {
217 const value = savedPlan as Plan
218 await update($, plan, () => value)
219 }
220}
221
222// The percent used of one rate-limit window, or null when the reading has none
223function windowPct(usage, kind: string): number | null {
224 const found = usage.rateLimits.find(l => l.kind === kind)
225 return found ? found.percentUsed : null
226}
227
228export const register: Register = on => {
229 // The lastAt the amber toast last fired for, so it fires once per cool-down
230 let warnedFor: number | null = null
231 // The session's cost and 5-hour percent when the current turn started
232 let costAtStart: number | null = null
233 let fiveAtStart: number | null = null
234
235 on('session.start', async ($, e, next) => {
236 await $.command.register({
237 name: 'cache-warm',
238 description: 'Cache band: show, hide, set the cache lifetime (ttl) or your Claude plan (plan)',
239 argumentHint: '[show | hide | ttl auto|5m|1h | plan auto|pro|max5|max20|api]',
240 immediate: true,
241 })
242 await loadChoices($)
243
244 // Redraw once a second so the countdown moves, and warn once when it gets short
245 $.clock.every(1000, async () => {
246 const at = await read($, lastAt)
247 if (at !== null && at !== warnedFor) {
248 const left = TTL_MS[await currentTtl($)] - ((await $.clock.now()) - at)
249 if (left > 0 && left <= WARN_MS) {
250 warnedFor = at
251 $.ui.toast('Cache cools in 5 min. Reply now, or /compact at a break.')
252 }
253 }
254 $.ui.invalidate('ui.render')
255 })
256
257 return next(e)
258 })
259
260 // /clear, /resume and /branch reset $.state, so load the saved choices again
261 on('classic.SessionStart', { source: ['clear', 'resume', 'fork'] }, async ($, e, next) => {
262 await loadChoices($)
263 return next(e)
264 })
265
266 // Track how much is typed, so the tips can tell a quick "ok" from a big ask
267 on('prompt.edit', async ($, e, next) => {
268 const box = await next(e)
269 const size = draftSize(box.text.length)
270 if (size !== (await read($, draft))) await update($, draft, () => size)
271 return box
272 })
273
274 on('prompt.submit', async ($, e, next) => {
275 await update($, draft, () => 'none')
276 return next(e)
277 })
278
279 // A main-conversation turn starts: note the cost so far, to price this message
280 on('turn.start', async ($, e, next) => {
281 const usage = await $.session.usage()
282 costAtStart = usage.cost ? usage.cost.usd : null
283 fiveAtStart = windowPct(usage, 'five_hour')
284 return next(e)
285 })
286
287 // Each request in the main conversation reads or writes the cache, which restarts its clock
288 on('turn.step', async function* ($, e, next) {
289 if (e.agentId === undefined) {
290 const at = await $.clock.now()
291 const named = e.model
292 const level = e.effort ?? null
293 await update($, lastAt, () => at)
294 await update($, model, () => named)
295 await update($, effort, () => level)
296 }
297 return yield* next(e)
298 })
299
300 // After a main-conversation turn, read the context size, the plan's lifetime and what the turn cost
301 on('turn.complete', async ($, e, next) => {
302 const result = await next(e)
303 if (e.agentId !== undefined) return result
304
305 const usage = await $.session.usage()
306 const n = usage.context.tokens
307 if (n !== undefined) await update($, tokens, () => n)
308 if (e.usage) {
309 const answered = e.usage.model
310 await update($, model, () => answered)
311 }
312
313 // A subscription reports its 5-hour and weekly windows. Within them Claude Code
314 // asks for the 1-hour cache; past them it draws on usage credits at 5 minutes
315 const five = windowPct(usage, 'five_hour')
316 const week = windowPct(usage, 'seven_day')
317 const isLimited = five !== null || week !== null
318 const isOverLimit = (five !== null && five >= 100) || (week !== null && week >= 100)
319 const ttl: Ttl = isLimited && !isOverLimit ? '1h' : '5m'
320 await update($, hasLimits, () => isLimited)
321 await update($, autoTtl, () => ttl)
322
323 // A window that reset during the turn reads lower than at the start: count from zero
324 const cost = usage.cost ? usage.cost.usd : null
325 const spent: LastMessage = {
326 usd: cost !== null && costAtStart !== null ? Math.max(0, cost - costAtStart) : null,
327 fivePct: five !== null && fiveAtStart !== null ? (five >= fiveAtStart ? five - fiveAtStart : five) : null,
328 }
329 await update($, lastMessage, () => spent)
330
331 if (week !== null) {
332 const first = await read($, weekStart)
333 // The weekly window reset since the first reading: count from this one
334 if (first === null || week < first) await update($, weekStart, () => week)
335 await update($, weekNow, () => week)
336 }
337
338 return result
339 })
340
341 // Each model has its own cache, so a switch starts cold
342 on('classic.PostModelSwitch', async ($, e, next) => {
343 const switched = e.to_model
344 await update($, lastAt, () => 0)
345 await update($, model, () => switched)
346 return next(e)
347 })
348
349 on('command.run', { command: 'cache-warm' }, async ($, e) => {
350 const words = (e.args ?? '').trim().split(/\s+/).filter(Boolean)
351
352 if (words[0] === 'hide') {
353 await update($, isHidden, () => true)
354 return { text: 'Cache band hidden. /cache-warm show brings it back.' }
355 }
356
357 if (words[0] === 'ttl') {
358 const choice = words[1]
359 if (choice !== 'auto' && choice !== '5m' && choice !== '1h') {
360 return { text: 'Use /cache-warm ttl auto, 5m or 1h. auto reads it from your plan.' }
361 }
362 const value: TtlMode = choice
363 await update($, mode, () => value)
364 await $.store.set('mode', value)
365 return { text: 'Cache lifetime set to ' + value + '.' }
366 }
367
368 if (words[0] === 'plan') {
369 const choice = words[1] ?? ''
370 if (!PLAN_CHOICES.includes(choice)) {
371 return { text: 'Use /cache-warm plan pro, max5, max20, api or auto. It prices your chat as a share of your plan.' }
372 }
373 const value = choice as Plan
374 await update($, plan, () => value)
375 await $.store.set('plan', value)
376 const named = value === 'pro' || value === 'max5' || value === 'max20' ? PLANS[value].label + ' ($' + PLANS[value].usd + '/month)' : value
377 return { text: 'Plan set to ' + named + '.' }
378 }
379
380 if (words.length > 0 && words[0] !== 'show') {
381 return { text: 'Use /cache-warm show, hide, ttl auto|5m|1h, or plan auto|pro|max5|max20|api.' }
382 }
383
384 await update($, isHidden, () => false)
385 const s = await snapshot($, false)
386 const lines = [cacheLine(s), moneyLine(s)].filter((line): line is Line => line !== null).map(line => line.text)
387 // Before the first reply there is no rate-limit reading to tell the plan by
388 if (s.at !== null) {
389 const chosen = await read($, mode)
390 lines.push('Cache lifetime ' + s.ttl + (chosen === 'auto' ? ', from your plan.' : ', set by you.'))
391 }
392 return { text: lines.join('\n') }
393 })
394
395 // The hint line under the prompt (terminal): what this send costs, while you type
396 on('ui.render', { component: 'PromptHint' }, async ($, e, next) => {
397 if (!e.props.isDraft || (await read($, isHidden))) return next(e)
398 const cost = sendCost(await snapshot($, e.props.isWorking))
399 if (cost === null) return next(e)
400 const tail = (e.props.tail ?? '') + ' · this send ≈ ' + money(cost.warm)
401 return next({ ...e, props: { ...e.props, tail } })
402 })
403
404 on('ui.render', { component: 'AbovePrompt' }, async ($, e, next) => {
405 if (e.props.hasSurvey || (await read($, isHidden))) return next(e)
406
407 const s = await snapshot($, e.props.isWorking)
408 const cache = cacheLine(s)
409 const cost = moneyLine(s)
410 const tip = tipLine(s)
411 const { Box, Text } = $.ui.resolve(e)
412 // Keep whatever other mods draw in the band, under these lines
413 const theirs = await next(e)
414
415 return (
416 <Box flexDirection="column">
417 <Text color={cache.color} dimColor={cache.isDim} wrap="truncate-end">
418 {cache.text}
419 </Text>
420 {cost && (
421 <Text dimColor wrap="truncate-end">
422 {cost.text}
423 </Text>
424 )}
425 {tip && (
426 <Text color={tip.color} wrap="truncate-end">
427 {tip.text}
428 </Text>
429 )}
430 {theirs}
431 </Box>
432 )
433 })
434}
435types/index.d.ts 40 lines1// How long a cache entry lives after the last request reads or writes it
2export type Ttl = '5m' | '1h'
3
4// 'auto' reads it from the plan; '5m' or '1h' is the person's own choice
5export type TtlMode = 'auto' | Ttl
6
7// Which Claude plan pays for the session; 'auto' only tells a subscription from an API key
8export type Plan = 'auto' | 'pro' | 'max5' | 'max20' | 'api'
9
10// How much is typed in the prompt box: nothing, a few words, a normal ask, a big one
11export type DraftSize = 'none' | 'short' | 'normal' | 'long'
12
13// What the last message cost: dollars at API prices, and points of the 5-hour limit
14export type LastMessage = { usd: number | null; fivePct: number | null }
15
16declare module 'claude-code' {
17 interface PluginState {
18 'cache-warm': {
19 // When the last main-conversation request went out, in ms; 0 after a model switch
20 lastAt: number | null
21 // Prompt tokens the next request re-sends
22 tokens: number
23 autoTtl: Ttl
24 mode: TtlMode
25 isHidden: boolean
26 // The model the last main-conversation request named, and its effort
27 model: string | null
28 effort: string | number | null
29 draft: DraftSize
30 // Whether the last reading had a subscription's 5-hour and weekly windows
31 hasLimits: boolean
32 lastMessage: LastMessage | null
33 // The weekly window's percent at the first reading of the session, and now
34 weekStart: number | null
35 weekNow: number | null
36 plan: Plan
37 }
38 }
39}
40