Prompt-cache meter as one compact line above the Claude Code prompt: how many tokens each request read from, wrote to and sent past the cache, a live countdown…

A Claude Code mod. It shows prompt-cache use as one line above the prompt and counts down to the moment the cache lapses.
cache 100% · read 566k · wrote 1.1k · new 2 · 59:38 / 1h · warm
cache N%: share of the last request's prompt that the cache served. Green from 90%, yellow from 50%, red below.read, wrote, new: tokens read from the cache, written to it, and sent uncached.59:38 / 1h: time left, then the cache lifetime. The clock turns yellow, then red as the cache runs out.warm, expires soon, expired: /compact, expired, missed, cold, uncached, or off.The line stays on one row. When the width is short it drops the advice, then new, then wrote. It draws below whatever other band mods return.
/cache opens a pane with the full advice and one row per turn. /cache stop closes it. Options (ttl, warnSeconds, compactAtTokens, band, status, toast) are in .claude-plugin/plugin.json.
Needs function hooks (early access): Claude Code 2.1.259 or later with CLAUDE_CODE_ENABLE_FUNCTION_HOOKS=1. The lifetime follows Claude Code's own cache rules; see hooks/cache.ts (decideTtl).
Test with claude plugin test plugins/cache-meter. Adapted from MIT-licensed work; see THIRD_PARTY_NOTICES.md.
hooks/cache-meter.tsx 442 lines1/**
2 * cache-meter — Claude Mod (EARLY ACCESS)
3 *
4 * A prompt-cache meter for Claude Code. Every main-loop request reports how
5 * many prompt tokens the cache served (`cache_read_input_tokens`), wrote
6 * (`cache_creation_input_tokens`) and sent uncached (`input_tokens`); this mod
7 * keeps those per request and per turn, counts down to the moment the cache
8 * lapses, and says what to do about it: keep going, /compact or /clear.
9 *
10 * - `turn.step` reads each main-loop request's usage (subagents have their
11 * own prefixes and are left out)
12 * - `$.clock.every(1000)` redraws the countdown, and only while its text
13 * changes: an idle, expired session costs nothing
14 * - one compact line above the prompt (the AbovePrompt component; it keeps what other
15 * band mods draw), an optional status
16 * line entry, and `/cache`, a pane with one row per turn
17 *
18 * The lifetime is counted from the start of the request that last wrote or read
19 * the cache, as Anthropic documents it. Which lifetime Claude Code asked for
20 * follows its documented rules (see decideTtl in ./cache.ts): FORCE_PROMPT_CACHING_5M,
21 * CLAUDE_CODE_PROMPT_CACHE_TTL, the promptCacheTtl setting, ENABLE_PROMPT_CACHING_1H,
22 * then the account (1 hour on a Claude subscription, 5 minutes otherwise). The
23 * API names the TTL of a write but the mod API passes on only the token counts,
24 * so the mod also watches the gaps between requests (a hit after more than 5
25 * minutes proves 1 hour; see observeTtl). `ttl: "5m" | "1h"` pins it.
26 *
27 * Needs CLAUDE_CODE_ENABLE_FUNCTION_HOOKS=1 (Claude Code >= 2.1.259).
28 *
29 * Options (pluginConfigs["cache-meter@personal"].options):
30 * ttl: "auto" | "5m" | "1h" cache lifetime (default auto)
31 * warnSeconds: number countdown threshold for the warning (default 60)
32 * compactAtTokens: number prompt size that makes an expired cache suggest /compact (default 100000)
33 * band: boolean row above the prompt (default true)
34 * status: boolean entry under the prompt (default false)
35 * toast: boolean toasts near expiry: at warnSeconds, then 10, 3, 2 and 1 s (default true)
36 */
37import type { EngineInterface, Register } from 'claude-code'
38import {
39 advise,
40 COUNTDOWN_MARKS,
41 byTurn,
42 fit,
43 fmtClock,
44 fmtTokens,
45 hitRatio,
46 isCachingDisabled,
47 accountOf,
48 decideTtl,
49 observeTtl,
50 lifeColor,
51 lifeRatio,
52 nextToastMark,
53 positive,
54 promptTokens,
55 remainingMs,
56 rowRatio,
57 ratioColor,
58 fitParts,
59 segments,
60} from './cache.ts'
61import type { Account, Advice, BandPart, CacheEnv, Sample, Ttl } from './cache.ts'
62
63const PANE = 'cache'
64const COMMAND = 'cache'
65const KEEP = 200
66// below this a lapsed cache costs too little to interrupt anyone about
67const TOAST_MIN_TOKENS = 20_000
68
69let samples: Sample[] = []
70let ttl: Ttl = '5m'
71let baseTtl: Ttl = '5m'
72let pinned = false
73let observed: Ttl | undefined
74let setting: unknown
75let account: Account = 'other'
76let ttlSource = 'default'
77let envSource = 'default'
78let env: CacheEnv = {}
79let timer: { cancel: () => void } | undefined
80let lastKey = ''
81let toastedFor = 0
82let toastLevel = Infinity
83let isPaneOpen = false
84
85type Policy = { warnMs: number; compactAtTokens: number }
86
87function current(policy: Policy, now: number) {
88 const last = samples[samples.length - 1]
89 const prev = samples[samples.length - 2]
90 const disabled = last ? isCachingDisabled(last.model, env) : isCachingDisabled('', env)
91 const advice: Advice = advise(last, prev, { ttl, ...policy }, now, disabled)
92 const left = last ? remainingMs(last, ttl, now) : 0
93 return { last, advice, left }
94}
95
96const COLOR: Record<Advice['kind'], string | undefined> = {
97 warm: 'green',
98 soon: 'yellow',
99 expired: 'red',
100 miss: 'red',
101 off: undefined,
102 cold: undefined,
103 uncached: undefined,
104}
105
106function shortLine(policy: Policy, now: number): string {
107 const { last, advice, left } = current(policy, now)
108 if (!last || advice.kind === 'off') return `cache: ${advice.text}`
109 const clock = left > 0 ? ` · ${fmtClock(left)}` : ''
110 return `cache ${Math.round(hitRatio(last) * 100)}%${clock}`
111}
112
113// the promptCacheTtl setting, from the settings files that can carry it (local over project over user)
114async function readSetting($: EngineInterface): Promise<unknown> {
115 const home = await $.env.get('HOME').catch(() => undefined)
116 const cwd = await $.session.cwd().catch(() => undefined)
117 const files = [cwd && `${cwd}/.claude/settings.local.json`, cwd && `${cwd}/.claude/settings.json`, home && `${home}/.claude/settings.json`]
118 for (const file of files) {
119 if (!file) continue
120 try {
121 const value = JSON.parse(await $.fs.read(file)).promptCacheTtl
122 if (value === '5m' || value === '1h') return value
123 } catch {
124 // missing or unreadable: the next file
125 }
126 }
127 return undefined
128}
129
130export const register: Register = (on, options) => {
131 const policy: Policy = {
132 warnMs: positive(options.warnSeconds, 60) * 1000,
133 compactAtTokens: positive(options.compactAtTokens, 100_000),
134 }
135 const showBand = options.band !== false
136 const showStatus = options.status === true
137 const wantToast = options.toast !== false
138
139 on('session.start', async ($, e, next) => {
140 const r = await next(e)
141 samples = []
142 lastKey = ''
143 toastedFor = 0
144 const none = () => undefined
145 env = {
146 enable1h: await $.env.get('ENABLE_PROMPT_CACHING_1H').catch(none),
147 force5m: await $.env.get('FORCE_PROMPT_CACHING_5M').catch(none),
148 ttlVar: await $.env.get('CLAUDE_CODE_PROMPT_CACHE_TTL').catch(none),
149 disableAll: await $.env.get('DISABLE_PROMPT_CACHING').catch(none),
150 disableHaiku: await $.env.get('DISABLE_PROMPT_CACHING_HAIKU').catch(none),
151 disableSonnet: await $.env.get('DISABLE_PROMPT_CACHING_SONNET').catch(none),
152 disableOpus: await $.env.get('DISABLE_PROMPT_CACHING_OPUS').catch(none),
153 }
154 pinned = options.ttl === '5m' || options.ttl === '1h'
155 observed = undefined
156 setting = await readSetting($)
157 account = accountOf((await $.session.usage().catch(() => undefined))?.rateLimits ?? [])
158 const choice = decideTtl(options.ttl, env, setting, account)
159 baseTtl = choice.ttl
160 ttl = baseTtl
161 envSource = choice.source
162 ttlSource = envSource
163
164 await $.command
165 .register({
166 name: COMMAND,
167 description: 'Prompt-cache usage per turn and the time left before it lapses (stop closes)',
168 argumentHint: '[stop]',
169 immediate: true,
170 })
171 .catch(err => $.ui.log(`cache-meter: /${COMMAND} not registered: ${err}`))
172 $.ui.log(`cache-meter loaded: ${ttl} cache (${ttlSource}), /${COMMAND} opens the table`, { to: 'debug' })
173
174 timer?.cancel()
175 timer = $.clock.every(1000, () => {
176 const now = Date.now()
177 const { last, advice, left } = current(policy, now)
178 const key = `${advice.kind}|${advice.text}|${left > 0 ? fmtClock(left) : ''}`
179 if (key !== lastKey) {
180 lastKey = key
181 if (showStatus) $.ui.status(shortLine(policy, now))
182 $.ui.invalidate('ui.render')
183 }
184 if (wantToast && last && left > 0 && promptTokens(last) >= TOAST_MIN_TOKENS) {
185 if (toastedFor !== last.startedAt) {
186 toastedFor = last.startedAt
187 toastLevel = Infinity
188 }
189 // the first toast comes at warnSeconds, then 10, 3, 2 and 1 seconds; a late tick skips to the newest one
190 const secs = Math.ceil(left / 1000)
191 const mark = nextToastMark(secs, policy.warnMs / 1000, toastLevel)
192 if (mark !== undefined) {
193 toastLevel = mark
194 const tail = secs <= COUNTDOWN_MARKS[0] ? 'send a message now' : `send a message to keep ${fmtTokens(promptTokens(last))} tokens warm`
195 $.ui.toast(`cache expires in ${secs >= 60 ? fmtClock(left) : `${secs}s`}: ${tail}`)
196 }
197 }
198 })
199 return r
200 })
201
202 on('session.end', async ($, e, next) => {
203 // /clear starts a new conversation in the same process: its cache is a new one
204 if (e.reason === 'clear') {
205 samples = []
206 lastKey = ''
207 toastedFor = 0
208 observed = undefined
209 ttl = baseTtl
210 ttlSource = envSource
211 $.ui.invalidate('ui.render')
212 return next(e)
213 }
214 timer?.cancel()
215 timer = undefined
216 return next(e)
217 })
218
219 // each main-loop request: what the cache did with it
220 on('turn.step', async function* ($, e, next) {
221 if (e.agentId) return yield* next(e)
222 const startedAt = Date.now()
223 const r = yield* next(e)
224 if (r.usage) {
225 const sample: Sample = {
226 turnId: e.turnId,
227 index: e.index,
228 model: r.usage.model || e.model,
229 startedAt,
230 read: r.usage.cache_read_input_tokens,
231 write: r.usage.cache_creation_input_tokens,
232 fresh: r.usage.input_tokens,
233 output: r.usage.output_tokens,
234 }
235 samples.push(sample)
236 if (samples.length > KEEP) samples = samples.slice(-KEEP)
237 if (!pinned) {
238 // the account can change under a session: a subscription running out of plan usage moves to usage credits
239 account = accountOf((await $.session.usage().catch(() => undefined))?.rateLimits ?? [])
240 const choice = decideTtl(options.ttl, env, setting, account)
241 baseTtl = choice.ttl
242 envSource = choice.source
243 if (observed === undefined) {
244 ttl = baseTtl
245 ttlSource = envSource
246 }
247 const seen = observeTtl(samples[samples.length - 2], sample, observed)
248 if (seen !== observed) {
249 observed = seen
250 ttl = seen ?? baseTtl
251 ttlSource = `observed from request timing; ${envSource} said ${baseTtl}`
252 $.ui.log(`cache-meter: cache lifetime is ${ttl} (${ttlSource})`, { to: 'debug' })
253 }
254 }
255 lastKey = ''
256 if (showStatus) $.ui.status(shortLine(policy, Date.now()))
257 $.ui.invalidate('ui.render')
258 }
259 return r
260 })
261
262 on('command.run', { command: COMMAND }, async ($, e) => {
263 if (e.args.trim().toLowerCase() === 'stop') {
264 await $.ui.close({ id: PANE }).catch(() => undefined)
265 isPaneOpen = false
266 return { text: 'cache table closed' }
267 }
268 isPaneOpen = true
269 await $.ui.open({ id: PANE, title: 'cache', focus: true })
270 $.ui.invalidate('ui.render')
271 const { advice } = current(policy, Date.now())
272 return { text: `${ttl} cache (${ttlSource}) · ${advice.text} · /${COMMAND} stop closes` }
273 })
274
275 on('ui.close', async ($, e, next) => {
276 if (e.id !== PANE) return next(e)
277 isPaneOpen = false
278 return next(e)
279 })
280
281 on('ui.press', async ($, e, next) => {
282 if (e.plugin !== $.plugin.name || e.requestId !== PANE) return next(e)
283 if (e.element === 'close') await $.ui.close({ id: PANE }).catch(() => undefined)
284 return next(e)
285 })
286
287 on('ui.render', { component: 'AbovePrompt' }, async ($, e, next) => {
288 if (!showBand || e.props.hasSurvey || isPaneOpen) return next(e)
289 const { last, advice, left } = current(policy, Date.now())
290 if (!last && advice.kind !== 'off') return next(e)
291 const below = await next(e)
292 const { Box, Text } = $.ui.resolve(e)
293 const columns = e.props.bodyColumns
294
295 const line = (children: JSX.Element | JSX.Element[]) => (
296 <Box flexDirection="column">
297 {below}
298 <Box key="row" flexDirection="row" columnGap={1}>{children}</Box>
299 </Box>
300 )
301 if (!last) return line(<Text dimColor wrap="truncate-end">{fit(`cache: ${advice.text}`, columns)}</Text>)
302
303 const pct = Math.round(hitRatio(last) * 100)
304 const counting = advice.kind !== 'uncached' && advice.kind !== 'off'
305 const parts: BandPart[] = [
306 { key: 'cache', text: `cache ${pct}%` },
307 { key: 'read', text: `read ${fmtTokens(last.read)}` },
308 { key: 'wrote', text: `wrote ${fmtTokens(last.write)}` },
309 { key: 'new', text: `new ${fmtTokens(last.fresh)}` },
310 ...(counting ? [{ key: 'clock' as const, text: `${fmtClock(left)} / ${ttl}` }] : []),
311 { key: 'advice', text: advice.short },
312 ]
313 type Cell = { key: string; text: string; color?: string; bold?: boolean }
314 const cells = fitParts(parts, columns).flatMap((part, i): Cell[] => {
315 const lead: Cell[] = i > 0 ? [{ key: `${part.key}:dot`, text: '·' }] : []
316 if (part.key === 'cache') {
317 return [...lead, { key: 'cache', text: 'cache', color: 'cyan', bold: true }, { key: 'pct', text: `${pct}%`, color: ratioColor(pct), bold: true }]
318 }
319 if (part.key === 'clock') {
320 const color = left > 0 ? lifeColor(left, ttl, policy.warnMs) : 'red'
321 return [...lead, { key: 'clock', text: fmtClock(left), color, bold: true }, { key: 'ttl', text: `/ ${ttl}` }]
322 }
323 const color = part.key === 'read' ? 'green' : part.key === 'wrote' ? 'yellow' : part.key === 'new' ? 'cyan' : undefined
324 return [...lead, { key: part.key, text: part.text, color }]
325 })
326 return line(
327 cells.map((cell, i) => (
328 <Text key={cell.key} color={cell.color} bold={cell.bold} dimColor={cell.color === undefined} wrap={i === cells.length - 1 ? 'truncate-end' : undefined}>
329 {cell.text}
330 </Text>
331 )),
332 )
333 })
334
335 on('ui.render', { component: 'Pane' }, async ($, e, next) => {
336 if (e.requestId !== PANE) return next(e)
337 const { Box, Text, Button } = $.ui.resolve(e)
338 const width = Math.max(30, e.props.bodyColumns - 1)
339 // HTML collapses runs of spaces and trims a text's ends; a no-break space keeps them
340 const sp = (t: string) => (e.surface === 'terminal' ? t : t.replace(/ /g, ' '))
341 const now = Date.now()
342 const { last, advice, left } = current(policy, now)
343 const all = byTurn(samples)
344 const counting = !!last && advice.kind !== 'uncached' && advice.kind !== 'off'
345 // the countdown goes green, then yellow, then red as the cache runs out
346 const clockColor = counting ? lifeColor(left, ttl, policy.warnMs) : undefined
347 const stateColor = advice.kind === 'expired' || advice.kind === 'miss' ? 'red' : (clockColor ?? COLOR[advice.kind])
348 const hitColor = (pct: number) => (pct >= 80 ? 'green' : pct >= 40 ? 'yellow' : 'red')
349 // solid bars are filled Boxes, not block characters, so HTML draws no seams between cells
350 const solid = (key: string, parts: [number, string | undefined][]) => (
351 <Box key={key} flexDirection="row" height={1} flexShrink={0}>
352 {parts.map(([w, c], i) => (w > 0 ? <Box key={`${key}:${i}`} width={w} height={1} flexShrink={0} backgroundColor={c} /> : null))}
353 </Box>
354 )
355 const cell = (key: string, w: number, text: string, c?: string, bold = false) => (
356 <Box key={key} width={w} flexShrink={0} justifyContent="flex-end">
357 <Text color={c} bold={bold} dimColor={!c}>{sp(text)}</Text>
358 </Box>
359 )
360
361 const barW = Math.min(width, 40)
362 const life = lifeRatio(left, ttl)
363 const lifeFilled = Math.round(life * barW)
364 const [sr, sw, sn] = last ? segments(last.read, last.write, last.fresh, barW) : [0, 0, 0]
365 const rows = all.slice(-Math.max(3, (e.viewport?.rows ?? 24) - 16))
366 const icon = advice.kind === 'warm' ? '●' : advice.kind === 'soon' ? '▲' : advice.kind === 'expired' || advice.kind === 'miss' ? '✖' : '○'
367
368 return (
369 <Box flexDirection="column">
370 <Box key="title" flexDirection="row" columnGap={1}>
371 <Text bold color="cyan">{sp('⚡ PROMPT CACHE')}</Text>
372 <Text dimColor>{sp(`· ${ttl} lifetime (${ttlSource})`)}</Text>
373 </Box>
374
375 <Box key="clock" flexDirection="column" marginTop={1}>
376 <Text bold color={clockColor}>{sp(counting ? `⏱ ${left > 0 ? fmtClock(left) : '0:00'}` : '⏱ --:--')}</Text>
377 {counting ? (
378 <Box flexDirection="row" columnGap={1}>
379 {solid('life', [[lifeFilled, clockColor], [barW - lifeFilled, 'gray']])}
380 <Text dimColor>{sp(`${Math.round(life * 100)}%`)}</Text>
381 </Box>
382 ) : null}
383 </Box>
384
385 <Box key="advice" marginTop={1} flexDirection="column">
386 <Text bold color={stateColor}>{sp(`${icon} ${advice.text}`)}</Text>
387 {last ? <Text dimColor>{sp(fit(`${last.model} · prompt ${fmtTokens(promptTokens(last))} tokens`, width))}</Text> : null}
388 </Box>
389
390 {last ? (
391 <Box key="stack" flexDirection="column" marginTop={1}>
392 <Box flexDirection="row" columnGap={1}>
393 {solid('stack', [[sr, 'green'], [sw, 'yellow'], [sn, 'cyan']])}
394 <Text bold color={hitColor(Math.round(hitRatio(last) * 100))}>{sp(`${Math.round(hitRatio(last) * 100)}% hit`)}</Text>
395 </Box>
396 <Box flexDirection="row" columnGap={2}>
397 <Text color="green">{sp(`■ read ${fmtTokens(last.read)}`)}</Text>
398 <Text color="yellow">{sp(`■ wrote ${fmtTokens(last.write)}`)}</Text>
399 <Text color="cyan">{sp(`■ new ${fmtTokens(last.fresh)}`)}</Text>
400 </Box>
401 </Box>
402 ) : null}
403
404 <Box key="table" flexDirection="column" marginTop={1}>
405 <Box key="head" flexDirection="row" columnGap={1}>
406 {cell('h:turn', 4, 'turn', 'cyan', true)}
407 {cell('h:steps', 5, 'steps', 'cyan', true)}
408 {cell('h:read', 6, 'read', 'green', true)}
409 {cell('h:wrote', 6, 'wrote', 'yellow', true)}
410 {cell('h:new', 5, 'new', 'cyan', true)}
411 {cell('h:hit', 4, 'hit', 'magenta', true)}
412 </Box>
413 {rows.length === 0 ? <Text dimColor>{sp('no requests yet')}</Text> : null}
414 {rows.map((row, i) => {
415 const n = all.length - rows.length + i + 1
416 const pct = Math.round(rowRatio(row) * 100)
417 return (
418 <Box key={`t:${row.turnId}`} flexDirection="row" columnGap={1}>
419 {cell(`c:turn:${row.turnId}`, 4, String(n))}
420 {cell(`c:steps:${row.turnId}`, 5, String(row.steps))}
421 {cell(`c:read:${row.turnId}`, 6, fmtTokens(row.read), 'green')}
422 {cell(`c:wrote:${row.turnId}`, 6, fmtTokens(row.write), 'yellow')}
423 {cell(`c:new:${row.turnId}`, 5, fmtTokens(row.fresh), 'cyan')}
424 {cell(`c:hit:${row.turnId}`, 4, `${pct}%`, hitColor(pct), true)}
425 </Box>
426 )
427 })}
428 </Box>
429
430 <Box key="foot" marginTop={1} flexDirection="column">
431 <Button key="close" label="close" onPress={() => {}} />
432 <Box key="legend" marginTop={1} flexDirection="column">
433 <Text color="green">{sp('■ read: served by the cache')}</Text>
434 <Text color="yellow">{sp('■ wrote: new cache entry')}</Text>
435 <Text color="cyan">{sp('■ new: sent uncached')}</Text>
436 </Box>
437 </Box>
438 </Box>
439 )
440 })
441}
442hooks/cache.ts 339 lines1/**
2 * cache.ts — the pure half of cache-meter: no `$`, no engine.
3 *
4 * What it models, from Anthropic's prompt-caching documentation:
5 * - the cache lives 5 minutes by default, 1 hour when asked for; a read
6 * refreshes the entry at no extra cost, and the lifetime is measured from
7 * the START of the request that wrote or read it
8 * - a request's prompt is `input_tokens` (uncached remainder) +
9 * `cache_read_input_tokens` + `cache_creation_input_tokens`
10 * - writes cost 1.25x base input for 5m and 2x for 1h; reads about 0.1x
11 * (less on some models), so an expired cache on a large context is the
12 * expensive moment
13 * - a prefix change (model, effort/thinking settings, tool set, system
14 * prompt) makes the next request write instead of read
15 *
16 * Claude Code's own switches (read from the environment):
17 * ENABLE_PROMPT_CACHING_1H=1 ask for the 1-hour TTL
18 * FORCE_PROMPT_CACHING_5M=1 force the 5-minute TTL, beating the above
19 * DISABLE_PROMPT_CACHING=1 no caching; DISABLE_PROMPT_CACHING_{HAIKU,SONNET,OPUS}
20 * turn it off for that model family only
21 */
22
23export type Ttl = '5m' | '1h'
24
25export type CacheEnv = {
26 enable1h?: string
27 force5m?: string
28 /** CLAUDE_CODE_PROMPT_CACHE_TTL: "5m" or "1h" for the main conversation */
29 ttlVar?: string
30 disableAll?: string
31 disableHaiku?: string
32 disableSonnet?: string
33 disableOpus?: string
34}
35
36/** One main-loop request, as the API reported it. */
37export type Sample = {
38 turnId: string
39 index: number
40 model: string
41 /** ms since the epoch when the request started: the cache's lifetime is counted from here */
42 startedAt: number
43 read: number
44 write: number
45 fresh: number
46 output: number
47}
48
49export type AdviceKind = 'off' | 'cold' | 'uncached' | 'warm' | 'soon' | 'expired' | 'miss'
50
51export type Advice = {
52 kind: AdviceKind
53 /** the full sentence, for the /cache pane and toasts */
54 text: string
55 /** a word or two, for the compact band */
56 short: string
57}
58
59export type Policy = {
60 ttl: Ttl
61 warnMs: number
62 compactAtTokens: number
63}
64
65export const isOn = (v: string | undefined) => v === '1' || v?.toLowerCase() === 'true'
66
67/** What the account is billed as, as far as the mod can tell. */
68export type Account = 'subscription' | 'credits' | 'other'
69
70export type TtlChoice = { ttl: Ttl; source: string }
71
72const asTtl = (v: unknown): Ttl | undefined => (v === '5m' || v === '1h' ? v : undefined)
73
74/**
75 * Which lifetime Claude Code asks for on the main conversation, in the order
76 * its documentation gives (https://code.claude.com/docs/en/prompt-caching,
77 * "Choose the TTL yourself"), after the mod's own `ttl` option:
78 *
79 * FORCE_PROMPT_CACHING_5M, CLAUDE_CODE_PROMPT_CACHE_TTL, the promptCacheTtl
80 * setting, ENABLE_PROMPT_CACHING_1H, then the default of the account: one
81 * hour on a Claude subscription within its plan usage, five minutes on usage
82 * credits, an API key or a cloud provider.
83 */
84export function decideTtl(option: unknown, env: CacheEnv, setting?: unknown, account?: Account): TtlChoice {
85 const pinned = asTtl(option)
86 if (pinned) return { ttl: pinned, source: 'the ttl option' }
87 if (isOn(env.force5m)) return { ttl: '5m', source: 'FORCE_PROMPT_CACHING_5M' }
88 const fromVar = asTtl(env.ttlVar)
89 if (fromVar) return { ttl: fromVar, source: 'CLAUDE_CODE_PROMPT_CACHE_TTL' }
90 const fromSetting = asTtl(setting)
91 if (fromSetting) return { ttl: fromSetting, source: 'the promptCacheTtl setting' }
92 if (isOn(env.enable1h)) return { ttl: '1h', source: 'ENABLE_PROMPT_CACHING_1H' }
93 if (account === 'subscription') return { ttl: '1h', source: 'Claude subscription default' }
94 if (account === 'credits') return { ttl: '5m', source: 'usage credits default' }
95 return { ttl: '5m', source: 'default' }
96}
97
98export const resolveTtl = (option: unknown, env: CacheEnv, setting?: unknown, account?: Account): Ttl =>
99 decideTtl(option, env, setting, account).ttl
100
101/**
102 * The account, from the rate-limit windows the last response reported: a
103 * five-hour or seven-day window means a Claude subscription, and one that is
104 * full means the next requests draw on usage credits. No window (an API key,
105 * a cloud provider, or no response yet) says nothing.
106 */
107export function accountOf(windows: readonly { kind: string; percentUsed: number }[]): Account {
108 const plan = windows.filter(w => w.kind === 'five_hour' || w.kind === 'seven_day')
109 if (plan.length === 0) return 'other'
110 return plan.some(w => w.percentUsed >= 100) ? 'credits' : 'subscription'
111}
112
113export function ttlMs(ttl: Ttl): number {
114 return ttl === '1h' ? 3_600_000 : 300_000
115}
116
117/** Caching switched off for this model by the environment. */
118export function isCachingDisabled(model: string, env: CacheEnv): boolean {
119 if (isOn(env.disableAll)) return true
120 const name = model.toLowerCase()
121 if (name.includes('haiku')) return isOn(env.disableHaiku)
122 if (name.includes('sonnet')) return isOn(env.disableSonnet)
123 if (name.includes('opus')) return isOn(env.disableOpus)
124 return false
125}
126
127export const promptTokens = (s: Sample) => s.read + s.write + s.fresh
128
129/** Share of the prompt the cache served, 0 to 1; 0 for an empty prompt. */
130export function hitRatio(s: Sample): number {
131 const total = promptTokens(s)
132 return total === 0 ? 0 : s.read / total
133}
134
135/** When the cache entry the sample touched lapses, ms since the epoch. */
136export const expiresAt = (s: Sample, ttl: Ttl) => s.startedAt + ttlMs(ttl)
137
138/** Zero for a request that read and wrote nothing: it created or refreshed no entry, so there is nothing to count down. */
139export function remainingMs(s: Sample, ttl: Ttl, now: number): number {
140 if (s.read + s.write === 0) return 0
141 return Math.max(0, expiresAt(s, ttl) - now)
142}
143
144/**
145 * Why a request that should have read the cache wrote it instead; undefined
146 * when it did not miss. A prompt that shrank is a /compact or /clear, not a
147 * miss, and the first request of a session has nothing to read.
148 */
149export function missReason(prev: Sample | undefined, cur: Sample, ttl: Ttl): string | undefined {
150 if (!prev) return undefined
151 const before = promptTokens(prev)
152 if (before === 0 || promptTokens(cur) < before * 0.7) return undefined
153 if (cur.read >= before * 0.5 || cur.write === 0) return undefined
154 if (cur.model !== prev.model) return `model changed (${prev.model} to ${cur.model})`
155 if (cur.startedAt - prev.startedAt > ttlMs(ttl)) return `the ${ttl} cache had lapsed`
156 return 'the prompt prefix changed (effort, tools, system prompt or CLAUDE.md)'
157}
158
159export function advise(last: Sample | undefined, prev: Sample | undefined, policy: Policy, now: number, disabled: boolean): Advice {
160 if (disabled) return { kind: 'off', short: 'off', text: 'prompt caching is off for this model (DISABLE_PROMPT_CACHING*)' }
161 if (!last) return { kind: 'cold', short: 'cold', text: 'no request yet: the first one writes the cache' }
162 if (last.read + last.write === 0) {
163 return { kind: 'uncached', short: 'uncached', text: 'this request was not cached (prompt under the model minimum, or caching off)' }
164 }
165 const miss = missReason(prev, last, policy.ttl)
166 const left = remainingMs(last, policy.ttl, now)
167 const size = promptTokens(last)
168 if (left <= 0) {
169 const big = size >= policy.compactAtTokens
170 return {
171 kind: 'expired',
172 short: big ? 'expired: /compact' : 'expired',
173 text: big
174 ? `expired: the next message rewrites ${fmtTokens(size)} tokens. /compact first, or /clear if the task is done`
175 : `expired: only ${fmtTokens(size)} tokens to rebuild, just keep going`,
176 }
177 }
178 if (left <= policy.warnMs) {
179 return { kind: 'soon', short: 'expires soon', text: 'expires soon: any message refreshes it for free' }
180 }
181 if (miss) return { kind: 'miss', short: 'missed', text: `cache missed: ${miss}` }
182 return { kind: 'warm', short: 'warm', text: 'warm: keep going' }
183}
184
185export function fmtTokens(n: number): string {
186 if (n < 1000) return String(n)
187 if (n < 100_000) return `${(n / 1000).toFixed(1).replace(/\.0$/, '')}k`
188 if (n < 1_000_000) return `${Math.round(n / 1000)}k`
189 return `${(n / 1_000_000).toFixed(1).replace(/\.0$/, '')}M`
190}
191
192/** m:ss, or h:mm:ss from an hour up. */
193export function fmtClock(ms: number): string {
194 const total = Math.max(0, Math.ceil(ms / 1000))
195 const h = Math.floor(total / 3600)
196 const m = Math.floor((total % 3600) / 60)
197 const s = total % 60
198 const pad = (n: number) => String(n).padStart(2, '0')
199 return h > 0 ? `${h}:${pad(m)}:${pad(s)}` : `${m}:${pad(s)}`
200}
201
202export function bar(ratio: number, width: number): string {
203 const filled = Math.round(Math.min(1, Math.max(0, ratio)) * width)
204 return '█'.repeat(filled) + '░'.repeat(width - filled)
205}
206
207export type TurnRow = {
208 turnId: string
209 steps: number
210 read: number
211 write: number
212 fresh: number
213 output: number
214}
215
216/** Samples grouped by turn, oldest first, each turn's requests summed. */
217export function byTurn(samples: readonly Sample[]): TurnRow[] {
218 const rows: TurnRow[] = []
219 for (const s of samples) {
220 let row = rows[rows.length - 1]
221 if (!row || row.turnId !== s.turnId) {
222 row = { turnId: s.turnId, steps: 0, read: 0, write: 0, fresh: 0, output: 0 }
223 rows.push(row)
224 }
225 row.steps += 1
226 row.read += s.read
227 row.write += s.write
228 row.fresh += s.fresh
229 row.output += s.output
230 }
231 return rows
232}
233
234export const rowRatio = (r: TurnRow) => {
235 const total = r.read + r.write + r.fresh
236 return total === 0 ? 0 : r.read / total
237}
238
239export function fit(text: string, width: number): string {
240 return text.length <= width ? text : `${text.slice(0, Math.max(0, width - 1))}…`
241}
242
243/** Colour of the hit percentage: green from 90%, yellow from 50%, red below. */
244export function ratioColor(pct: number): 'green' | 'yellow' | 'red' {
245 return pct >= 90 ? 'green' : pct >= 50 ? 'yellow' : 'red'
246}
247
248export type BandPart = { key: 'cache' | 'read' | 'wrote' | 'new' | 'clock' | 'advice'; text: string }
249
250/** The widest prefix of the parts that fits `columns` joined by ' · '; advice goes first, then new, then wrote. */
251export function fitParts(parts: readonly BandPart[], columns: number): BandPart[] {
252 const width = (list: readonly BandPart[]) => list.reduce((n, p) => n + p.text.length, 0) + 3 * Math.max(0, list.length - 1)
253 let kept = [...parts]
254 for (const drop of ['advice', 'new', 'wrote'] as const) {
255 if (width(kept) <= columns) break
256 kept = kept.filter(p => p.key !== drop)
257 }
258 return kept
259}
260
261export function positive(v: unknown, fallback: number): number {
262 return typeof v === 'number' && Number.isFinite(v) && v > 0 ? v : fallback
263}
264
265/** Share of the cache lifetime left, 0 to 1. */
266export function lifeRatio(leftMs: number, ttl: Ttl): number {
267 return Math.min(1, Math.max(0, leftMs / ttlMs(ttl)))
268}
269
270/**
271 * Widths of the three stacked-bar segments (read, wrote, new) over `width`
272 * cells: proportional, each non-empty part at least one cell, summing to width.
273 */
274export function segments(read: number, write: number, fresh: number, width: number): [number, number, number] {
275 const total = read + write + fresh
276 if (total === 0 || width <= 0) return [0, 0, 0]
277 const parts = [read, write, fresh]
278 const cells = parts.map(p => (p > 0 ? Math.max(1, Math.round((p / total) * width)) : 0))
279 let over = cells.reduce((a, b) => a + b, 0) - width
280 while (over !== 0) {
281 const i = over > 0 ? cells.indexOf(Math.max(...cells)) : parts.indexOf(Math.max(...parts))
282 cells[i] = (cells[i] ?? 0) + (over > 0 ? -1 : 1)
283 over += over > 0 ? -1 : 1
284 }
285 return [cells[0] ?? 0, cells[1] ?? 0, cells[2] ?? 0]
286}
287
288/** Seconds left at which a toast counts down after the one at the warning threshold. */
289export const COUNTDOWN_MARKS = [10, 3, 2, 1] as const
290
291/**
292 * The toast mark to fire now, or undefined. `level` is the mark last fired for
293 * this cache entry (Infinity before any); a late tick skips straight to the
294 * newest mark crossed, so a stalled clock never replays old ones.
295 */
296export function nextToastMark(secsLeft: number, warnSecs: number, level: number): number | undefined {
297 const marks = [warnSecs, ...COUNTDOWN_MARKS].filter(m => m <= warnSecs)
298 const due = marks.filter(m => secsLeft <= m && m < level)
299 return due.length ? Math.min(...due) : undefined
300}
301
302export type LifeColor = 'green' | 'yellow' | 'red'
303
304/** Countdown colour: green while there is plenty, yellow below 40% of the lifetime, red from the warning threshold down. */
305export function lifeColor(leftMs: number, ttl: Ttl, warnMs: number): LifeColor {
306 if (leftMs <= warnMs) return 'red'
307 return leftMs / ttlMs(ttl) <= 0.4 ? 'yellow' : 'green'
308}
309
310// requests are timed from their start, so a little slack keeps a hit that
311// landed just inside the lifetime from reading as proof of the longer one
312const SLACK_MS = 10_000
313
314/**
315 * What the traffic says about the cache lifetime, given the request before and
316 * `known`, what earlier requests already showed.
317 *
318 * - a hit (the cache served at least half of the previous prompt) more than
319 * 5 minutes after the previous request began proves the 1-hour lifetime,
320 * and nothing later undoes it: a miss afterwards is more likely a changed
321 * prefix than a lapse
322 * - a miss with the same model and a prompt that did not shrink, 5 minutes to
323 * an hour after the previous request, says the entry lapsed: 5 minutes
324 * (weaker: a changed prefix looks the same, so a later hit overrules it)
325 *
326 * Needed because the API names the TTL of a write (`cache_creation.ephemeral_*`)
327 * but Claude Code's mod API passes on only the four token counts.
328 */
329export function observeTtl(prev: Sample | undefined, cur: Sample, known: Ttl | undefined): Ttl | undefined {
330 if (!prev || prev.read + prev.write === 0 || cur.model !== prev.model) return known
331 const gap = cur.startedAt - prev.startedAt
332 const before = promptTokens(prev)
333 if (gap <= ttlMs('5m') + SLACK_MS) return known
334 if (cur.read >= before * 0.5) return '1h'
335 if (known === '1h') return known
336 const lapsed = cur.write > 0 && promptTokens(cur) >= before * 0.7 && gap < ttlMs('1h') + SLACK_MS
337 return lapsed ? '5m' : known
338}
339