Shows how much of each request the prompt cache served, and tells you when the cache was rebuilt and the likely reason.

A mod for Claude Code that shows how much of each request the prompt cache served, and tells you the moment the cache is paid for again, with the likely reason.
Claude Code sends the whole conversation with every request. The prompt cache is what keeps that cheap and fast: text the API has seen recently is read from the cache instead of being processed again. When the cache is lost, the next request pays for the whole conversation at once. That usually happens silently. Cache Watch makes it visible.
A line above the prompt after each turn:
cache 94% read from cache 182k in · 2.1k out · 6 requests details
The percentage is the share of the turn's input tokens that came from the cache. It is green from 70% and a warning colour under 30%.
A notice when the cache is rebuilt:
Cache rebuilt: 61k written again, because 12 minutes passed since the last request, long enough for the cache to lapse.
/cache opens a pane with the session's totals, a bar for each of the last ten turns, the subagents' numbers counted apart, and every rebuild with its turn and cause.
Each request reports four counts: input tokens read from the cache, written to it, sent uncached, and output tokens. The mod compares each request of the main conversation with the one before:
When both hold, it is a rebuild. The cause is the first of these that fits:
| Order | Cause | How it is known | | :- | :- | :- | | 1 | The model changed | The two requests were answered by different models | | 2 | The conversation was compacted | A compaction ran between them | | 3 | The cache lapsed | More than 5 minutes passed between them | | 4 | The start of the prompt changed | None of the above, so something early changed: the system prompt or the list of tools |
/plugin install prompt-cache-watch --marketplace ramankrishna/prompt-cache-watch
Answer y to add the marketplace, then pick a scope. Needs Claude Code v2.1.287 or later.
Like every mod, it runs with the same access to your machine as Claude Code itself. The full statement is in PRIVACY.md.
/clear and /resume start the numbers over, and nothing is kept between sessions.claude plugin validate .
claude plugin test .
claude --plugin-dir .
MIT
hooks/register.tsx 209 lines1// Cache Watch: how much of each request the prompt cache served, and a notice
2// the moment the cache is paid for again.
3//
4// It only reads the token counts the API reports for each request. It changes
5// nothing about the request, asks no model anything, and leaves the machine alone.
6
7import { atom, read, update } from 'claude-code'
8import type { EngineInterface, Register } from 'claude-code'
9
10import type { Step } from '../types'
11import {
12 addAgentStep,
13 addStep,
14 bar,
15 EMPTY,
16 endTurn,
17 hitOf,
18 percent,
19 rebuildLine,
20 sentOf,
21 shownTurn,
22 startTurn,
23 tokens,
24 toneOf,
25 totalsLine,
26} from './cache'
27
28const PANE = 'cache'
29
30const watch = atom({ plugin: 'prompt-cache-watch', key: 'watch' } as const, EMPTY)
31
32const openPane = ($: EngineInterface) =>
33 $.ui.open({ id: PANE, title: 'Cache Watch', focus: true, closeOnEscape: true })
34
35export const register: Register = on => {
36 on('session.start', async ($, e, next) => {
37 await $.command.register({
38 name: 'cache',
39 description: 'Show how much of each request the prompt cache served, and when it was rebuilt',
40 })
41
42 return next(e)
43 })
44
45 on('command.run', { command: 'cache' }, async $ => {
46 await openPane($)
47
48 return {}
49 })
50
51 on('turn.start', async ($, e, next) => {
52 await update($, watch, was => startTurn(was))
53
54 return next(e)
55 })
56
57 // One request to the model: forward it untouched, then count what it cost.
58 on('turn.step', async function* ($, e, next) {
59 const startedAt = await $.clock.now()
60 const result = yield* next(e)
61
62 if (result.usage === null) return result
63
64 const step: Step = {
65 at: await $.clock.now(),
66 model: result.usage.model,
67 input: result.usage.input_tokens,
68 output: result.usage.output_tokens,
69 read: result.usage.cache_read_input_tokens,
70 write: result.usage.cache_creation_input_tokens,
71 }
72
73 if (e.agentId !== undefined) {
74 await update($, watch, was => addAgentStep(was, step))
75
76 return result
77 }
78
79 const { rebuild } = addStep(await read($, watch), step, startedAt)
80
81 await update($, watch, was => addStep(was, step, startedAt).watch)
82
83 if (rebuild !== null) $.ui.toast(`Cache rebuilt: ${rebuildLine(rebuild)}.`, { timeoutMs: 8000 })
84
85 return result
86 })
87
88 on('turn.complete', async ($, e, next) => {
89 if (e.agentId === undefined) await update($, watch, was => endTurn(was))
90
91 return next(e)
92 })
93
94 // A compaction rewrites the conversation, so the next request starts a new prefix.
95 on('session.compact', async ($, e, next) => {
96 const out = await next(e)
97
98 if (e.agentId === undefined && e.trigger !== 'precompute' && out.skip === undefined) {
99 await update($, watch, was => ({ ...was, hasCompacted: true }))
100 }
101
102 return out
103 })
104
105 // -------------------------------------------------------------- drawing
106
107 on('ui.render', { component: 'AbovePrompt' }, async ($, e, next) => {
108 const now = await read($, watch)
109 const row = shownTurn(now)
110
111 if (e.props.hasSurvey || row === null) return next(e)
112
113 const { Box, Button, Text } = $.ui.resolve(e)
114 const theirs = await next(e)
115 const share = hitOf(row)
116 const tone = toneOf(share)
117 const head = `${percent(share)} read from cache`
118
119 return (
120 <Box flexDirection="column">
121 {theirs}
122 <Box flexDirection="row" columnGap={2}>
123 <Text inverse> cache </Text>
124 {tone === null ? <Text>{head}</Text> : <Text color={tone}>{head}</Text>}
125 <Text dimColor>{totalsLine(row)}</Text>
126 {row.rebuilds > 0 && (
127 <Text color="warning">
128 rebuilt {row.rebuilds === 1 ? 'once' : `${row.rebuilds} times`} this turn
129 </Text>
130 )}
131 <Button key="open" label="details" hotkey="c" plain onPress={() => openPane($)} />
132 </Box>
133 </Box>
134 )
135 })
136
137 on('ui.render', { component: 'Pane', requestId: PANE }, async ($, e) => {
138 const { Box, Text } = $.ui.resolve(e)
139 const now = await read($, watch)
140 const rows = [...now.turns, ...(now.current !== null && now.current.requests > 0 ? [now.current] : [])]
141 const share = hitOf(now.main)
142
143 return (
144 <Box flexDirection="column" rowGap={1}>
145 {now.main.requests === 0 ? (
146 <Box flexDirection="column">
147 <Text>No requests yet this session.</Text>
148 <Text dimColor>Numbers appear after Claude's first reply.</Text>
149 </Box>
150 ) : (
151 <Box flexDirection="column">
152 <Text>
153 This session: {percent(share)} of {tokens(sentOf(now.main))} input tokens read from cache
154 </Text>
155 <Text dimColor>
156 {tokens(now.main.read)} read {'·'} {tokens(now.main.write)} written {'·'} {tokens(now.main.input)}{' '}
157 uncached {'·'} {tokens(now.main.output)} out {'·'} {now.main.requests} requests
158 </Text>
159 {now.agents.requests > 0 && (
160 <Text dimColor>
161 Subagents, counted apart: {percent(hitOf(now.agents))} of {tokens(sentOf(now.agents))} read from cache{' '}
162 {'·'} {now.agents.requests} requests
163 </Text>
164 )}
165 </Box>
166 )}
167 {rows.length > 0 && (
168 <Box flexDirection="column">
169 <Text dimColor>Turn by turn, newest first. The bar is the share read from cache.</Text>
170 {rows
171 .slice(-10)
172 .reverse()
173 .map(row => (
174 <Box flexDirection="row" columnGap={2}>
175 <Box width={4} flexShrink={0}>
176 <Text dimColor>#{row.n}</Text>
177 </Box>
178 <Text color={toneOf(hitOf(row)) ?? 'text'}>{bar(hitOf(row))}</Text>
179 <Box width={4} flexShrink={0}>
180 <Text>{percent(hitOf(row))}</Text>
181 </Box>
182 <Text dimColor>{totalsLine(row)}</Text>
183 {row.rebuilds > 0 && <Text color="warning">rebuilt</Text>}
184 </Box>
185 ))}
186 </Box>
187 )}
188 <Box flexDirection="column">
189 {now.rebuilds.length === 0 ? (
190 <Text dimColor>No rebuilds this session.</Text>
191 ) : (
192 <Text color="warning">
193 {now.rebuilds.length} {now.rebuilds.length === 1 ? 'rebuild' : 'rebuilds'} this session
194 </Text>
195 )}
196 {now.rebuilds
197 .slice(-6)
198 .reverse()
199 .map(rebuild => (
200 <Text dimColor>
201 Turn #{rebuild.turn}: {rebuildLine(rebuild)}
202 </Text>
203 ))}
204 </Box>
205 </Box>
206 )
207 })
208}
209hooks/cache.ts 156 lines1// How requests are counted and a cache rebuild is told, with no engine in it:
2// every function here is pure, so the tests and the hooks agree.
3
4import type { Rebuild, Step, Totals, TurnRow, Watch } from '../types'
5
6/** A cached prefix smaller than this is too small to call a rebuild over. */
7export const MIN_PREFIX = 2048
8
9/** A request that reads less than this share of the last prefix rebuilt it. */
10export const KEPT_SHARE = 0.5
11
12/** The shortest time the cache is kept between requests, in milliseconds. Some plans keep it longer. */
13export const SHORTEST_LIFETIME = 5 * 60_000
14
15const NONE: Totals = { requests: 0, input: 0, output: 0, read: 0, write: 0 }
16
17export const EMPTY: Watch = {
18 last: null,
19 hasCompacted: false,
20 turnCount: 0,
21 current: null,
22 turns: [],
23 rebuilds: [],
24 main: NONE,
25 agents: NONE,
26}
27
28// --------------------------------------------------------------- counting
29
30/** Every input token a request sent: fresh, read from the cache, and written to it. */
31export const sentOf = (totals: Pick<Totals, 'input' | 'read' | 'write'>): number =>
32 totals.input + totals.read + totals.write
33
34/** The share of a request's input the cache served, or null when it sent nothing. */
35export function hitOf(totals: Pick<Totals, 'input' | 'read' | 'write'>): number | null {
36 const sent = sentOf(totals)
37
38 return sent === 0 ? null : totals.read / sent
39}
40
41const plus = (totals: Totals, step: Step): Totals => ({
42 requests: totals.requests + 1,
43 input: totals.input + step.input,
44 output: totals.output + step.output,
45 read: totals.read + step.read,
46 write: totals.write + step.write,
47})
48
49/**
50 * Tells a rebuild: the last request left a prefix in the cache, and this one
51 * read less than half of it back. The cause is the first of these that fits:
52 * the model changed, the conversation was compacted, the cache had time to
53 * lapse, or else the start of the prompt changed.
54 */
55export function detect(
56 last: Step,
57 now: Step,
58 startedAt: number,
59 hasCompacted: boolean,
60 turn: number,
61): Rebuild | null {
62 const prefix = last.read + last.write
63
64 if (prefix < MIN_PREFIX || now.read >= prefix * KEPT_SHARE) return null
65
66 const idle = startedAt - last.at
67 const cause =
68 last.model !== now.model
69 ? `the model changed from ${last.model} to ${now.model}`
70 : hasCompacted
71 ? 'the conversation was compacted'
72 : idle > SHORTEST_LIFETIME
73 ? `${Math.round(idle / 60_000)} minutes passed since the last request, long enough for the cache to lapse`
74 : 'something near the start of the prompt changed, such as the system prompt or the list of tools'
75
76 return { at: now.at, turn, lost: prefix - now.read, rewritten: now.write, cause }
77}
78
79/** Begins a turn's row; a turn left open by an interrupt is closed first. */
80export function startTurn(watch: Watch): Watch {
81 const closed = endTurn(watch)
82 const n = closed.turnCount + 1
83
84 return { ...closed, turnCount: n, current: { n, ...NONE, rebuilds: 0 } }
85}
86
87export function endTurn(watch: Watch): Watch {
88 if (watch.current === null) return watch
89
90 return {
91 ...watch,
92 current: null,
93 turns: watch.current.requests === 0 ? watch.turns : [...watch.turns, watch.current].slice(-50),
94 }
95}
96
97/** Counts one main-loop request, and notes a rebuild when it was one. */
98export function addStep(watch: Watch, step: Step, startedAt: number): { watch: Watch; rebuild: Rebuild | null } {
99 const open = watch.current === null ? startTurn(watch) : watch
100 const row = open.current as TurnRow
101 const rebuild = open.last === null ? null : detect(open.last, step, startedAt, open.hasCompacted, row.n)
102
103 return {
104 rebuild,
105 watch: {
106 ...open,
107 last: step,
108 hasCompacted: false,
109 current: { ...row, ...plus(row, step), rebuilds: row.rebuilds + (rebuild === null ? 0 : 1) },
110 main: plus(open.main, step),
111 rebuilds: rebuild === null ? open.rebuilds : [...open.rebuilds, rebuild].slice(-30),
112 },
113 }
114}
115
116/** Counts a subagent's request. A subagent has a cache of its own, so no rebuild is told. */
117export const addAgentStep = (watch: Watch, step: Step): Watch => ({ ...watch, agents: plus(watch.agents, step) })
118
119// ------------------------------------------------------------------ words
120
121/** A token count at a glance: 950, 1.2k, 124k, 1.3M. */
122export function tokens(count: number): string {
123 if (count < 1000) return String(count)
124 if (count < 10_000) return `${(count / 1000).toFixed(1)}k`
125 if (count < 1_000_000) return `${Math.round(count / 1000)}k`
126
127 return `${(count / 1_000_000).toFixed(1)}M`
128}
129
130export const percent = (share: number | null): string => (share === null ? 'n/a' : `${Math.round(share * 100)}%`)
131
132/** A ten-cell bar for a share between 0 and 1. */
133export function bar(share: number | null): string {
134 const filled = share === null ? 0 : Math.round(share * 10)
135
136 return '▇'.repeat(filled) + '▁'.repeat(10 - filled)
137}
138
139const plural = (count: number, word: string): string => `${count} ${word}${count === 1 ? '' : 's'}`
140
141export const totalsLine = (totals: Totals): string =>
142 `${tokens(sentOf(totals))} in · ${tokens(totals.output)} out · ${plural(totals.requests, 'request')}`
143
144export type Tone = 'success' | 'warning' | null
145
146/** How a cache share reads at a glance: most of it served, or most of it paid for again. */
147export const toneOf = (share: number | null): Tone =>
148 share === null ? null : share >= 0.7 ? 'success' : share < 0.3 ? 'warning' : null
149
150/** The turn the band speaks for: the one running, or the last one finished. */
151export const shownTurn = (watch: Watch): TurnRow | null =>
152 watch.current !== null && watch.current.requests > 0 ? watch.current : (watch.turns.at(-1) ?? null)
153
154export const rebuildLine = (rebuild: Rebuild): string =>
155 `${tokens(rebuild.rewritten)} written again, because ${rebuild.cause}`
156types/index.d.ts 54 lines1/** One model request, as the API counted it. */
2export type Step = {
3 /** When the response ended, in milliseconds. */
4 at: number
5 model: string
6 /** Input tokens neither read from nor written to the cache. */
7 input: number
8 output: number
9 /** Input tokens the cache served. */
10 read: number
11 /** Input tokens written to the cache. */
12 write: number
13}
14
15export type Totals = { requests: number; input: number; output: number; read: number; write: number }
16
17/** One turn's requests, summed. */
18export type TurnRow = Totals & {
19 /** The turn's number in the session, from 1. */
20 n: number
21 rebuilds: number
22}
23
24/** One time the cached prefix was paid for again. */
25export type Rebuild = {
26 at: number
27 turn: number
28 /** Cached tokens the request did not get back. */
29 lost: number
30 /** Tokens the request wrote to the cache in their place. */
31 rewritten: number
32 /** The likeliest reason, in words. */
33 cause: string
34}
35
36export type Watch = {
37 /** The main conversation's last request, or null before the first. */
38 last: Step | null
39 /** True from a compaction until the next request. */
40 hasCompacted: boolean
41 turnCount: number
42 current: TurnRow | null
43 turns: TurnRow[]
44 rebuilds: Rebuild[]
45 main: Totals
46 agents: Totals
47}
48
49declare module 'claude-code' {
50 interface PluginState {
51 'prompt-cache-watch': { watch: Watch }
52 }
53}
54