Status-line bar draining over the prompt-cache window; compacts once just before it expires

A Claude Code mod (function-hook plugin) that tracks the prompt-cache window of the main conversation.
🟢 ▰▰▰▰▰▰▰▰▱▱ 47:12 beside the model name. The dot goes 🟢 above half the window, 🟡 down to 15%, 🔴 below, ⚪ cold once it has expired. The window restarts from when a main-thread request is sent, once the model has answered it (an interrupted answer counts); subagent and compaction requests run on their own cache prefix and are ignored./compact. Each cache generation (named by its request time) gets one shot: it's spent before the call, a skip or a mid-turn refusal leaves it spent, and only a new request that reaches the model starts a generation with a fresh shot. It skips contexts under 40k tokens and calls the compaction off if a request re-warmed the cache while it was getting ready. 📦 marks a generation where a compaction actually stood.The window is assumed to be the 1-hour TTL (TTL_MS). Sessions on the 5-minute TTL (usage overage, some API-key setups) will see the wrong countdown. The thresholds are constants at the top of hooks/register.tsx.
Install: claude plugin install cache-timer@kosta-plugins. Check: claude plugin validate plugins/cache-timer.
hooks/register.tsx 138 lines1import { atom, read, update } from 'claude-code'
2import type { EngineInterface, Register } from 'claude-code'
3
4// This session's requests use the 1-hour prompt-cache TTL; it drops to 5m in usage overage.
5const TTL_MS = 60 * 60 * 1000
6// Warn two minutes ahead of the compaction, so there's time to send something instead.
7const WARN_MS = 12 * 60 * 1000
8// Compact while the cache is still warm, so the summarizer's request reads the
9// prefix from cache and the next prompt re-caches only the short summary.
10const COMPACT_MS = 10 * 60 * 1000
11// Below this the context is cheap to re-cache anyway; not worth losing detail.
12const MIN_COMPACT_TOKENS = 40_000
13const BAR = 10
14
15// Each flag records the generation (the lastHit it was set for) rather than a boolean, so a
16// stale tick writing for an old generation can never block or mark a newer one.
17const lastHit = atom({ plugin: 'cache-timer', key: 'lastHit' } as const, 0)
18const now = atom({ plugin: 'cache-timer', key: 'now' } as const, 0)
19const warnedFor = atom({ plugin: 'cache-timer', key: 'warnedFor' } as const, 0)
20const attemptedFor = atom({ plugin: 'cache-timer', key: 'attemptedFor' } as const, 0)
21const compactedFor = atom({ plugin: 'cache-timer', key: 'compactedFor' } as const, 0)
22
23// Set synchronously, so two ticks can't both pass the state read before either writes it.
24let compacting = false
25
26// The status line is plain text, so the dot carries the color.
27function dotFor(fraction: number) {
28 if (fraction > 0.5) return '🟢'
29 if (fraction > 0.15) return '🟡'
30 return '🔴'
31}
32
33async function show($: EngineInterface) {
34 const hit = await read($, lastHit)
35 if (hit === 0) return $.ui.status(undefined)
36 const isCompacted = (await read($, compactedFor)) === hit
37 const left = Math.max(0, TTL_MS - ((await read($, now)) - hit))
38 if (left === 0) return $.ui.status(`⚪ ${'▱'.repeat(BAR)} ${isCompacted ? 'cold · compacted' : 'cold'}`)
39
40 const fraction = left / TTL_MS
41 const filled = Math.max(1, Math.round(fraction * BAR))
42 const m = Math.floor(left / 60000)
43 const s = Math.floor((left % 60000) / 1000)
44 $.ui.status(`${dotFor(fraction)} ${'▰'.repeat(filled)}${'▱'.repeat(BAR - filled)} ${m}:${String(s).padStart(2, '0')}${isCompacted ? ' 📦' : ''}`)
45}
46
47async function isWorthCompacting($: EngineInterface) {
48 const { context } = await $.session.usage()
49 return (context.tokens ?? 0) >= MIN_COMPACT_TOKENS
50}
51
52// `hit` is the generation the tick saw; a request reaching the model meanwhile moves lastHit and calls it off.
53async function autoCompact($: EngineInterface, hit: number) {
54 if (compacting) return
55 compacting = true
56 try {
57 await update($, attemptedFor, () => hit)
58 if (!(await isWorthCompacting($))) return
59 // Recheck right before compacting: a request that just reached the model re-warmed the cache.
60 if ((await read($, lastHit)) !== hit) return
61
62 const { skip } = await $.session.compact({
63 instructions: 'Session went idle; keep open tasks, decisions, file paths and the current plan.',
64 })
65 if (skip) {
66 $.ui.toast(`Auto-compact skipped: ${skip}`)
67 } else {
68 await update($, compactedFor, () => hit)
69 $.ui.toast('Cache about to expire: compacted the conversation once')
70 }
71 } catch {
72 // Rejects while a turn runs. Keep this generation's shot spent: that turn's request starts a
73 // new generation with a fresh shot. Giving it back here would retry every tick.
74 } finally {
75 compacting = false
76 }
77}
78
79export const register: Register = on => {
80 on('session.start', async ($, e, next) => {
81 const r = await next(e)
82 $.clock.every(1000, async () => {
83 const t = await $.clock.now()
84 await update($, now, () => t)
85 const hit = await read($, lastHit)
86 const left = TTL_MS - (t - hit)
87
88 // Only while the two minutes are still ahead: after a sleep that jumps past them, warning now
89 // would announce a compaction that is already starting.
90 if (hit > 0 && left > COMPACT_MS && left <= WARN_MS && (await read($, warnedFor)) !== hit) {
91 await update($, warnedFor, () => hit)
92 if (await isWorthCompacting($)) {
93 $.ui.toast('Prompt cache: auto-compacting in 2m unless you send something')
94 }
95 }
96 if (hit > 0 && left > 0 && left <= COMPACT_MS && (await read($, attemptedFor)) !== hit) {
97 await autoCompact($, hit)
98 }
99 await show($)
100 })
101 return r
102 })
103
104 on('turn.step', async function* ($, e, next) {
105 // The cache is read and written when the request is sent, so the window starts here,
106 // not when a long response finishes streaming.
107 let sentAt = await $.clock.now()
108 // A content chunk means the model answered, so the request touched the cache, even if the
109 // stream is then interrupted or fails. Engine chunks don't count: a retry marker can come
110 // before any response.
111 let reachedModel = false
112 const stream = next(e)
113 try {
114 // `for await` closes the stream if this hook is cancelled mid-stream, as `yield*` would,
115 // so the request's own cleanup still runs; `stream.result` hands the step's result up.
116 for await (const chunk of stream) {
117 if (chunk.kind !== 'engine') {
118 reachedModel = true
119 } else if (!reachedModel) {
120 // Before any answer an engine item may mark a retry, which sends the request again;
121 // restarting here keeps the window from starting early by the retry's backoff.
122 sentAt = await $.clock.now()
123 }
124 yield chunk
125 }
126 return await stream.result
127 } finally {
128 // Subagents and the compaction fork run on their own cache prefix; only main-thread requests count.
129 if (e.agentId === undefined && reachedModel) {
130 await update($, lastHit, () => sentAt)
131 const t = await $.clock.now()
132 await update($, now, () => t)
133 await show($)
134 }
135 }
136 })
137}
138types/index.d.ts 19 lines1/** Milliseconds since the epoch. */
2export type CacheClock = number
3
4declare module 'claude-code' {
5 interface PluginState {
6 'cache-timer': {
7 /** When the last main-thread request that reached the model was sent; it names the cache generation. */
8 lastHit: CacheClock
9 now: CacheClock
10 /** The generation the 12-minute warning was shown for. */
11 warnedFor: CacheClock
12 /** The generation whose one auto-compact shot is spent. */
13 attemptedFor: CacheClock
14 /** The generation a compaction actually stood for (drives 📦). */
15 compactedFor: CacheClock
16 }
17 }
18}
19