Context window weather, tok/s, prompt-cache hit rate and TTL, and cache auto-warm, in the band above the prompt

Mods for Claude Code, packaged as a plugin marketplace.
In Claude Code:
/plugin marketplace add mrzzmrzz/claude-code-mods
/plugin install omni-token@claude-code-mods
A live forecast of your context window, shown in the band above the prompt and updated after every turn:
Cloudy 67% 134.4k / 200k ▂▃▄▅▆█ ▲ +98.3k last turn 95 tok/s cache 96% 58:12
| Fill | Forecast (color) |
|---|---|
| < 25% | Clear (yellow) |
| 25–49% | Cloudy (cyan) |
| 50–74% | Showers (blue) |
| 75–89% | Storm (magenta) |
| ≥ 90% | Compact soon (red) |
tok/s is the last turn's output tokens (thinking included) over the time from each request's start to its response's end.cache 96% is the last turn's prompt-cache hit rate: cache reads over all input tokens.58:12 counts down to when the prompt cache expires: the last main-loop request's start plus the cache TTL. It turns yellow under 5 minutes and reads expired after.| Option | Values | Default |
|---|---|---|
cacheTtl | 5m, 1h | 1h |
autoWarm | true, false | false |
warmHours | hours | 24 |
With autoWarm on, while the session is idle the mod sends one tiny forked request over the main thread's own prefix shortly before the cache expires (3 minutes before on 1h, 1 minute on 5m). The cache read restarts the TTL, so the next real message reads the cache instead of rewriting the whole context. The fork never enters the transcript.
warmHours after the last turn started, and never warms an expired cache (that would pay the full write it exists to avoid) or a context under 50k tokens.warmed 3x shows how many warms ran since the last turn; warm missed means a warm found the cache already cold, and warming stops until the next turn./omni-token| Command | Effect |
|---|---|
/omni-token | Status: auto-warm on or off and why, cache TTL and time left, warms since the last turn, option names |
/omni-token warm off | Stop auto-warming in this session, at once; other sessions and the autoWarm option are unchanged |
/omni-token warm on | Auto-warm this session even with autoWarm off |
/omni-token warm reset | Follow the autoWarm option again |
On a 1-hour TTL, keeping a cache warm costs 0.2/7.8 of a cold restart per hour, so it pays off for gaps under about 39 hours, if you come back.
The API's usage figures don't say which TTL a session runs on, so set it to match yours: 1h on most Claude subscriptions, 5m on the API default or in usage overage.
Load a mod straight from this checkout, with hot reload on save:
claude --plugin-dir ./plugins/omni-token
Check one before committing:
claude plugin validate ./plugins/omni-token
claude plugin test ./plugins/omni-tokenhooks/register.tsx 262 lines1import { atom, read, update } from 'claude-code'
2import type { EngineInterface, Register } from 'claude-code'
3
4import type { Cache, Reading, Warm } from '../types'
5
6const MAX_TURNS = 12
7const BARS = '▁▂▃▄▅▆▇█'
8
9const history = atom({ plugin: 'omni-token', key: 'history' } as const, [])
10// Output tokens per second of the last turn, timed from each request's start.
11const speed = atom({ plugin: 'omni-token', key: 'speed' } as const, null)
12// Last turn's cache hit rate and when the last main-loop request touched the cache.
13const cache = atom({ plugin: 'omni-token', key: 'cache' } as const, null)
14// Wall clock, ticked every second so the cache countdown redraws.
15const clock = atom({ plugin: 'omni-token', key: 'now' } as const, 0)
16const warm = atom({ plugin: 'omni-token', key: 'warm' } as const, { count: 0, missed: false })
17const warmOverride = atom({ plugin: 'omni-token', key: 'warmOverride' } as const, null)
18
19// Below this a cold restart is cheap, so warming isn't worth it.
20const MIN_WARM_TOKENS = 50_000
21const WARM_PROMPT = 'Cache keep-alive. Reply with exactly: ok'
22
23function forecast(percent: number) {
24 if (percent >= 90) return { word: 'Compact soon', color: 'red' }
25 if (percent >= 75) return { word: 'Storm', color: 'magenta' }
26 if (percent >= 50) return { word: 'Showers', color: 'blue' }
27 if (percent >= 25) return { word: 'Cloudy', color: 'cyan' }
28 return { word: 'Clear', color: 'yellow' }
29}
30
31function short(n: number) {
32 const trim = (x: number) => x.toFixed(1).replace(/\.0$/, '')
33 if (Math.abs(n) >= 1e6) return `${trim(n / 1e6)}M`
34 if (Math.abs(n) >= 1e3) return `${trim(n / 1e3)}k`
35 return `${Math.round(n)}`
36}
37
38function sparkline(readings: Reading[]) {
39 const peak = Math.max(...readings.map(r => r.tokens), 1)
40 return readings
41 .map(r => BARS[Math.min(BARS.length - 1, Math.floor((r.tokens / peak) * (BARS.length - 1) + 0.5))])
42 .join('')
43}
44
45function countdown(ms: number) {
46 const total = Math.ceil(ms / 1000)
47 const m = Math.floor(total / 60)
48 const sec = String(total % 60).padStart(2, '0')
49 return `${m}:${sec}`
50}
51
52// One forked request over the main thread's prefix: a cache read restarts its TTL.
53async function warmCache($: EngineInterface, startedAt: number) {
54 const r = await $.model.fork({ prompt: WARM_PROMPT })
55 if (!('usage' in r)) return
56 const hit = r.usage.cache_read_input_tokens > 0
57 await update($, warm, w => ({ count: w.count + 1, missed: !hit }))
58 if (hit) await update($, cache, c => (c ? { ...c, lastRequestAt: startedAt } : c))
59}
60
61type WarmConfig = { autoWarm: boolean; ttlMs: number; marginMs: number; warmMs: number }
62
63// What the hooks know of the main loop right now; starts over on a reload.
64const live = { isBusy: false, isWarming: false, lastTurnAt: 0 }
65
66async function maybeWarm($: EngineInterface, cfg: WarmConfig) {
67 if (live.isBusy || live.isWarming || live.lastTurnAt === 0) return
68 if (!((await read($, warmOverride)) ?? cfg.autoWarm)) return
69 const now = await $.clock.now()
70 if (now - live.lastTurnAt > cfg.warmMs) return
71 const [cached, readings, w] = [await read($, cache), await read($, history), await read($, warm)]
72 if (cached === null || w.missed) return
73 const left = cached.lastRequestAt + cfg.ttlMs - now
74 // Already expired: a warm would pay a full write, the very cost it exists to avoid.
75 if (left <= 0 || left > cfg.marginMs) return
76 const tokens = readings[readings.length - 1]?.tokens ?? 0
77 if (tokens < MIN_WARM_TOKENS) return
78 live.isWarming = true
79 try {
80 await warmCache($, now)
81 } finally {
82 live.isWarming = false
83 }
84}
85
86async function runCommand($: EngineInterface, cfg: WarmConfig, args: string) {
87 const [verb, value] = args.trim().toLowerCase().split(/\s+/)
88 if (verb === 'warm' && (value === 'on' || value === 'off')) {
89 await update($, warmOverride, () => value === 'on')
90 return value === 'on'
91 ? 'Auto-warm on for this session.'
92 : 'Auto-warm off for this session. Other sessions and the autoWarm option are unchanged.'
93 }
94 if (verb === 'warm' && value === 'reset') {
95 await update($, warmOverride, () => null)
96 return `Auto-warm follows the autoWarm option again (${cfg.autoWarm ? 'on' : 'off'}).`
97 }
98 if (verb !== undefined && verb !== '' && verb !== 'status') {
99 return 'Usage: /omni-token [status] | /omni-token warm on|off|reset'
100 }
101
102 const override = await read($, warmOverride)
103 const isOn = override ?? cfg.autoWarm
104 const cached = await read($, cache)
105 const w = await read($, warm)
106 const now = await $.clock.now()
107 const left = cached === null ? null : cached.lastRequestAt + cfg.ttlMs - now
108 const lines = [
109 `auto-warm: ${isOn ? 'on' : 'off'} (${override === null ? 'from the autoWarm option' : 'set for this session'})`,
110 `cache TTL: ${cfg.ttlMs === 5 * 60_000 ? '5m' : '1h'}, expires in ${left === null ? 'n/a' : left <= 0 ? 'expired' : countdown(left)}`,
111 `keep warm for: ${cfg.warmMs / 3_600_000}h after the last turn`,
112 `warms since the last turn: ${w.count}${w.missed ? ' (last one found the cache cold; stopped)' : ''}`,
113 ]
114 if (isOn && live.lastTurnAt === 0) lines.push('Warming starts after your next message.')
115 lines.push('Options: cacheTtl, autoWarm, warmHours (claude plugin configure omni-token@claude-code-mods)')
116 return lines.join('\n')
117}
118
119async function measure($: EngineInterface) {
120 const { context } = await $.session.usage()
121 if (context.tokens === undefined) return
122 const reading: Reading = { tokens: context.tokens, window: context.window }
123 await update($, history, h => [...h, reading].slice(-MAX_TURNS))
124}
125
126export const register: Register = (on, options) => {
127 const ttlMs = options.cacheTtl === '5m' ? 5 * 60_000 : 60 * 60_000
128 // Warm this long before expiry: room for the request to reach the API.
129 const marginMs = options.cacheTtl === '5m' ? 60_000 : 3 * 60_000
130 const cfg: WarmConfig = {
131 autoWarm: options.autoWarm === true,
132 ttlMs,
133 marginMs,
134 warmMs: (typeof options.warmHours === 'number' ? options.warmHours : 24) * 3_600_000,
135 }
136
137 // Per main-loop turn: output tokens and milliseconds from request to response end.
138 let streamed = { turnId: '', tokens: 0, ms: 0 }
139
140 on('turn.step', async function* ($, e, next) {
141 if (e.agentId !== undefined || !live.isBusy) return yield* next(e)
142 if (streamed.turnId !== e.turnId) streamed = { turnId: e.turnId, tokens: 0, ms: 0 }
143
144 // Timed from the request, so thinking (streamed, summarized or not) and
145 // prefill latency all count against the output tokens.
146 const startedAt = await $.clock.now()
147 const result = yield* next(e)
148 if (result.usage) {
149 streamed.tokens += result.usage.output_tokens
150 streamed.ms += (await $.clock.now()) - startedAt
151 // The cache's TTL restarts when a request reads or writes it.
152 await update($, cache, c => ({ hitRate: c?.hitRate ?? null, lastRequestAt: startedAt }))
153 }
154 return result
155 })
156
157 on('turn.start', async ($, e, next) => {
158 live.isBusy = true
159 live.lastTurnAt = await $.clock.now()
160 await update($, warm, () => ({ count: 0, missed: false }))
161 return next(e)
162 })
163
164 on('session.start', async ($, e, next) => {
165 const result = await next(e)
166 await $.command.register({
167 name: 'omni-token',
168 description: 'Show omni-token status; "warm on|off|reset" toggles cache auto-warm for this session',
169 })
170 const tick = async () => {
171 const t = await $.clock.now()
172 await update($, clock, () => t)
173 }
174 await tick()
175 $.clock.every(1000, () => void tick().then(() => maybeWarm($, cfg)))
176 // Seed one reading on a fresh load (or a hot reload mid-session).
177 if ((await read($, history)).length === 0) await measure($)
178 return result
179 })
180
181 on('command.run', { command: 'omni-token' }, async ($, e) => ({ text: await runCommand($, cfg, e.args) }))
182
183 on('turn.complete', async ($, e, next) => {
184 const result = await next(e)
185 if (e.agentId !== undefined) return result
186 live.isBusy = false
187 await measure($)
188 if (streamed.turnId === e.turnId && streamed.ms > 0) {
189 const tps = streamed.tokens / (streamed.ms / 1000)
190 await update($, speed, () => tps)
191 }
192 if (e.usage) {
193 const u = e.usage
194 const input = u.input_tokens + u.cache_read_input_tokens + u.cache_creation_input_tokens
195 if (input > 0) {
196 const hitRate = u.cache_read_input_tokens / input
197 await update($, cache, c => (c ? { ...c, hitRate } : c))
198 }
199 }
200 return result
201 })
202
203 on('ui.render', { component: 'AbovePrompt' }, async ($, e, next) => {
204 const readings = await read($, history)
205 const tps = await read($, speed)
206 const cached = await read($, cache)
207 const nowMs = await read($, clock)
208 const warmed = await read($, warm)
209 if (e.props.hasSurvey || readings.length === 0) return next(e)
210
211 const { Box, Text } = $.ui.resolve(e)
212 const now = readings[readings.length - 1]!
213 const percent = Math.round((now.tokens / now.window) * 100)
214 const sky = forecast(percent)
215 const prev = readings[readings.length - 2] ?? null
216 const delta = prev ? now.tokens - prev.tokens : null
217
218 return (
219 <Box>
220 <Text color={sky.color} bold>
221 {sky.word}
222 </Text>
223 <Text>
224 {' '}
225 {percent}%{' '}
226 </Text>
227 <Text dimColor>
228 {short(now.tokens)} / {short(now.window)}
229 {' '}
230 </Text>
231 <Text color={sky.color}>{sparkline(readings)}</Text>
232 {delta !== null ? (
233 <Text dimColor>
234 {' '}
235 {delta >= 0 ? `▲ +${short(delta)}` : `▼ −${short(-delta)}`} last turn
236 </Text>
237 ) : null}
238 {tps !== null ? (
239 <Text dimColor>
240 {' '}{Math.round(tps)} tok/s
241 </Text>
242 ) : null}
243 {cached !== null ? cacheInfo(Text, cached, cached.lastRequestAt + ttlMs - nowMs) : null}
244 {warmed.missed ? (
245 <Text color="red">{' '}warm missed</Text>
246 ) : warmed.count > 0 ? (
247 <Text dimColor>{' '}warmed {warmed.count}x</Text>
248 ) : null}
249 </Box>
250 )
251 })
252}
253
254// Text is the surface's own element, from $.ui.resolve.
255function cacheInfo(Text: any, cached: Cache, left: number) {
256 const rate = cached.hitRate === null ? '' : ` ${Math.round(cached.hitRate * 100)}%`
257 const label = <Text dimColor>{' '}cache{rate} </Text>
258 if (left <= 0) return <Text>{label}<Text color="red">expired</Text></Text>
259 if (left < 5 * 60_000) return <Text>{label}<Text color="yellow">{countdown(left)}</Text></Text>
260 return <Text>{label}<Text dimColor>{countdown(left)}</Text></Text>
261}
262types/index.d.ts 22 lines1export type Reading = { tokens: number; window: number }
2
3/** The last main-loop request's cache figures. */
4export type Cache = { hitRate: number | null; lastRequestAt: number }
5
6/** Auto-warming since the last turn: how many warms, and whether the last one found the cache cold. */
7export type Warm = { count: number; missed: boolean }
8
9declare module 'claude-code' {
10 interface PluginState {
11 'omni-token': {
12 history: Reading[]
13 speed: number | null
14 cache: Cache | null
15 now: number
16 warm: Warm
17 /** `/omni-token warm on|off` for this session; null follows the autoWarm option. */
18 warmOverride: boolean | null
19 }
20 }
21}
22