Shows how much of the context each model request reuses versus reprocesses, measured from the prompt-cache token counts.

A hand-owned decode loop over 0.5-4B open weights for training loops, up to 8B for inference; post-train → RL → inspect, on one machine. Qwen3 on Apple Silicon via MLX.
workbench/ — Python backendengine/ — hand-owned MLX generation loop, KV-cache reuse, control queue, tapscontext/ — event-sourced, editable context (typed segments), token assembly, eviction policiesserver/ — FastAPI WebSocket app, wire protocol, chat-template framingconfig/ — durable task configuration (steering, saved by name)attachments/ — text/PDF/OCR/VLM extraction pipeline and transient storetools/ — safe tool registry (calculator, current-time, context search)cli/ — terminal client: streaming, context inspection, priced edits, per-token logprobsstatic/ — minimal fallback clientfrontend/ — Next.js app (chat UI, context inspector, media viewer)tests/ — pytest suiteexperiments/ — engine ↔ stock mlx-lm parity harnesshooks/register.tsx 299 lines1import { atom, read, update } from 'claude-code'
2import type { Register } from 'claude-code'
3
4import {
5 type Fixture,
6 type StoredSession,
7 compactionRecord,
8 compactionsKey,
9 fixtureSource,
10 isCompactionsKey,
11 isSessionKey,
12 replay,
13 sampleOf,
14 sessionsReport,
15 shouldTell,
16 staleKeys,
17 storeKey,
18} from './record'
19import {
20 bar,
21 isMainThread,
22 pct,
23 retain,
24 sessionLines,
25 spendText,
26 shapeText,
27 statusText,
28 tok,
29 totals,
30 view,
31} from './cost'
32
33const PANE = 'lumen-cache'
34const TITLE = 'Cache horizon'
35
36/** Enough main-thread history for a long session; the oldest fall off the front. */
37const KEEP_MAIN = 400
38
39/**
40 * Forks are not a signal (see isMainThread), so they get a small separate
41 * budget: enough to total what subagents cost, never enough to crowd out the
42 * main thread.
43 */
44const KEEP_FORKS = 60
45
46const samples = atom({ plugin: 'lumen-cache', key: 'samples' } as const, [])
47const compactions = atom({ plugin: 'lumen-cache', key: 'compactions' } as const, [])
48
49/** Enough to keep old verdicts true without growing without bound. */
50const KEEP_COMPACTIONS = 50
51
52/** Sessions kept in the cross-session store; the store caps at 4 MiB of JSON. */
53const KEEP_SESSIONS = 20
54
55/** The slice of `$` the persistence helpers use. */
56type Store = {
57 store: {
58 set: (key: string, value: unknown) => Promise<void>
59 keys: () => Promise<string[]>
60 delete: (key: string) => Promise<void>
61 }
62 session: { id: () => Promise<string> }
63 ui: { toast: (text: string) => void }
64}
65
66/** Sessions already told their store writes are failing. */
67const told = new Set<string>()
68
69// The store is a convenience copy; the atoms are the record. A rejected
70// write (a full 4 MiB store, say) must not take the hook down with it, or
71// recording and the status line stop for the rest of the process. One toast
72// per session: the status line is rewritten every request, so a note there
73// would be gone at once, and a toast per request would be noise.
74async function persist($: Store, key: string, value: unknown): Promise<void> {
75 try {
76 await $.store.set(key, value)
77 } catch (err) {
78 if (shouldTell(told, await $.session.id())) {
79 $.ui.toast(
80 `lumen-cache: could not save to the store (${err instanceof Error ? err.message : String(err)}). Measuring continues, but this session will not be kept.`,
81 )
82 }
83 }
84}
85
86// The store is a 4 MiB budget shared by every session ever recorded.
87async function prune($: Store): Promise<void> {
88 try {
89 for (const stale of staleKeys(await $.store.keys(), KEEP_SESSIONS)) {
90 await $.store.delete(stale)
91 }
92 } catch {
93 // Pruning is housekeeping; failing it must not end the session hook.
94 }
95}
96
97export const register: Register = on => {
98 on('session.start', async ($, e, next) => {
99 await $.command.register({
100 name: 'lumen-cache',
101 description: 'Show how much of the context each request reuses vs reprocesses',
102 })
103 await $.command.register({
104 name: 'lumen-cache-export',
105 description: 'Write this session\u2019s measurements out as a replayable fixture',
106 })
107 await $.command.register({
108 name: 'lumen-cache-sessions',
109 description: 'Review every recorded session: what each one reprocessed and how much of it was rebuilt',
110 })
111 // Opened unasked it waits for a wide terminal; /lumen-cache opens it anywhere.
112 void $.ui.open({ id: PANE, title: TITLE })
113
114 await prune($)
115
116 return next(e)
117 })
118
119 // /clear ends the conversation and fires no session.start for the next one,
120 // so this is the only place a cleared session's keys can be pruned behind.
121 on('session.end', async ($, e, next) => {
122 await prune($)
123
124 return next(e)
125 })
126
127 on('command.run', { command: 'lumen-cache-export' }, async $ => {
128 const sampled = await read($, samples)
129 if (sampled.length === 0) return { text: 'Nothing measured yet \u2014 nothing to export.' }
130
131 const id = await $.session.id()
132 const captured: Omit<Fixture, 'expected'> = {
133 name: `capture-${id.slice(0, 8)}`,
134 capturedAt: new Date().toISOString().slice(0, 10),
135 note: `Captured live from session ${id}.`,
136 compactions: await read($, compactions),
137 samples: sampled,
138 }
139 const fixture: Fixture = { ...captured, expected: replay(captured) }
140 const path = `${await $.session.cwd()}/${fixture.name}.ts`
141 await $.fs.write(path, fixtureSource(fixture))
142
143 return {
144 text: [
145 `Wrote ${sampled.length} measurements to ${path}`,
146 '',
147 'To make it a regression test, move it into hooks/fixtures/ and add:',
148 ` import { fixture as ${fixture.name.replace(/-/g, '_')} } from './fixtures/${fixture.name}'`,
149 '',
150 'expected was filled from the current logic, so review it before trusting it as a baseline.',
151 ].join('\n'),
152 }
153 })
154
155 // Every session ever recorded, read straight back out of the store. The
156 // decisions are all in sessionsReport, which takes the raw keys and values
157 // so a test can hand it an old session's records as the store holds them.
158 on('command.run', { command: 'lumen-cache-sessions' }, async $ => {
159 const entries: StoredSession[] = []
160 for (const key of (await $.store.keys()).filter(k => isSessionKey(k) || isCompactionsKey(k))) {
161 entries.push({ key, value: await $.store.get(key) })
162 }
163
164 return { text: sessionsReport(entries, await $.session.id()) }
165 })
166
167 on('command.run', { command: 'lumen-cache' }, async $ => {
168 await $.ui.open({ id: PANE, title: TITLE })
169 const v = view(await read($, samples), await read($, compactions))
170
171 return {
172 text: v.isEmpty
173 ? `${TITLE} pane opened. No model request measured yet.`
174 : `${TITLE} pane opened. ${v.session.requests} main-thread requests; ${tok(v.session.reprocessed)} tokens reprocessed, ${pct(v.session.rebuiltShare)}% of them rebuilt.`,
175 }
176 })
177
178 // A compaction rewrites the whole prefix, so it is the one cause we can name.
179 //
180 // The matcher requires at least one message. `/compact` with nothing to
181 // compact dispatches with `messages: []`, and next() refuses to be passed an
182 // empty transcript ("a compaction leaves at least one"), so an unguarded hook
183 // throws and is skipped — noisily, in the transcript. No compaction happened
184 // in that case, so there is nothing to record either.
185 on('session.compact', { messages: [{}] }, async ($, e, next) => {
186 const done = await next(e)
187 // A veto leaves the conversation as it was: nothing compacted, nothing to record.
188 if (done.skip !== undefined) return done
189
190 const record = compactionRecord(Date.now(), e.trigger, done)
191 await update($, compactions, list => [...list, record].slice(-KEEP_COMPACTIONS))
192 // Mirrored, not moved: $.state dies with the session, and a collapse whose
193 // compaction time died with it can never be attributed again.
194 await persist($, compactionsKey(await $.session.id()), await read($, compactions))
195
196 return done
197 })
198
199 // turn.step streams, so the hook is a generator and next(e) is the stream.
200 on('turn.step', async function* ($, e, next) {
201 // Before the stream drains: the interval a cache cares about starts here.
202 const startedAt = Date.now()
203 const result = yield* next(e)
204 // Both halves of the step make the record: the request's shape from `e`,
205 // its cost from the response. Null when the API reported no usage.
206 const sample = sampleOf(e, result, Date.now(), startedAt)
207 if (sample === null) return result
208
209 await update($, samples, list => retain([...list, sample], KEEP_MAIN, KEEP_FORKS))
210
211 const all = await read($, samples)
212 // Kept across sessions too, so captures accumulate rather than vanish.
213 await persist($, storeKey(await $.session.id()), all)
214
215 $.ui.status(statusText(totals(all.filter(isMainThread))))
216
217 return result
218 })
219
220 on('ui.render', { component: 'Pane', requestId: PANE }, async ($, e) => {
221 const { Box, Text } = $.ui.resolve(e)
222 const v = view(await read($, samples), await read($, compactions))
223 const width = Math.max(10, Math.min(36, (e.props.bodyColumns ?? 44) - 8))
224
225 if (v.isEmpty || v.last === null) {
226 return (
227 <Box flexDirection="column">
228 <Text dimColor>No model request measured yet.</Text>
229 <Text dimColor>The pane fills on the first response.</Text>
230 </Box>
231 )
232 }
233
234 const last = v.last
235
236 return (
237 <Box flexDirection="column" gap={1}>
238 <Box flexDirection="column">
239 <Text bold>Last request</Text>
240 <Text color={last.isCollapse ? 'yellow' : 'green'}>{bar(last.rate, width)}</Text>
241 <Text>{pct(last.rate)}% reused</Text>
242 <Text dimColor>
243 {tok(last.held)} held · {tok(last.reprocessed)} reprocessed
244 </Text>
245 {shapeText(last) !== null && <Text dimColor>{shapeText(last)}</Text>}
246 </Box>
247
248 <Box flexDirection="column">
249 <Text bold>Session · main thread</Text>
250 <Text>{sessionLines(v.session)[0]}</Text>
251 <Text>{sessionLines(v.session)[1]}</Text>
252 {sessionLines(v.session)[2] !== '' && <Text dimColor>{sessionLines(v.session)[2]}</Text>}
253 {spendText(v.compaction) !== null && <Text dimColor>{spendText(v.compaction)}</Text>}
254 </Box>
255
256 {v.forks !== null && (
257 <Box flexDirection="column">
258 <Text bold dimColor>Subagents</Text>
259 <Text dimColor>
260 {v.forks.requests} requests · {tok(v.forks.reprocessed)} reprocessed
261 </Text>
262 <Text dimColor>A fork pays for the whole prefix; not a signal.</Text>
263 </Box>
264 )}
265
266 {v.collapses.length > 0 && (
267 <Box flexDirection="column">
268 <Text bold color="yellow">
269 {v.collapses.length === 1 ? '1 collapse' : `${v.collapses.length} collapses`}
270 </Text>
271 {v.collapses.map(one => (
272 <Box flexDirection="column">
273 <Text color="yellow">
274 {tok(one.lost)} lost · {pct(one.share)}% of the prior context
275 </Text>
276 <Text dimColor>{tok(one.reprocessed)} reprocessed</Text>
277 <Text dimColor>{one.why}</Text>
278 {one.engine !== null && (
279 <Text color={one.engine.disagrees ? 'red' : undefined} dimColor={!one.engine.disagrees}>
280 {one.engine.text}
281 </Text>
282 )}
283 {one.notes.map(note => (
284 <Text dimColor>{note}</Text>
285 ))}
286 </Box>
287 ))}
288 </Box>
289 )}
290
291 <Box flexDirection="column">
292 <Text dimColor>cost = T - a, where a is what the cache held.</Text>
293 <Text dimColor>Measured, not estimated.</Text>
294 </Box>
295 </Box>
296 )
297 })
298}
299hooks/record.ts 533 lines1/**
2 * The harness: a real session's samples, kept across sessions and replayable
3 * as a regression test.
4 *
5 * The pane answers "what is happening now". This answers "did that change" —
6 * a captured run carries what it produced when it was taken, so a later edit
7 * to the cost logic that alters a real session's verdicts fails a test instead
8 * of quietly redrawing.
9 */
10import type { TurnStepInput, TurnStepResult } from 'claude-code'
11
12import type { Cause, CompactionSpend, Compaction, CompactionUsage, CompactTrigger, Sample, Shape, StoredCompaction } from './cost'
13import { CAUSE_TEXT, allCollapses, compactionSpend, spendText, isMainThread, pct, shape, tallyText, tok, totals, view } from './cost'
14
15/**
16 * The record one measured request leaves: the request's shape beside its cost.
17 *
18 * `null` when the step carried no usage — the API reported nothing, so there
19 * is nothing measured to record. The fields are taken from both halves of the
20 * step: `messageCount` and `effort` from the request the engine is about to
21 * send, `stopReason` and the tool calls from the response it got back.
22 *
23 * `startedAt` is when the step began, taken by the caller before the response
24 * streamed; `at` is when it ended. Left off when not given.
25 *
26 * A field the engine did not give is left off the record rather than written
27 * as a zero, so a reader can tell silence from a measurement.
28 */
29export const sampleOf = (
30 e: Pick<TurnStepInput, 'agentId' | 'effort' | 'messageCount'>,
31 r: Pick<TurnStepResult, 'turnId' | 'index' | 'stopReason' | 'toolUses' | 'usage'>,
32 at: number,
33 startedAt?: number,
34): Sample | null => {
35 const usage = r.usage
36 if (usage === null) return null
37
38 return {
39 turnId: r.turnId,
40 index: r.index,
41 ...(e.agentId === undefined ? {} : { agentId: e.agentId }),
42 model: usage.model,
43 reused: usage.cache_read_input_tokens,
44 written: usage.cache_creation_input_tokens,
45 uncached: usage.input_tokens,
46 output: usage.output_tokens,
47 at,
48 ...(startedAt === undefined ? {} : { startedAt }),
49 messageCount: e.messageCount,
50 stopReason: r.stopReason,
51 toolUseCount: r.toolUses.length,
52 ...(e.effort === undefined ? {} : { effort: e.effort }),
53 }
54}
55
56export type Verdict = {
57 requests: number
58 reused: number
59 reprocessed: number
60 collapses: { lost: number; cause: Cause }[]
61}
62
63export type Fixture = {
64 name: string
65 capturedAt: string
66 /** How it was captured and anything the numbers do not say for themselves. */
67 note: string
68 compactions: StoredCompaction[]
69 samples: Sample[]
70 /** What this capture produced when taken. A diff here is a regression. */
71 expected: Verdict
72}
73
74/** What a capture produces under the current logic. */
75export const replay = (f: Pick<Fixture, 'samples' | 'compactions'>): Verdict => {
76 const v = view(f.samples, f.compactions)
77
78 return {
79 requests: v.session.requests,
80 reused: v.session.reused,
81 reprocessed: v.session.reprocessed,
82 collapses: v.collapses.map(c => ({ lost: c.lost, cause: c.cause })),
83 }
84}
85
86/** The prefix every session's key carries in the cross-session store. */
87export const SESSION_PREFIX = 'session:'
88
89/** Where a session's samples live in the plugin's cross-session store. */
90export const storeKey = (sessionId: string): string => `${SESSION_PREFIX}${sessionId}`
91
92/** Is this store key one of the recorded sessions, rather than anything else? */
93export const isSessionKey = (key: string): boolean => key.startsWith(SESSION_PREFIX)
94
95/** The session a key names; the key itself when there is no prefix to strip. */
96export const sessionIdFromKey = (key: string): string =>
97 isSessionKey(key) ? key.slice(SESSION_PREFIX.length) : key
98
99/** The prefix every session's compaction times carry in the store. */
100export const COMPACTIONS_PREFIX = 'compactions:'
101
102/** Where a session's compaction times live, beside its samples under `storeKey`. */
103export const compactionsKey = (sessionId: string): string => `${COMPACTIONS_PREFIX}${sessionId}`
104
105/** Is this store key one session's compaction times? */
106export const isCompactionsKey = (key: string): boolean => key.startsWith(COMPACTIONS_PREFIX)
107
108/**
109 * Which keys to delete so the store keeps the newest `keep` sessions.
110 *
111 * Sessions are dropped oldest first, in the order the store lists them. A
112 * session's compaction times go with it, and compaction times whose session
113 * has no samples kept are dropped too — they attribute nothing.
114 */
115export const staleKeys = (keys: readonly string[], keep: number): string[] => {
116 const sessions = keys.filter(isSessionKey)
117 const stale = sessions.slice(0, Math.max(0, sessions.length - keep))
118 const kept = new Set(sessions.slice(stale.length).map(sessionIdFromKey))
119 const orphans = keys.filter(
120 k => isCompactionsKey(k) && !kept.has(k.slice(COMPACTIONS_PREFIX.length)),
121 )
122
123 return [...stale, ...orphans]
124}
125
126/**
127 * Should this failed store write be announced? Once per session: the first
128 * failure says so and records the session in `told`; later ones stay quiet,
129 * so a full store costs one toast rather than one per request.
130 */
131export const shouldTell = (told: Set<string>, sessionId: string): boolean => {
132 if (told.has(sessionId)) return false
133 told.add(sessionId)
134
135 return true
136}
137
138/**
139 * Is this stored value one of this mod's records, with the counts the
140 * arithmetic reads intact?
141 *
142 * The store holds JSON written by every earlier build of this mod, and it is
143 * a file on disk besides. The four numbers checked here are the ones every
144 * version has written and every function in cost.ts reads; `output` is not
145 * among them because nothing computes from it. The enriched fields are not
146 * checked at all — a record without them is old, not broken.
147 */
148export const isSample = (value: unknown): value is Sample => {
149 if (typeof value !== 'object' || value === null) return false
150 const r = value as Record<string, unknown>
151
152 return [r.reused, r.written, r.uncached, r.at].every(
153 n => typeof n === 'number' && Number.isFinite(n),
154 )
155}
156
157/**
158 * A stored session's records, read back. Anything that is not a record with
159 * its counts intact is dropped rather than carried into the arithmetic, where
160 * one missing count would turn a whole session's figures into NaN.
161 */
162export const readSamples = (value: unknown): Sample[] =>
163 Array.isArray(value) ? value.filter(isSample) : []
164
165const finite = (n: unknown): n is number => typeof n === 'number' && Number.isFinite(n)
166
167const TRIGGERS: readonly CompactTrigger[] = ['manual', 'auto', 'plugin', 'precompute']
168
169/**
170 * The record one compaction leaves: when it ran, and whatever of the engine's
171 * own figures it supplied.
172 *
173 * A figure the event did not give is left off rather than written as zero. A
174 * `usage` is kept only whole: with one of its four counts missing or not a
175 * number, the request is unknown, and a partial one would price it low.
176 */
177export const compactionRecord = (
178 at: number,
179 trigger: CompactTrigger,
180 r: { tokensBefore?: number; tokensAfter?: number; usage?: Partial<CompactionUsage> | null },
181): Compaction => {
182 const u = r.usage
183
184 return {
185 at,
186 trigger,
187 ...(finite(r.tokensBefore) ? { tokensBefore: r.tokensBefore } : {}),
188 ...(finite(r.tokensAfter) ? { tokensAfter: r.tokensAfter } : {}),
189 ...(usageOf(u) === null ? {} : { usage: usageOf(u) as CompactionUsage }),
190 }
191}
192
193const usageOf = (u: unknown): CompactionUsage | null => {
194 if (typeof u !== 'object' || u === null) return null
195 const r = u as Record<string, unknown>
196 if (
197 !finite(r.input_tokens) ||
198 !finite(r.output_tokens) ||
199 !finite(r.cache_read_input_tokens) ||
200 !finite(r.cache_creation_input_tokens)
201 ) {
202 return null
203 }
204
205 return {
206 input_tokens: r.input_tokens,
207 output_tokens: r.output_tokens,
208 cache_read_input_tokens: r.cache_read_input_tokens,
209 cache_creation_input_tokens: r.cache_creation_input_tokens,
210 }
211}
212
213/**
214 * One stored compaction, read back: a bare timestamp stays a bare timestamp
215 * (it is an old record, not a broken one), an object keeps only the fields
216 * that are intact, and anything without a time is dropped.
217 */
218export const readCompaction = (value: unknown): StoredCompaction | null => {
219 if (finite(value)) return value
220 if (typeof value !== 'object' || value === null) return null
221 const r = value as Record<string, unknown>
222 if (!finite(r.at)) return null
223
224 return {
225 at: r.at,
226 ...(TRIGGERS.includes(r.trigger as CompactTrigger) ? { trigger: r.trigger as CompactTrigger } : {}),
227 ...(finite(r.tokensBefore) ? { tokensBefore: r.tokensBefore } : {}),
228 ...(finite(r.tokensAfter) ? { tokensAfter: r.tokensAfter } : {}),
229 ...(usageOf(r.usage) === null ? {} : { usage: usageOf(r.usage) as CompactionUsage }),
230 }
231}
232
233/**
234 * A stored session's compactions, read back; null when none were ever
235 * recorded for it. The two are different facts: no list means the session
236 * predates the record, an empty list means it was recorded that none ran.
237 */
238export const readCompactions = (value: unknown): StoredCompaction[] | null =>
239 Array.isArray(value)
240 ? value.map(readCompaction).filter((c): c is StoredCompaction => c !== null)
241 : null
242
243/** One session as the review lists it. Main-thread requests only, as the pane's are. */
244export type SessionRow = {
245 sessionId: string
246 requests: number
247 /** Tokens the cache served, summed over requests: a billing figure, not a size. */
248 served: number
249 reprocessed: number
250 /** Reprocessed tokens that had been cached and were lost. */
251 rebuilt: number
252 /** `rebuilt / reprocessed`; 0 for a session that reprocessed nothing. */
253 rebuiltShare: number
254 collapses: number
255 /** How many of them each cause accounts for; only `model-change` and `compaction` are ever proven. */
256 causes: Record<Cause, number>
257 /** What its compactions' own requests cost, beyond `reprocessed`; zeros when none were recorded. */
258 compaction: CompactionSpend
259 /** Compaction collapses whose engine sizes could be compared with the measured loss, and how many disagreed. */
260 engine: { checked: number; disagreed: number }
261 /** Were this session's compaction times recorded? False for any session from before they were. */
262 compactionsRecorded: boolean
263 /** The latest record's timestamp, 0 when none; what the rows sort on. */
264 lastSeen: number
265}
266
267/** A key and the value under it, as `$.store` hands them over. */
268export type StoredSession = {
269 key: string
270 value: unknown
271}
272
273export const sessionRow = (
274 sessionId: string,
275 samples: readonly Sample[],
276 compactions: readonly StoredCompaction[] | null = null,
277): SessionRow => {
278 const main = samples.filter(isMainThread)
279 const t = totals(main)
280 // The pane's own attribution, over the same samples: the two views cannot
281 // disagree about an event. With no record, nothing can be proven.
282 const flagged = allCollapses(samples, compactions ?? [])
283 const causes: Record<Cause, number> = { 'model-change': 0, compaction: 0, unattributed: 0 }
284 for (const c of flagged) causes[c.cause] += 1
285
286 const engine = { checked: 0, disagreed: 0 }
287 for (const c of flagged) {
288 if (c.engine === null || !c.engine.compared) continue
289 engine.checked += 1
290 if (c.engine.disagrees) engine.disagreed += 1
291 }
292
293 let lastSeen = 0
294 for (const s of samples) if (s.at > lastSeen) lastSeen = s.at
295
296 return {
297 sessionId,
298 requests: t.requests,
299 served: t.reused,
300 reprocessed: t.reprocessed,
301 rebuilt: t.rebuilt,
302 rebuiltShare: t.rebuiltShare,
303 collapses: flagged.length,
304 causes,
305 compaction: compactionSpend(compactions ?? []),
306 engine,
307 compactionsRecorded: compactions !== null,
308 lastSeen,
309 }
310}
311
312/** Every recorded session's figures, newest first; other keys are not sessions. */
313export const sessionRows = (entries: readonly StoredSession[]): SessionRow[] =>
314 entries
315 .filter(entry => isSessionKey(entry.key))
316 .map(entry => {
317 const id = sessionIdFromKey(entry.key)
318 const times = entries.find(other => other.key === compactionsKey(id))
319
320 return sessionRow(
321 id,
322 readSamples(entry.value),
323 times === undefined ? null : readCompactions(times.value),
324 )
325 })
326 .sort((a, b) => b.lastSeen - a.lastSeen)
327
328/** Every recorded session summed. Rates are recomputed, never averaged. */
329export type AllSessions = {
330 sessions: number
331 requests: number
332 served: number
333 reprocessed: number
334 rebuilt: number
335 rebuiltShare: number
336 collapses: number
337}
338
339export const acrossSessions = (rows: readonly SessionRow[]): AllSessions => {
340 let requests = 0
341 let served = 0
342 let reprocessed = 0
343 let rebuilt = 0
344 let collapses = 0
345
346 for (const row of rows) {
347 requests += row.requests
348 served += row.served
349 reprocessed += row.reprocessed
350 rebuilt += row.rebuilt
351 collapses += row.collapses
352 }
353
354 return {
355 sessions: rows.length,
356 requests,
357 served,
358 reprocessed,
359 rebuilt,
360 // The session shares are over different numbers of tokens, so averaging
361 // them would weight a short session like a long one.
362 rebuiltShare: reprocessed === 0 ? 0 : rebuilt / reprocessed,
363 collapses,
364 }
365}
366
367/** The recorded request shape across every session, main thread only. */
368export const sessionsShape = (entries: readonly StoredSession[]): Shape =>
369 shape(
370 entries
371 .filter(entry => isSessionKey(entry.key))
372 .flatMap(entry => readSamples(entry.value).filter(isMainThread)),
373 )
374
375/**
376 * A row's collapses, with the causes only where they are proven. A session
377 * with no compaction record says so rather than implying its collapses were
378 * not compactions; one with a record names what the record accounts for.
379 */
380export const collapseText = (row: SessionRow): string => {
381 const count = plural(row.collapses, 'collapse', 'collapses')
382 if (row.collapses === 0) return count
383 if (!row.compactionsRecorded) {
384 // A model change is read off the records themselves, so it is named even
385 // where no compaction times were kept.
386 const known = row.causes['model-change']
387 return known === 0
388 ? `${count} (no compaction times recorded)`
389 : `${count} (${known} model-change, no compaction times recorded)`
390 }
391
392 const named = (['model-change', 'compaction', 'unattributed'] as const)
393 .filter(cause => row.causes[cause] > 0)
394 .map(cause => `${row.causes[cause]} ${cause}`)
395
396 return `${count} (${named.join(', ')})`
397}
398
399/**
400 * What the compactions' own requests cost, summed over the rows, said apart
401 * from the totals above it: `turn.step` never sees a summary request, so none
402 * of it is in those figures. Nothing when no compaction was recorded.
403 */
404export const spendLines = (rows: readonly SessionRow[]): string[] => {
405 const all = compactionSpend([])
406 for (const row of rows) {
407 all.compactions += row.compaction.compactions
408 all.withUsage += row.compaction.withUsage
409 all.reprocessed += row.compaction.reprocessed
410 all.output += row.compaction.output
411 }
412 const text = spendText(all)
413
414 return text === null ? [] : ['', `Compactions: ${text}`]
415}
416
417/**
418 * Where the engine's own sizes and the measured loss disagreed, said once the
419 * session rows are summed. Silent when nothing could be checked or all agreed:
420 * agreement needs no report, disagreement is the thing to see. The rows
421 * themselves are in the pane.
422 */
423export const engineLines = (rows: readonly SessionRow[]): string[] => {
424 let checked = 0
425 let disagreed = 0
426 for (const row of rows) {
427 checked += row.engine.checked
428 disagreed += row.engine.disagreed
429 }
430
431 return disagreed === 0
432 ? []
433 : [
434 '',
435 `Engine sizes disagreed with the measured loss on ${disagreed} of ${plural(checked, 'compaction collapse', 'compaction collapses')} that could be checked.`,
436 ]
437}
438
439/** What the report says about causes, given the rows it lists. */
440export const causeNote = (rows: readonly SessionRow[]): string[] => {
441 const blind = rows.filter(
442 row => !row.compactionsRecorded && row.collapses > row.causes['model-change'],
443 ).length
444 const unproven = rows.some(row => row.compactionsRecorded && row.causes.unattributed > 0)
445 const note: string[] = []
446
447 if (blind > 0) {
448 note.push(
449 `${plural(blind, 'session', 'sessions')} with collapses kept no compaction times, so no cause is named here:`,
450 'a changed prefix and a cache entry that lapsed are the same four numbers.',
451 )
452 }
453 if (unproven) {
454 note.push(
455 `Where compaction times were recorded, unattributed means none fell in that collapse's window (${CAUSE_TEXT.unattributed}).`,
456 )
457 }
458 if (note.length === 0) {
459 note.push(
460 `Causes named only where proven: ${CAUSE_TEXT.compaction}; ${CAUSE_TEXT['model-change']} (the two records name different models).`,
461 )
462 }
463
464 return note
465}
466
467const plural = (n: number, one: string, many: string): string => `${n} ${n === 1 ? one : many}`
468
469/**
470 * The whole cross-session review as plain text: a line per session, the
471 * sessions summed, and what the records say about the requests themselves.
472 *
473 * `current` marks the running session's own row where it is already recorded.
474 */
475export const sessionsReport = (
476 entries: readonly StoredSession[],
477 current?: string,
478): string => {
479 const rows = sessionRows(entries)
480 if (rows.length === 0) return 'No session recorded yet — nothing to review.'
481
482 const all = acrossSessions(rows)
483 const s = sessionsShape(entries)
484 const width = Math.max(...rows.map(row => row.sessionId.length))
485
486 const lines = [
487 `${plural(all.sessions, 'recorded session', 'recorded sessions')}, main-thread requests only.`,
488 '',
489 ...rows.map(row =>
490 [
491 ` ${row.sessionId.padEnd(width)}`,
492 `${String(row.requests).padStart(4)} req`,
493 `${tok(row.reprocessed).padStart(7)} reprocessed`,
494 `${String(pct(row.rebuiltShare)).padStart(3)}% rebuilt`,
495 collapseText(row),
496 row.sessionId === current ? '(this session)' : '',
497 ]
498 .join(' ')
499 .trimEnd(),
500 ),
501 '',
502 `All sessions: ${plural(all.requests, 'request', 'requests')} · ${tok(all.reprocessed)} reprocessed · ${pct(all.rebuiltShare)}% rebuilt · ${plural(all.collapses, 'collapse', 'collapses')}`,
503 '',
504 'Rebuilt: reprocessed tokens the cache had held a request earlier and did not serve. The rest was new content.',
505 '',
506 `Request shape, recorded for ${s.described} of ${plural(s.requests, 'request', 'requests')}:`,
507 ` messages carried: ${Math.round(s.meanMessages)} mean · ${s.peakMessages} peak`,
508 ` tool calls asked for: ${s.toolUses} over ${plural(s.withToolUses, 'request', 'requests')}`,
509 ` stops: ${tallyText(s.stops) || 'none recorded'}`,
510 ` effort: ${tallyText(s.efforts) || 'none recorded'}`,
511 ...spendLines(rows),
512 ...engineLines(rows),
513 '',
514 ...causeNote(rows),
515 ]
516
517 return lines.join('\n')
518}
519
520/**
521 * The fixture module an export writes. `expected` is filled from the CURRENT
522 * logic, so a capture records today's behaviour — review it before trusting
523 * it as a baseline, exactly as any golden file.
524 */
525export const fixtureSource = (f: Fixture): string =>
526 [
527 "import type { Fixture } from '../record'",
528 '',
529 `/** ${f.note} */`,
530 `export const fixture: Fixture = ${JSON.stringify(f, null, 2)}`,
531 '',
532 ].join('\n')
533hooks/cost.ts 815 lines1/**
2 * The cost of rewriting context is set by WHERE the edit lands, not how big it is.
3 *
4 * Lumen's model: a context of T tokens edited at position `a` costs T - a to
5 * re-prefill, because causal attention invalidates every cached key after the
6 * edit. `a` is `first_invalid_token`; a one-word change forty turns back can
7 * cost far more than rewriting a whole paragraph in the latest turn.
8 *
9 * Claude Code reports that same quantity per request, MEASURED, as the four
10 * token counts of `ModelUsage`:
11 *
12 * T = input_tokens + cache_creation_input_tokens + cache_read_input_tokens
13 * a = cache_read_input_tokens (the prefix the cache held)
14 * T - a = input_tokens + cache_creation_input_tokens
15 *
16 * Nothing in this file is an estimate. What it cannot do is say *why* the
17 * prefix was re-sent — see `attribute`.
18 */
19
20/**
21 * Why the model stopped, in `TurnStepResult.stopReason`'s own spelling.
22 * `TurnStopReason` is declared but not exported from 'claude-code', so the
23 * union is mirrored here; `null` is the engine's "no response arrived".
24 */
25export type StopReason =
26 | 'end_turn'
27 | 'max_tokens'
28 | 'stop_sequence'
29 | 'tool_use'
30 | 'pause_turn'
31 | 'compaction'
32 | 'refusal'
33 | 'model_context_window_exceeded'
34 | null
35
36/** How hard the request asked the model to think (`TurnStepInput.effort`). */
37export type Effort = 'low' | 'medium' | 'high' | 'xhigh' | 'max' | number
38
39export type Sample = {
40 turnId: string
41 index: number
42 /** A subagent's loop id; absent on the main thread. */
43 agentId?: string
44 model: string
45 /** `cache_read_input_tokens` — the prefix served from cache. Lumen's `a`. */
46 reused: number
47 /** `cache_creation_input_tokens` — input this request wrote to the cache. */
48 written: number
49 /** `input_tokens` — input neither read from nor written to the cache. */
50 uncached: number
51 output: number
52 at: number
53 /**
54 * When the step BEGAN, before the response streamed. `at` is taken after the
55 * stream drains, so `at - prev.at` carries this step's duration; the interval
56 * the cache cares about runs between two starts. Optional: records from
57 * before it was kept have none, and `gapMs` reads that as unknown, not zero.
58 */
59 startedAt?: number
60 /**
61 * `TurnStepInput.messageCount` — how many messages the request carried.
62 *
63 * Optional, with the three below it: a record written before the harness
64 * grew these fields carries none of them, and a record that does not say
65 * is not a record that says zero. Everything that reads them says which.
66 */
67 messageCount?: number
68 /** `TurnStepResult.stopReason`; `null` when no response arrived. */
69 stopReason?: StopReason
70 /** `TurnStepResult.toolUses.length` — tool calls the response asked for. */
71 toolUseCount?: number
72 /** `TurnStepInput.effort`; absent for a model without an effort setting. */
73 effort?: Effort
74}
75
76export type Totals = {
77 requests: number
78 /**
79 * `cache_read_input_tokens` summed over the requests: what the cache SERVED,
80 * a billing quantity. Not a size — one prefix is counted once per request
81 * that read it — so it is never "held".
82 */
83 reused: number
84 /** Tokens put through the model: `rebuilt + fresh`. */
85 reprocessed: number
86 /** Reprocessed tokens that had been cached and were lost: work redone. */
87 rebuilt: number
88 /** Reprocessed tokens that were never cached: genuinely new content. */
89 fresh: number
90 /**
91 * `rebuilt / reprocessed`; 0 when nothing was reprocessed. Unlike `rate`
92 * it does not climb as a conversation grows, so it compares sessions.
93 */
94 rebuiltShare: number
95 /**
96 * `reused / (reused + reprocessed)` over the whole run. Kept for completeness,
97 * not for display: it rises with conversation length on its own, so it can
98 * neither compare two sessions nor show improvement.
99 */
100 rate: number
101}
102
103/**
104 * Why a prefix was re-sent. `unattributed` is the honest default: a cache read
105 * collapsing means EITHER the prefix changed OR the cache entry lapsed on TTL,
106 * and the token counts cannot tell those apart.
107 */
108export type Cause = 'model-change' | 'compaction' | 'unattributed'
109
110/** What triggered a compaction, in `SessionCompactTrigger`'s spelling. */
111export type CompactTrigger = 'manual' | 'auto' | 'plugin' | 'precompute'
112
113/** The summarizer's own request, in `ModelUsage`'s spelling. */
114export type CompactionUsage = {
115 input_tokens: number
116 output_tokens: number
117 cache_read_input_tokens: number
118 cache_creation_input_tokens: number
119}
120
121/**
122 * One compaction as the engine reported it.
123 *
124 * Only `at` is always known. Every other field is whatever that event actually
125 * supplied: `tokensBefore` and `tokensAfter` are absent when core did not
126 * record them (and `tokensAfter` always on `precompute`), `usage` when a hook
127 * answered in core's place, a summary was reused, or the response reported
128 * none. A field that is absent is unknown, never zero.
129 */
130export type Compaction = {
131 at: number
132 trigger?: CompactTrigger
133 /** The conversation's size before, in tokens, as the engine counted it. */
134 tokensBefore?: number
135 /** Its size afterwards, in tokens. */
136 tokensAfter?: number
137 /** The summarizer's own request, which `turn.step` never sees. */
138 usage?: CompactionUsage
139}
140
141/**
142 * What the store holds: a bare timestamp from every build before the engine's
143 * own figures were kept, a record from every build since. Both stay readable.
144 */
145export type StoredCompaction = number | Compaction
146
147/** When a stored compaction ran. */
148export const compactionAt = (c: StoredCompaction): number => (typeof c === 'number' ? c : c.at)
149
150/** A stored compaction as a record; a bare timestamp carries nothing but its time. */
151export const compactionOf = (c: StoredCompaction): Compaction => (typeof c === 'number' ? { at: c } : c)
152
153/**
154 * Did this event rewrite the transcript? A `precompute` computes a summary to
155 * keep for the compaction that comes and installs nothing, so the prefix of the
156 * next request is untouched by it: it cannot be the cause of a collapse. (The
157 * summarizer's request still happened, so its `usage` is still a cost.)
158 */
159export const rewrites = (c: StoredCompaction): boolean => typeof c === 'number' || c.trigger !== 'precompute'
160
161/** T — the whole input side of one request. Output is not part of the window. */
162export const windowTokens = (s: Sample): number => s.reused + s.written + s.uncached
163
164/** `a` — how far into the context the cache was still valid. */
165export const frontier = (s: Sample): number => s.reused
166
167/** T - a — what this request had to put through the model again. */
168export const reprocessed = (s: Sample): number => s.written + s.uncached
169
170/** `a / T`, in 0..1. A request with no input side reads as 0, not NaN. */
171export const reuseRate = (s: Sample): number => {
172 const t = windowTokens(s)
173 return t === 0 ? 0 : s.reused / t
174}
175
176/** A fork pays for the whole prefix by construction, so it is never a signal. */
177export const isMainThread = (s: Sample): boolean => s.agentId === undefined
178
179/**
180 * The samples worth keeping: the newest `keepMain` main-thread ones and the
181 * newest `keepForks` fork ones, in their original order.
182 *
183 * The two are capped apart so a burst of subagent steps can never push
184 * main-thread history off the end and shrink the session's reported totals.
185 */
186export const retain = (
187 samples: readonly Sample[],
188 keepMain: number,
189 keepForks: number,
190): Sample[] => {
191 let main = samples.filter(isMainThread).length
192 let forks = samples.length - main
193 const kept: Sample[] = []
194
195 for (const s of samples) {
196 if (isMainThread(s)) {
197 main -= 1
198 if (main < keepMain) kept.push(s)
199 } else {
200 forks -= 1
201 if (forks < keepForks) kept.push(s)
202 }
203 }
204
205 return kept
206}
207
208/** Reprocessed tokens that had been cached and were lost: work redone. */
209export const rebuilt = (s: Sample, prev: Sample | undefined): number =>
210 prev === undefined ? 0 : Math.min(lostGround(s, prev), reprocessed(s))
211
212/** Reprocessed tokens that were never cached: genuinely new content. */
213export const fresh = (s: Sample, prev: Sample | undefined): number =>
214 reprocessed(s) - rebuilt(s, prev)
215
216/**
217 * The samples MUST be in order, one thread: each is read against its
218 * predecessor to tell rebuilt work from new content.
219 */
220export const totals = (list: readonly Sample[]): Totals => {
221 let reused = 0
222 let repro = 0
223 let redone = 0
224 for (let i = 0; i < list.length; i += 1) {
225 const prev = i > 0 ? list[i - 1] : undefined
226 reused += frontier(list[i])
227 repro += reprocessed(list[i])
228 redone += rebuilt(list[i], prev)
229 }
230 const t = reused + repro
231 return {
232 requests: list.length,
233 reused,
234 reprocessed: repro,
235 rebuilt: redone,
236 fresh: repro - redone,
237 rebuiltShare: repro === 0 ? 0 : redone / repro,
238 rate: t === 0 ? 0 : reused / t,
239 }
240}
241
242// --- the request's shape: what it carried, beside what it cost ---
243//
244// Every reader here has to tell "not recorded" from "zero", because the oldest
245// records in the store predate these fields. A record that does not say how
246// many tools the step called is not a record of a step that called none.
247
248/** Does this record carry the request shape, or does it predate it? */
249export const isDescribed = (s: Sample): s is Sample & { messageCount: number } =>
250 typeof s.messageCount === 'number' && Number.isFinite(s.messageCount)
251
252/** Tool calls the response asked for; `null` when the record does not say. */
253export const toolUses = (s: Sample): number | null =>
254 typeof s.toolUseCount === 'number' && Number.isFinite(s.toolUseCount) ? s.toolUseCount : null
255
256/** The stop reason as a tally key; `null` when the record does not say. */
257export const stopLabel = (s: Sample): string | null =>
258 s.stopReason === undefined ? null : s.stopReason === null ? 'none' : s.stopReason
259
260/** The effort as a tally key; `null` when the record does not say. */
261export const effortLabel = (s: Sample): string | null =>
262 s.effort === undefined ? null : String(s.effort)
263
264/** A tally of the labels seen, by label. Only labels actually seen appear. */
265export type Tally = Record<string, number>
266
267/**
268 * What the recorded requests looked like. `described` says how many of them
269 * carry the shape at all, so a reading always carries the size of its own
270 * evidence rather than reading silence as zero.
271 */
272export type Shape = {
273 requests: number
274 /** Records carrying a message count — the shape's own coverage. */
275 described: number
276 /** Messages per described request. 0 when none is described, never NaN. */
277 meanMessages: number
278 /** The largest message count seen; 0 when none is recorded. */
279 peakMessages: number
280 /** Records that report a tool-use count. */
281 withToolUses: number
282 /** Tool calls over those records. */
283 toolUses: number
284 stops: Tally
285 efforts: Tally
286}
287
288export const shape = (list: readonly Sample[]): Shape => {
289 let described = 0
290 let messages = 0
291 let peakMessages = 0
292 let withToolUses = 0
293 let calls = 0
294 const stops: Tally = {}
295 const efforts: Tally = {}
296
297 for (const s of list) {
298 if (isDescribed(s)) {
299 described += 1
300 messages += s.messageCount
301 peakMessages = Math.max(peakMessages, s.messageCount)
302 }
303
304 const used = toolUses(s)
305 if (used !== null) {
306 withToolUses += 1
307 calls += used
308 }
309
310 const stop = stopLabel(s)
311 if (stop !== null) stops[stop] = (stops[stop] ?? 0) + 1
312
313 const effort = effortLabel(s)
314 if (effort !== null) efforts[effort] = (efforts[effort] ?? 0) + 1
315 }
316
317 return {
318 requests: list.length,
319 described,
320 meanMessages: described === 0 ? 0 : messages / described,
321 peakMessages,
322 withToolUses,
323 toolUses: calls,
324 stops,
325 efforts,
326 }
327}
328
329/** A tally as text, commonest first: `tool_use 14 · end_turn 5`. */
330export const tallyText = (tally: Tally): string =>
331 Object.entries(tally)
332 .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
333 .map(([label, n]) => `${label} ${n}`)
334 .join(' · ')
335
336/**
337 * What the cache holds after a request: the prefix it served (and so still
338 * holds) plus what the request wrote into it. NOT the window — `uncached` is,
339 * by the engine's definition, neither read from nor written to the cache, so
340 * it was never there to keep or to lose.
341 */
342export const cached = (s: Sample): number => s.reused + s.written
343
344/**
345 * Tokens the cache held after the previous request but did not serve for this
346 * one: the ground the cache lost.
347 *
348 * This is the quantity the pane is about, and it is not the same as a low reuse
349 * rate. A request that APPENDS 20k tokens to a fully cached 12k context reuses
350 * only 37% of its window, but it lost nothing — the 20k was new and had to be
351 * read whatever happened. A request whose prefix was rewritten serves less than
352 * the previous request cached, and that shortfall is what the edit actually cost.
353 *
354 * The baseline is what was cached, not the previous window: input the engine
355 * left uncached was never in the cache, so a request cannot lose it.
356 */
357export const lostGround = (s: Sample, prev: Sample): number =>
358 Math.max(0, cached(prev) - frontier(s))
359
360/**
361 * Lost ground as a share of what the cache held a request earlier. The
362 * denominator is the cache, not the window: the question is how much of the
363 * cached prefix survived, and the loss can never exceed it, so the share stays
364 * in 0..1 and does not shift with how much input happened to go uncached.
365 */
366export const lostShare = (s: Sample, prev: Sample): number => {
367 const before = cached(prev)
368 return before === 0 ? 0 : lostGround(s, prev) / before
369}
370
371/** A prior cache below this has too little in it to lose to read anything into. */
372export const MIN_PRIOR_WINDOW = 5_000
373
374/** Above this share of the prior context lost, the prefix was rewritten. */
375export const LOST_SHARE = 0.25
376
377/**
378 * Did this request pay to re-send a prefix it had already cached?
379 *
380 * False for a subagent step (a fork always pays), false for the first request
381 * (nothing was cached to lose), and false when the prior cache was too small
382 * for its loss to mean anything.
383 */
384export const isCollapse = (s: Sample, prev: Sample | undefined): boolean => {
385 if (!isMainThread(s)) return false
386 if (prev === undefined) return false
387 if (cached(prev) < MIN_PRIOR_WINDOW) return false
388
389 return lostShare(s, prev) > LOST_SHARE
390}
391
392/**
393 * Name the cause only where it is positively known. A compaction rewrites the
394 * whole prefix and the engine says when one ran, so that one is certain;
395 * everything else stays `unattributed`, because a changed prefix and a cache
396 * entry that lapsed on TTL are the same four numbers.
397 *
398 * Every compaction's timestamp is kept, not just the latest. With one
399 * timestamp, a second compaction silently relabelled the first one's collapse
400 * as unattributed on the next render — the verdicts are recomputed each time,
401 * so a stored history is the only thing that keeps an old label true.
402 *
403 * `prev` is required: isCollapse already refuses a request with nothing before
404 * it, so a "first request" cause was unreachable from `view`.
405 */
406export const attribute = (s: Sample, prev: Sample, compactions: readonly StoredCompaction[]): Cause =>
407 // Checked first: the two records name two different models, which needs no
408 // outside clock. It stays a collapse; the whole prefix really was re-sent.
409 s.model !== prev.model
410 ? 'model-change'
411 : compactedBetween(s, prev, compactions)
412 ? 'compaction'
413 : 'unattributed'
414
415/** The compactions that rewrote the transcript between two requests; the window is open at `prev.at`. */
416export const compactionsBetween = (
417 s: Sample,
418 prev: Sample,
419 compactions: readonly StoredCompaction[],
420): Compaction[] =>
421 compactions
422 .filter(c => rewrites(c) && compactionAt(c) > prev.at && compactionAt(c) <= s.at)
423 .map(compactionOf)
424
425const compactedBetween = (s: Sample, prev: Sample, compactions: readonly StoredCompaction[]): boolean =>
426 compactionsBetween(s, prev, compactions).length > 0
427
428/** How a cause reads, wherever a collapse is described: the pane and the review share it. */
429export const CAUSE_TEXT: Record<Cause, string> = {
430 'model-change': 'the model changed',
431 compaction: 'compaction rewrote the prefix',
432 unattributed: 'cause not attributable from token counts',
433}
434
435/** The cause for one collapse, with the records' own detail where it has some. */
436export const causeText = (cause: Cause, s: Sample, prev: Sample): string =>
437 cause === 'model-change' ? `${CAUSE_TEXT[cause]} (${prev.model} → ${s.model})` : CAUSE_TEXT[cause]
438
439/**
440 * The time from the previous request's start to this one's, in ms; `null` when
441 * either record lacks `startedAt` or the two do not run forward. Never zero as
442 * a stand-in: an unknown gap must not read as "no time passed".
443 */
444export const gapMs = (s: Sample, prev: Sample): number | null => {
445 const a = prev.startedAt
446 const b = s.startedAt
447 if (typeof a !== 'number' || typeof b !== 'number') return null
448 if (!Number.isFinite(a) || !Number.isFinite(b)) return null
449
450 return b >= a ? b - a : null
451}
452
453/** A duration for a sentence: `45s`, `12m`, `3.3h`. */
454export const gapText = (ms: number): string => {
455 if (ms < 60_000) return `${Math.floor(ms / 1_000)}s`
456 if (ms < 3_600_000) return `${Math.floor(ms / 60_000)}m`
457
458 return `${(ms / 3_600_000).toFixed(1)}h`
459}
460
461/** Below this many tokens apart, two sizes of one event agree whatever their ratio. */
462export const AGREE_FLOOR = 2_000
463
464/** Above the floor, two sizes agree while they are within this share of the larger. */
465export const AGREE_SHARE = 0.15
466
467/** Do two token sizes of one event agree? Inclusive at the edge. */
468export const sizesAgree = (a: number, b: number): boolean =>
469 Math.abs(a - b) <= Math.max(AGREE_FLOOR, AGREE_SHARE * Math.max(Math.abs(a), Math.abs(b)))
470
471/**
472 * How much the engine says the compaction shrank the conversation:
473 * `tokensBefore - tokensAfter`; null unless it recorded BOTH. Negative when it
474 * reports the conversation grew. Never computed from one side alone.
475 */
476export const engineRemoved = (c: Compaction): number | null =>
477 typeof c.tokensBefore === 'number' &&
478 Number.isFinite(c.tokensBefore) &&
479 typeof c.tokensAfter === 'number' &&
480 Number.isFinite(c.tokensAfter)
481 ? c.tokensBefore - c.tokensAfter
482 : null
483
484/**
485 * THE ORACLE. Two measurements of one compaction, from different sides:
486 *
487 * - `removed` (engine): how much the conversation SHRANK, `tokensBefore -
488 * tokensAfter`.
489 * - `lost` (this mod): how much cached prefix the NEXT request failed to
490 * serve, `lostGround`.
491 *
492 * They are not the same quantity and need not be equal. The relationship the
493 * check ASSUMES is `lost ≈ removed`; the two sources of divergence it allows
494 * for are the engine's sizes being its own counts rather than the API's (a few
495 * percent), and a few hundred tokens of prefix wobble. It is NOT adjusted for
496 * what the next request adds (a new user turn) or must write afresh (the
497 * summary): both land in `written`, which `lostGround` never reads.
498 *
499 * ASSUMPTION NOT YET CHECKED AGAINST A LIVE COMPACTION: what survives a
500 * compaction is only the unchanged head (system prompt, tools); every message
501 * from the first replaced one on is invalid. If so, `lost` tracks the whole
502 * old conversation (`tokensBefore`), not the shrink, and this check disagrees
503 * on every compaction by about `tokensAfter`. That would be this check doing
504 * its job: it is the first outside number the mod has had. The quantity
505 * compared lives only in `engineRemoved`.
506 *
507 * The tolerance is `max(AGREE_FLOOR, AGREE_SHARE of the larger)`. 15%: the
508 * engine's own sizes are estimates that run several percent off the API's, so
509 * anything tighter cries wolf on a correct number; a missing term of the size
510 * of a system prompt (~25k) or a summary (~10-20k) is well over 15% of a
511 * 25-50k compaction, so anything looser hides the error it exists to catch.
512 * The 2k floor keeps a small compaction from disagreeing over a rounding.
513 *
514 * Neither number is preferred and neither is adjusted to match the other.
515 */
516export type EngineCheck = { removed: number; lost: number; agrees: boolean }
517
518export const checkAgainstEngine = (lost: number, c: Compaction): EngineCheck | null => {
519 const removed = engineRemoved(c)
520
521 return removed === null ? null : { removed, lost, agrees: sizesAgree(removed, lost) }
522}
523
524/** The check as a row: both numbers, the engine's own sizes, and whether they agree. */
525export const engineText = (check: EngineCheck, c: Compaction): string => {
526 const said =
527 check.removed < 0
528 ? `engine reported the conversation grew by ${tok(-check.removed)}`
529 : `engine reported ${tok(check.removed)} removed`
530 const sizes = `engine sized the conversation ${tok(c.tokensBefore ?? 0)} → ${tok(c.tokensAfter ?? 0)}`
531 const verdict = check.agrees
532 ? 'agree within tolerance'
533 : `DISAGREE by ${tok(Math.abs(check.removed - check.lost))}, not reconciled`
534
535 return `${said}; measured ${tok(check.lost)} lost — ${verdict} (${sizes})`
536}
537
538/** What a collapse's row says about the engine's own figures. */
539/** `compared` is false where the note says why no comparison was made. */
540export type EngineNote = { text: string; compared: boolean; disagrees: boolean }
541
542/**
543 * Only for a collapse a compaction caused, and only where ONE compaction ran in
544 * the window: with two, one loss belongs to both and no pairing is honest.
545 * Null for any other cause, where there is nothing to check.
546 */
547export const engineNote = (
548 lost: number,
549 cause: Cause,
550 between: readonly Compaction[],
551): EngineNote | null => {
552 if (cause !== 'compaction') return null
553 if (between.length > 1) {
554 return { text: `${between.length} compactions ran in this window; engine sizes not compared`, compared: false, disagrees: false }
555 }
556 const check = checkAgainstEngine(lost, between[0])
557 if (check === null) {
558 return { text: 'engine did not record both sizes — nothing to check against', compared: false, disagrees: false }
559 }
560
561 return { text: engineText(check, between[0]), compared: true, disagrees: !check.agrees }
562}
563
564/**
565 * What a compaction's own request put through the model: `input_tokens +
566 * cache_creation_input_tokens`, the same T - a that `reprocessed` reads for a
567 * step. Cache reads are not reprocessing.
568 */
569export const summarizerReprocessed = (u: CompactionUsage): number =>
570 u.input_tokens + u.cache_creation_input_tokens
571
572/**
573 * The summarizer's request as a sentence. `turn.step` never fires for it, so it
574 * is in no total the pane or the review prints; the sentence says so.
575 */
576export const summarizerText = (u: CompactionUsage): string =>
577 `the summary request itself: ${tok(summarizerReprocessed(u))} reprocessed (${tok(u.cache_read_input_tokens)} served from cache), ${tok(u.output_tokens)} out — not in the reprocessed total`
578
579/**
580 * What compactions cost beyond the requests `turn.step` measured. `withUsage`
581 * is the evidence behind `reprocessed`: a compaction whose usage was never
582 * supplied (a hook answered, a summary was reused, an old bare timestamp) is
583 * counted in `compactions` and contributes nothing it does not know.
584 */
585export type CompactionSpend = {
586 compactions: number
587 withUsage: number
588 reprocessed: number
589 output: number
590}
591
592export const compactionSpend = (list: readonly StoredCompaction[]): CompactionSpend => {
593 let withUsage = 0
594 let repro = 0
595 let output = 0
596 for (const c of list) {
597 const u = compactionOf(c).usage
598 if (u === undefined) continue
599 withUsage += 1
600 repro += summarizerReprocessed(u)
601 output += u.output_tokens
602 }
603
604 return { compactions: list.length, withUsage, reprocessed: repro, output }
605}
606
607/**
608 * The spend as one line, or null when no compaction was recorded. Said as
609 * additional to the session's figures, never part of them.
610 */
611export const spendText = (c: CompactionSpend): string | null => {
612 if (c.compactions === 0) return null
613 const n = `${c.compactions} ${c.compactions === 1 ? 'compaction' : 'compactions'}`
614 if (c.withUsage === 0) return `${n} · summary request usage not recorded`
615
616 return `${n} · ${tok(c.reprocessed)} reprocessed by the summary requests (usage known for ${c.withUsage}), not counted above`
617}
618
619/**
620 * What the records narrow without naming a cause: facts to read beside the
621 * collapse. Nothing here says which whole-prefix event occurred.
622 *
623 * - A hold of exactly 0. An edit at position `a` leaves `a` tokens held, so a
624 * partial edit always holds something; zero leaves only whole-prefix events
625 * (the entry lapsed, the model changed, /clear, a new conversation). Said only
626 * for an unattributed collapse, where it narrows something; a named cause
627 * needs no narrowing. It does not say WHICH event.
628 * - The gap since the previous request, only where both starts were recorded.
629 * Not a cause: the mod cannot read the cache lifetime, so a long gap is shown
630 * and never named.
631 * - A compaction that also ran, when the model change took the cause.
632 * - For a compaction, what its own summarizing request cost, where the engine
633 * reported it: a request `turn.step` never saw, so additional to the totals.
634 */
635export const observations = (
636 s: Sample,
637 prev: Sample,
638 cause: Cause,
639 compactions: readonly StoredCompaction[],
640): string[] => {
641 const notes: string[] = []
642 if (cause === 'unattributed' && frontier(s) === 0) {
643 notes.push('nothing was held — not a partial edit')
644 }
645 if (cause === 'model-change' && compactedBetween(s, prev, compactions)) {
646 notes.push('a compaction also ran in this window')
647 }
648 if (cause === 'compaction') {
649 for (const c of compactionsBetween(s, prev, compactions)) {
650 if (c.usage !== undefined) notes.push(summarizerText(c.usage))
651 }
652 }
653 const gap = gapMs(s, prev)
654 if (gap !== null) notes.push(`${gapText(gap)} since the previous request`)
655
656 return notes
657}
658
659/** A fixed-width meter. Rates outside 0..1 clamp rather than overrun the box. */
660export const bar = (rate: number, width: number): string => {
661 const w = Math.max(1, Math.floor(width))
662 const safe = Number.isFinite(rate) ? Math.max(0, Math.min(1, rate)) : 0
663 const filled = Math.max(0, Math.min(w, Math.round(safe * w)))
664 return '█'.repeat(filled) + '░'.repeat(w - filled)
665}
666
667/** Token counts, short enough for a status line. */
668export const tok = (n: number): string => {
669 if (!Number.isFinite(n)) return '0'
670 const v = Math.max(0, Math.round(n))
671 if (v >= 1_000_000) return `${(v / 1_000_000).toFixed(2)}M`
672 if (v >= 10_000) return `${(v / 1_000).toFixed(1)}k`
673 return String(v)
674}
675
676/** A whole percentage, for display only. */
677export const pct = (rate: number): number =>
678 Number.isFinite(rate) ? Math.round(Math.max(0, Math.min(1, rate)) * 100) : 0
679
680/** The session's own three lines for the pane: how much, how much of it was rebuilt, how much new. */
681export const sessionLines = (t: Totals): [string, string, string] => [
682 `${t.requests} requests · ${tok(t.reprocessed)} reprocessed`,
683 t.reprocessed === 0
684 ? 'nothing reprocessed'
685 : `${pct(t.rebuiltShare)}% of that was rebuilt — work the cache had held and lost`,
686 t.reprocessed === 0 ? '' : `${tok(t.fresh)} was new content, never cached before`,
687]
688
689/** The status line: the number that compares sessions, then what it is a share of. */
690export const statusText = (t: Totals): string =>
691 `cache ${pct(t.rebuiltShare)}% rebuilt · ${tok(t.reprocessed)} reprocessed`
692
693/** One flagged request: what it cost, and the most we can honestly say about why. */
694export type Collapse = {
695 /** Tokens the cache held a request ago and did not serve for this one. */
696 lost: number
697 /** `lost` as a share of what the cache held a request earlier. */
698 share: number
699 reprocessed: number
700 cause: Cause
701 /** The cause as a sentence, with the models named for a model change. */
702 why: string
703 /** What the records narrow beside it; see `observations`. */
704 notes: string[]
705 /** The engine's own sizes against the measured loss, for a compaction; see `engineNote`. */
706 engine: EngineNote | null
707}
708
709/**
710 * Everything the pane decides, decided here. The render hook is then a plain
711 * mapping from this to elements, which keeps every judgement in a function a
712 * test can call directly — the plugin test kit has no `state` noun, so a
713 * mounted pane can only ever be read in the states the plugin starts in.
714 */
715export type View = {
716 isEmpty: boolean
717 last: {
718 rate: number
719 held: number
720 reprocessed: number
721 /** Ground lost against the request before it; 0 when there was none. */
722 lost: number
723 isCollapse: boolean
724 /** Messages the request carried; null when the record predates the field. */
725 messageCount: number | null
726 /** Tool calls the response asked for; null when the record does not say. */
727 toolUses: number | null
728 } | null
729 session: Totals
730 /** Null when no subagent ran; never folded into `session`. */
731 forks: Totals | null
732 /** What the summary requests cost; additional to `session`, never folded into it. */
733 compaction: CompactionSpend
734 /** Most recent first, at most `SHOW_COLLAPSES`. */
735 collapses: Collapse[]
736}
737
738export const SHOW_COLLAPSES = 3
739
740/**
741 * Every flagged request, oldest first. The pane shows the tail of this; a
742 * review across sessions counts the whole of it.
743 */
744export const allCollapses = (
745 all: readonly Sample[],
746 compactions: readonly StoredCompaction[],
747): Collapse[] => {
748 const main = all.filter(isMainThread)
749 const flagged: Collapse[] = []
750
751 for (let i = 0; i < main.length; i += 1) {
752 const s = main[i]
753 const prev = i > 0 ? main[i - 1] : undefined
754 if (isCollapse(s, prev) && prev !== undefined) {
755 const cause = attribute(s, prev, compactions)
756 flagged.push({
757 lost: lostGround(s, prev),
758 share: lostShare(s, prev),
759 reprocessed: reprocessed(s),
760 cause,
761 why: causeText(cause, s, prev),
762 notes: observations(s, prev, cause, compactions),
763 engine: engineNote(lostGround(s, prev), cause, compactionsBetween(s, prev, compactions)),
764 })
765 }
766 }
767
768 return flagged
769}
770
771/**
772 * The last request's shape as one line, or null when that record predates the
773 * fields. A step that called no tools reports none; a step that never said
774 * reports nothing at all.
775 */
776export const shapeText = (last: {
777 messageCount: number | null
778 toolUses: number | null
779}): string | null => {
780 const parts: string[] = []
781 if (last.messageCount !== null) parts.push(`${last.messageCount} messages`)
782 if (last.toolUses !== null) parts.push(`${last.toolUses} tool calls`)
783
784 return parts.length === 0 ? null : parts.join(' · ')
785}
786
787export const view = (all: readonly Sample[], compactions: readonly StoredCompaction[]): View => {
788 const main = all.filter(isMainThread)
789 const forked = all.filter(s => !isMainThread(s))
790 const flagged = allCollapses(all, compactions)
791
792 const last = main.length > 0 ? main[main.length - 1] : undefined
793 const beforeLast = main.length > 1 ? main[main.length - 2] : undefined
794
795 return {
796 isEmpty: main.length === 0,
797 last:
798 last === undefined
799 ? null
800 : {
801 rate: reuseRate(last),
802 held: frontier(last),
803 reprocessed: reprocessed(last),
804 lost: beforeLast === undefined ? 0 : lostGround(last, beforeLast),
805 isCollapse: isCollapse(last, beforeLast),
806 messageCount: isDescribed(last) ? last.messageCount : null,
807 toolUses: toolUses(last),
808 },
809 session: totals(main),
810 forks: forked.length === 0 ? null : totals(forked),
811 compaction: compactionSpend(compactions),
812 collapses: flagged.slice(-SHOW_COLLAPSES).reverse(),
813 }
814}
815types/index.d.ts 73 lines1/** `TurnStepResult.stopReason`'s union, which 'claude-code' does not export. */
2export type CacheStopReason =
3 | 'end_turn'
4 | 'max_tokens'
5 | 'stop_sequence'
6 | 'tool_use'
7 | 'pause_turn'
8 | 'compaction'
9 | 'refusal'
10 | 'model_context_window_exceeded'
11 | null
12
13/** `TurnStepInput.effort`. */
14export type CacheEffort = 'low' | 'medium' | 'high' | 'xhigh' | 'max' | number
15
16/**
17 * One measured request: what it cost, and what the request itself looked like.
18 *
19 * The four fields below `at` arrived after the first version of this mod, so
20 * every one of them is optional: the state and the cross-session store both
21 * hold records written without them, and a record that does not say is never
22 * read as a record that says zero. This mirrors `Sample` in hooks/cost.ts.
23 */
24export type CacheSample = {
25 turnId: string
26 index: number
27 agentId?: string
28 model: string
29 reused: number
30 written: number
31 uncached: number
32 output: number
33 at: number
34 /** When the step began, before it streamed; absent on records from before it was kept. */
35 startedAt?: number
36 /** `TurnStepInput.messageCount` — messages the request carried. */
37 messageCount?: number
38 /** Why the model stopped; `null` when no response arrived. */
39 stopReason?: CacheStopReason
40 /** `TurnStepResult.toolUses.length` — tool calls the response asked for. */
41 toolUseCount?: number
42 /** Absent for a model without an effort setting. */
43 effort?: CacheEffort
44}
45
46/**
47 * One compaction. Only `at` is always known; the rest is whatever the event
48 * supplied, and an absent field is unknown, never zero. Mirrors `Compaction`
49 * in hooks/cost.ts.
50 */
51export type CacheCompaction = {
52 at: number
53 trigger?: 'manual' | 'auto' | 'plugin' | 'precompute'
54 tokensBefore?: number
55 tokensAfter?: number
56 usage?: {
57 input_tokens: number
58 output_tokens: number
59 cache_read_input_tokens: number
60 cache_creation_input_tokens: number
61 }
62}
63
64declare module 'claude-code' {
65 interface PluginState {
66 'lumen-cache': {
67 samples: CacheSample[]
68 /** A bare number is a record from before the engine's figures were kept. */
69 compactions: (number | CacheCompaction)[]
70 }
71 }
72}
73