Measures how $.model.fork behaves as a prompt-cache keep-alive, to settle the numbers cost-ledger's keep-alive will rely on. Not shipped.

A throwaway mod that measures how $.model.fork behaves as a prompt-cache keep-alive, before cost-ledger's keep-alive is built on it. It logs every request's token counts, Claude Code's running cost and the rate-limit readings, and forks the transcript on demand or on a timer while the session is idle.
It lives outside plugins/ so it is never published, and nothing in the marketplace lists it.
cache_read close to the context size)?$.session.usage().cost)? If it is, cost-ledger must not add it a second time.fork takes no max_tokens, and the session's thinking setting applies.Load it next to cost-ledger for one session:
claude --plugin-dir tests/cost-ledger-probe
Every record goes to ~/claude-costs/probe-<date>-<session>.jsonl (%USERPROFILE% on Windows) as it happens. A reload or restart of the same session keeps adding to the same file. Timers still pending are lost on a reload.
| Command | Does |
|---|---|
/probe ping | Forks now |
/probe schedule 4 8 12 | Forks at those minutes after the last turn ended |
/probe burst 20 30 | 20 forks, each started 30 s after the previous one returned |
/probe mark <text> | Writes a label into the log, such as control start |
/probe report | Summarises forks, the first request of each turn, rate limits and cache lifetime |
A prompt you send cancels any pings still pending, and a ping is skipped (and logged as skipped) while a turn is running.
Before each one, build a context of at least 50k tokens, for example by asking Claude to read a few large files, so a hit and a miss differ by a clear margin. Use /probe mark before each run so the log is easy to split afterwards. Don't run other Claude sessions during experiment 5.
| # | Steps | Read in the report |
|---|---|---|
| 1, 4 | End a turn, wait 2 min, /probe ping. Repeat at your usual effort and at a lower one | Fork read close to ctx (above 90%); fork output |
| 2 | Test: end a turn, /probe schedule 4 8 12, send a one-word prompt 15 min after the turn ended. Control: the same with no schedule | First request of the follow-up turn: the test shows a large read, the control a large write |
| 3 | /probe ping with no turn in between | Fork Δcost against its tokens priced at list rates: about equal means Claude Code counts the fork; zero means cost-ledger has to |
| 5 | /probe report to note the limits, /probe burst 20 30, then a one-word prompt and /probe report again. Then the same wait and prompt without the burst | The difference between the two moves, divided by 20 |
For experiment 2, check the reported cache lifetime first. If it is 1h, use /probe schedule 55 110 with the prompt at 115 minutes; the 15-minute version proves nothing on a one-hour cache.
The rate-limit percentages come with one decimal place and are only what the last response reported. That is why experiment 5 uses a burst and takes its readings after a real prompt.
Bring the JSONL file back when you're done. It holds everything the report leaves out.
claude plugin validate tests/cost-ledger-probe
claude plugin test tests/cost-ledger-probehooks/register.ts 273 lines1import type { EngineInterface, Register } from 'claude-code'
2
3import type { ProbeRecord, Snapshot, Tokens } from './probe'
4import { contextOf, fromJsonl, logFileName, parseCommand, PING_PROMPT, report, scheduleDelays, toJsonl, USAGE } from './probe'
5
6type Usage = {
7 input_tokens: number
8 output_tokens: number
9 cache_read_input_tokens: number
10 cache_creation_input_tokens: number
11}
12
13const tokensOf = (u: Usage): Tokens => ({
14 input: u.input_tokens,
15 output: u.output_tokens,
16 cacheRead: u.cache_read_input_tokens,
17 cacheWrite: u.cache_creation_input_tokens,
18})
19
20// Module state starts over on a reload; the log itself is read back from its
21// file, so a reload mid-experiment loses only pending timers.
22const state = {
23 records: [] as ProbeRecord[],
24 logPath: undefined as string | undefined,
25 runningTurn: null as string | null,
26 lastTurnEnd: undefined as number | undefined,
27 contextTokens: null as number | null,
28 lastMeasured: '',
29 pending: [] as { cancel: () => void }[],
30 queue: Promise.resolve() as Promise<void>,
31}
32
33function serially(task: () => Promise<void>) {
34 const run = state.queue.then(task)
35 state.queue = run.catch(() => undefined)
36 return run
37}
38
39async function writeLog($: EngineInterface) {
40 const path = state.logPath
41 if (path === undefined) return
42 try {
43 await $.fs.write(path, toJsonl(state.records))
44 } catch (err) {
45 $.ui.toast(`cost-ledger-probe: could not write ${path}: ${String(err)}`)
46 }
47}
48
49function add($: EngineInterface, record: ProbeRecord) {
50 return serially(async () => {
51 state.records.push(record)
52 await writeLog($)
53 })
54}
55
56async function snapshot($: EngineInterface): Promise<Snapshot> {
57 const usage = await $.session.usage()
58 return { costUsd: usage.cost?.usd ?? null, rateLimits: usage.rateLimits.map(l => ({ ...l })) }
59}
60
61async function ping($: EngineInterface, label: string) {
62 if (state.runningTurn !== null) {
63 await add($, { at: await $.clock.now(), kind: 'skip', label, why: 'a turn is running' })
64 return
65 }
66 const before = await snapshot($)
67 const startedAt = await $.clock.now()
68 const result = await $.model.fork({ prompt: PING_PROMPT })
69 const after = await snapshot($)
70 const at = await $.clock.now()
71 const common = { at, kind: 'fork' as const, label, startedAt, contextTokens: state.contextTokens, before, after }
72 let record: ProbeRecord
73 if (result.isAnswered) record = { ...common, result: 'answered', tokens: tokensOf(result.usage) }
74 else if (result.reason === 'nothing-to-fork') record = { ...common, result: result.reason }
75 else if (result.reason === 'api-error')
76 record = { ...common, result: `api-error ${result.error}`, status: result.status, tokens: tokensOf(result.usage) }
77 else record = { ...common, result: result.reason, tokens: tokensOf(result.usage) }
78 await add($, record)
79 const read = record.tokens?.cacheRead ?? 0
80 const ctx = state.contextTokens ? ` of ~${state.contextTokens}` : ''
81 $.ui.status(`probe ${label}: ${record.result}, cache read ${read}${ctx}`)
82}
83
84// A timer's ping has no caller to reject to: a fork the engine refuses is
85// logged as a skip instead of surfacing as an unhandled rejection.
86async function pingFromTimer($: EngineInterface, label: string) {
87 try {
88 await ping($, label)
89 } catch (err) {
90 await add($, { at: await $.clock.now(), kind: 'skip', label, why: String(err) })
91 }
92}
93
94// A pending timer leaves the list as it fires, so a prompt cancels, and
95// reports, only the pings still to come.
96function later($: EngineInterface, ms: number, run: () => Promise<void>) {
97 const timer = $.clock.after(ms, () => {
98 state.pending = state.pending.filter(t => t !== timer)
99 void run()
100 })
101 state.pending.push(timer)
102}
103
104// Each fork of a burst starts only after the previous one returned, so slow
105// replies stretch the spacing rather than overlapping.
106async function burst($: EngineInterface, i: number, count: number, seconds: number) {
107 await pingFromTimer($, `burst ${i}/${count}`)
108 if (i < count && state.runningTurn === null) {
109 later($, seconds * 1000, () => burst($, i + 1, count, seconds))
110 }
111}
112
113function cancelPending() {
114 const n = state.pending.length
115 for (const timer of state.pending) timer.cancel()
116 state.pending = []
117 return n
118}
119
120export const register: Register = on => {
121 Object.assign(state, {
122 records: [],
123 logPath: undefined,
124 runningTurn: null,
125 lastTurnEnd: undefined,
126 contextTokens: null,
127 lastMeasured: '',
128 pending: [],
129 queue: Promise.resolve(),
130 })
131
132 on('session.start', async ($, e, next) => {
133 await $.command.register({
134 name: 'probe',
135 description: 'cost-ledger-probe: ping (fork now), schedule <minutes...>, burst <count> <seconds>, mark <text>, report',
136 argumentHint: 'ping | schedule <min...> | burst <n> <s> | mark <text> | report',
137 })
138 const home = (await $.env.get('USERPROFILE')) ?? (await $.env.get('HOME'))
139 const sessionId = await $.session.id()
140 const now = await $.clock.now()
141 if (home !== undefined) {
142 const path = `${home}/claude-costs/${logFileName(now, sessionId)}`
143 state.logPath = path
144 if (await $.fs.exists(path)) state.records = fromJsonl(String(await $.fs.read(path)))
145 } else {
146 $.ui.toast('cost-ledger-probe: no home folder found; records are kept in memory only')
147 }
148 await add($, { at: now, kind: 'session', sessionId, version: (await $.session.version()).version })
149 return next(e)
150 })
151
152 on('classic.SessionStart', async ($, e, next) => {
153 if (e.source === 'resume' || e.source === 'fork' || e.source === 'clear' || e.source === 'compact') {
154 await add($, {
155 at: await $.clock.now(),
156 kind: 'resume',
157 source: e.source,
158 secondsSinceLastResponse: e.seconds_since_last_response,
159 contextTokens: e.context_tokens,
160 likelyExpired: e.prompt_cache_likely_expired,
161 estimatedCacheWriteUsd: e.estimated_cache_write_usd,
162 })
163 }
164 if (e.source === 'clear' || e.source === 'compact') state.contextTokens = null
165 else if (e.context_tokens !== undefined) state.contextTokens = e.context_tokens
166 return next(e)
167 })
168
169 on('classic.PreModelSwitch', async ($, e, next) => {
170 await add($, { at: await $.clock.now(), kind: 'ttl', source: 'PreModelSwitch', ttl: e.cache_ttl })
171 return next(e)
172 })
173
174 on('classic.PostModelSwitch', async ($, e, next) => {
175 await add($, { at: await $.clock.now(), kind: 'ttl', source: 'PostModelSwitch', ttl: e.cache_ttl })
176 return next(e)
177 })
178
179 on('turn.start', async ($, e, next) => {
180 state.runningTurn = e.turnId
181 // A real prompt ends whatever schedule or burst was waiting: the gap it was
182 // meant to measure no longer exists.
183 const cancelled = cancelPending()
184 const at = await $.clock.now()
185 if (cancelled > 0) await add($, { at, kind: 'mark', text: `turn started; ${cancelled} pending ping(s) cancelled` })
186 await add($, { at, kind: 'turn', phase: 'start', turnId: e.turnId })
187 return next(e)
188 })
189
190 on('turn.step', async function* ($, e, next) {
191 const startedAt = await $.clock.now()
192 const result = yield* next(e)
193 const usage = result.usage
194 if (usage) {
195 const tokens = tokensOf(usage)
196 if (e.agentId === undefined) state.contextTokens = contextOf(tokens)
197 await add($, {
198 at: await $.clock.now(),
199 kind: 'step',
200 turnId: e.turnId,
201 agentId: e.agentId,
202 model: usage.model,
203 startedAt,
204 tokens,
205 })
206 }
207 return result
208 })
209
210 on('turn.complete', async ($, e, next) => {
211 const result = await next(e)
212 if (e.agentId === undefined) {
213 state.runningTurn = null
214 const at = await $.clock.now()
215 state.lastTurnEnd = at
216 await add($, { at, kind: 'turn', phase: 'end', turnId: e.turnId })
217 }
218 return result
219 })
220
221 on('session.measure', async ($, e, next) => {
222 const snap: Snapshot = { costUsd: e.cost?.usd ?? null, rateLimits: e.rateLimits.map(l => ({ ...l })) }
223 const key = JSON.stringify(snap)
224 if (key !== state.lastMeasured) {
225 state.lastMeasured = key
226 await add($, { at: await $.clock.now(), kind: 'measure', snapshot: snap })
227 }
228 return next(e)
229 })
230
231 on('command.run', { command: 'probe' }, async ($, e) => {
232 const command = parseCommand(e.args)
233 const now = await $.clock.now()
234 switch (command.kind) {
235 case 'help':
236 return { text: `${command.error ? `${command.error}\n` : ''}${USAGE}` }
237 case 'report':
238 return { text: `${report(state.records)}\n\nLog: ${state.logPath ?? '(memory only)'}` }
239 case 'mark':
240 await add($, { at: now, kind: 'mark', text: command.text })
241 return { text: `Marked: ${command.text}` }
242 case 'ping':
243 // Not awaited: the command returns at once and the fork's record lands
244 // when it finishes, as a scheduled ping's would.
245 void pingFromTimer($, 'ping')
246 return { text: 'Forking now; /probe report shows the result.' }
247 case 'schedule': {
248 const anchor = state.lastTurnEnd
249 if (anchor === undefined) return { text: 'No turn has finished in this session yet; send a prompt first.' }
250 cancelPending()
251 const lines: string[] = []
252 for (const { minutes, delayMs } of scheduleDelays(anchor, now, command.minutes)) {
253 if (delayMs <= 0) {
254 lines.push(` +${minutes}m: already past, skipped`)
255 continue
256 }
257 later($, delayMs, () => pingFromTimer($, `+${minutes}m`))
258 lines.push(` +${minutes}m: in ${(delayMs / 60_000).toFixed(1)} min`)
259 }
260 await add($, { at: now, kind: 'mark', text: `schedule ${command.minutes.join(' ')} after turn end` })
261 return { text: `Pings after the last turn's end:\n${lines.join('\n')}\nA new prompt cancels any still pending.` }
262 }
263 case 'burst': {
264 cancelPending()
265 const { count, seconds } = command
266 await add($, { at: now, kind: 'mark', text: `burst ${count} x ${seconds}s` })
267 void burst($, 1, count, seconds)
268 return { text: `Burst of ${count} pings, ${seconds}s apart, started. A new prompt stops it.` }
269 }
270 }
271 })
272}
273hooks/probe.ts 202 lines1// The probe's records and the arithmetic over them, free of engine calls so the
2// tests can check it directly. Each record is one line of the JSONL log.
3
4export type Limit = { kind: string; percentUsed: number; resetsAt?: string }
5
6export type Snapshot = { costUsd: number | null; rateLimits: Limit[] }
7
8export type Tokens = { input: number; output: number; cacheRead: number; cacheWrite: number }
9
10export type ProbeRecord = { at: number } & (
11 | { kind: 'session'; sessionId: string; version: string }
12 | { kind: 'turn'; phase: 'start' | 'end'; turnId: string }
13 | { kind: 'step'; turnId: string; agentId?: string; model: string; startedAt: number; tokens: Tokens }
14 | { kind: 'measure'; snapshot: Snapshot }
15 | {
16 kind: 'fork'
17 label: string
18 startedAt: number
19 contextTokens: number | null
20 result: 'answered' | string
21 status?: number | null
22 tokens?: Tokens
23 before: Snapshot
24 after: Snapshot
25 }
26 | { kind: 'skip'; label: string; why: string }
27 | { kind: 'ttl'; source: string; ttl: string }
28 | { kind: 'resume'; source: string; secondsSinceLastResponse?: number; contextTokens?: number; likelyExpired?: boolean; estimatedCacheWriteUsd?: number }
29 | { kind: 'mark'; text: string }
30)
31
32// Short, tool-free and asking for one word, so each ping's output stays small;
33// the output tokens it still costs are one of the things being measured.
34export const PING_PROMPT =
35 'This is an automated prompt-cache keep-alive check, not a request from the user. ' +
36 'Do not use any tools. Reply with exactly: ok'
37
38export type Command =
39 | { kind: 'ping' }
40 | { kind: 'schedule'; minutes: number[] }
41 | { kind: 'burst'; count: number; seconds: number }
42 | { kind: 'mark'; text: string }
43 | { kind: 'report' }
44 | { kind: 'help'; error?: string }
45
46export const USAGE =
47 'Usage: /probe ping | schedule <minutes...> | burst <count> <seconds> | mark <text> | report'
48
49const isPositive = (n: number) => Number.isFinite(n) && n > 0
50
51export function parseCommand(args: string): Command {
52 const [verb = '', ...rest] = args.trim().split(/\s+/).filter(Boolean)
53 switch (verb) {
54 case 'ping':
55 return { kind: 'ping' }
56 case 'schedule': {
57 const minutes = rest.map(Number)
58 if (minutes.length === 0 || !minutes.every(isPositive)) {
59 return { kind: 'help', error: 'schedule takes one or more positive minute offsets, e.g. schedule 4 8 12' }
60 }
61 return { kind: 'schedule', minutes: [...minutes].sort((a, b) => a - b) }
62 }
63 case 'burst': {
64 const [count, seconds] = rest.map(Number)
65 if (count === undefined || seconds === undefined || !Number.isInteger(count) || !isPositive(count) || !isPositive(seconds)) {
66 return { kind: 'help', error: 'burst takes a whole count and a spacing in seconds, e.g. burst 20 30' }
67 }
68 return { kind: 'burst', count, seconds }
69 }
70 case 'mark': {
71 const text = rest.join(' ')
72 return text === '' ? { kind: 'help', error: 'mark takes a label, e.g. mark control start' } : { kind: 'mark', text }
73 }
74 case 'report':
75 case '':
76 return { kind: 'report' }
77 default:
78 return { kind: 'help', error: `unknown subcommand "${verb}"` }
79 }
80}
81
82// The context the next request sends, counted as cost-ledger counts it: the
83// last response's input, cache read, cache write and output together.
84export const contextOf = (t: Tokens) => t.input + t.cacheRead + t.cacheWrite + t.output
85
86export function dayKey(ms: number): string {
87 const d = new Date(ms)
88 const pad = (n: number) => String(n).padStart(2, '0')
89 return `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())}`
90}
91
92export const logFileName = (ms: number, sessionId: string) => `probe-${dayKey(ms)}-${sessionId.slice(0, 8)}.jsonl`
93
94export const toJsonl = (records: readonly ProbeRecord[]) => records.map(r => JSON.stringify(r)).join('\n') + '\n'
95
96export function fromJsonl(text: string): ProbeRecord[] {
97 const records: ProbeRecord[] = []
98 for (const line of text.split('\n')) {
99 if (line.trim() === '') continue
100 try {
101 records.push(JSON.parse(line) as ProbeRecord)
102 } catch {
103 // a line cut short by a crash mid-write is dropped, not fatal
104 }
105 }
106 return records
107}
108
109// Delays for `schedule`: each offset counts from the end of the last main
110// turn. Offsets already past are reported rather than fired late, since a late
111// ping would measure a different idle gap from the one asked for.
112export function scheduleDelays(anchor: number, now: number, minutes: readonly number[]) {
113 return minutes.map(m => ({ minutes: m, delayMs: anchor + m * 60_000 - now }))
114}
115
116const time = (ms: number) => new Date(ms).toTimeString().slice(0, 8)
117const minutes = (ms: number) => `${(ms / 60_000).toFixed(1)}m`
118const pct = (part: number, whole: number | null) => (whole ? `${Math.round((part / whole) * 100)}%` : '?')
119const usd = (n: number | null) => (n === null ? '?' : `$${n.toFixed(4)}`)
120
121function costDelta(before: Snapshot, after: Snapshot): string {
122 if (before.costUsd === null || after.costUsd === null) return '?'
123 return usd(after.costUsd - before.costUsd)
124}
125
126function limitsText(limits: readonly Limit[]): string {
127 return limits.length === 0 ? 'none reported' : limits.map(l => `${l.kind} ${l.percentUsed}%`).join(', ')
128}
129
130// What the five Phase 0 questions read from the log: each fork's cache hit and
131// cost, the first request of each turn after an idle gap, and how the
132// rate-limit readings moved. The raw JSONL holds everything else.
133export function report(records: readonly ProbeRecord[]): string {
134 const lines: string[] = []
135 let lastRequest: number | undefined
136 let lastTurnStart: number | undefined
137 let awaitingFirstStep = false
138 const forks: string[] = []
139 const firstSteps: string[] = []
140 const limits: Limit[][] = []
141 const ttls = new Set<string>()
142 const marks: string[] = []
143
144 for (const r of records) {
145 switch (r.kind) {
146 case 'turn':
147 if (r.phase === 'start') {
148 lastTurnStart = r.at
149 awaitingFirstStep = true
150 }
151 break
152 case 'step': {
153 if (r.agentId === undefined && awaitingFirstStep && lastTurnStart !== undefined) {
154 const idle = lastRequest === undefined ? '-' : minutes(r.startedAt - lastRequest)
155 firstSteps.push(
156 ` ${time(r.startedAt)} idle ${idle.padStart(6)} read ${String(r.tokens.cacheRead).padStart(8)}` +
157 ` write ${String(r.tokens.cacheWrite).padStart(8)} input ${r.tokens.input} ${r.model}`,
158 )
159 awaitingFirstStep = false
160 }
161 if (r.agentId === undefined) lastRequest = r.startedAt
162 break
163 }
164 case 'fork': {
165 const idle = lastRequest === undefined ? '-' : minutes(r.startedAt - lastRequest)
166 const t = r.tokens
167 forks.push(
168 t
169 ? ` ${time(r.startedAt)} ${r.label.padEnd(10)} idle ${idle.padStart(6)} ctx ${String(r.contextTokens ?? '?').padStart(8)}` +
170 ` read ${String(t.cacheRead).padStart(8)} (${pct(t.cacheRead, r.contextTokens)}) write ${t.cacheWrite}` +
171 ` input ${t.input} output ${t.output} Δcost ${costDelta(r.before, r.after)} ${r.result}`
172 : ` ${time(r.startedAt)} ${r.label.padEnd(10)} ${r.result}${r.status ? ` (${r.status})` : ''}`,
173 )
174 if (t) lastRequest = r.startedAt
175 if (r.after.rateLimits.length > 0) limits.push(r.after.rateLimits)
176 break
177 }
178 case 'measure':
179 if (r.snapshot.rateLimits.length > 0) limits.push(r.snapshot.rateLimits)
180 break
181 case 'ttl':
182 ttls.add(`${r.ttl} (${r.source})`)
183 break
184 case 'mark':
185 marks.push(` ${time(r.at)} ${r.text}`)
186 break
187 case 'skip':
188 forks.push(` ${time(r.at)} ${r.label.padEnd(10)} skipped: ${r.why}`)
189 break
190 }
191 }
192
193 lines.push(`Forks (${forks.length})`, ...(forks.length ? forks : [' none yet']))
194 lines.push('', 'First request of each turn', ...(firstSteps.length ? firstSteps : [' none yet']))
195 const first = limits[0]
196 const last = limits[limits.length - 1]
197 lines.push('', `Rate limits: first ${first ? limitsText(first) : 'none reported'}; last ${last ? limitsText(last) : 'none reported'}`)
198 lines.push(`Cache lifetime reported: ${ttls.size ? [...ttls].join(', ') : 'not yet'}`)
199 if (marks.length) lines.push('', 'Marks', ...marks)
200 return lines.join('\n')
201}
202