Keeps a ledger of what output-trimmer, subagent-router and auto-effort save, shown by /savings

A Claude Code mod that keeps a ledger of what the other token-saving mods in this collection save. Run /savings to see it:
Since 2026-10-08, at API prices:
output-trimmer: 1.2M input tokens not re-read, about $0.24
subagent-router: 340k tokens on a cheaper model, about $1.10
auto-effort on: 42 prompts, $0.31 and 2k output tokens per prompt, 0.40% of the 5-hour window per prompt
auto-effort off: 9 prompts, $0.44 and 3k output tokens per prompt, 0.55% of the 5-hour window per prompt
Prompts differ, so the auto-effort rows only mean something after a few days of each.
/auto-effort off for a few days of ordinary work now and then.Dollar figures use Anthropic's API prices. On a subscription you don't pay per token, so the window share per prompt is the closer figure to what you use up.
Two of the prices are guesses: Haiku cache reads at a tenth of its input price, and cache writes at twice the input price, the rate for the 1-hour cache.
The ledger is kept between sessions. If two sessions run at once, the last one to save wins.
/savings shows the ledger/savings reset clears it and starts a new one from todayInstall it with the mods it measures:
/plugin install savings-meter --marketplace noash-xrc/claude-tools
Answer y to add the marketplace, then pick a scope.
claude plugin marketplace update noash-tools
claude plugin update savings-meter@noash-tools
Restart Claude Code to load the new version.
MIT
hooks/register.ts 128 lines1import { atom, read } from 'claude-code'
2import type { EngineInterface, Register, TurnStepResult } from 'claude-code'
3
4type Usage = NonNullable<TurnStepResult['usage']>
5type Group = { prompts: number; output: number; usd: number; window: number }
6type Ledger = {
7 since: string
8 trimmer: { tokens: number; usd: number }
9 router: { tokens: number; usd: number }
10 effort: Record<'on' | 'off', Group>
11}
12
13// API dollars per million tokens: input, output, cache read.
14// ponytail: the Haiku cache read is a guess at a tenth of its input price, and cache writes take the 1-hour rate of twice the input price
15const PRICES: Record<string, [number, number, number]> = {
16 opus: [4, 20, 0.2],
17 sonnet: [2, 10, 0.2],
18 haiku: [0.1, 0.5, 0.01],
19}
20const price = (model: string) => Object.entries(PRICES).find(([family]) => model.includes(family))?.[1]
21const cost = (model: string, u: Usage) => {
22 const p = price(model)
23 if (!p) return undefined
24 const [input, output, cached] = p
25 return (u.input_tokens * input + u.cache_creation_input_tokens * input * 2 + u.cache_read_input_tokens * cached + u.output_tokens * output) / 1e6
26}
27
28const effortAtom = atom({ plugin: 'auto-effort', key: 'effort' } as const, null)
29const trimmedAtom = atom({ plugin: 'output-trimmer', key: 'trimmed' } as const, 0)
30const routedAtom = atom({ plugin: 'subagent-router', key: 'routed' } as const, [])
31
32const group = (): Group => ({ prompts: 0, output: 0, usd: 0, window: 0 })
33const fresh = (since: string): Ledger => ({ since, trimmer: { tokens: 0, usd: 0 }, router: { tokens: 0, usd: 0 }, effort: { on: group(), off: group() } })
34
35let ledger: Ledger | undefined
36let mainModel = ''
37// trimmed tokens already dropped from the context by a compaction
38let trimmedBase = 0
39let window: number | undefined
40let mode: 'on' | 'off' = 'off'
41
42// ponytail: two sessions at once each write their own copy, so the last one to save wins
43const load = async ($: EngineInterface) =>
44 (ledger ??= ((await $.store.get('ledger')) as Ledger | undefined) ?? fresh(new Date(await $.clock.now()).toISOString().slice(0, 10)))
45const save = ($: EngineInterface) => $.store.set('ledger', ledger)
46
47const k = (n: number) => (n >= 1e6 ? `${(n / 1e6).toFixed(1)}M` : n >= 1000 ? `${Math.round(n / 1000)}k` : String(Math.round(n)))
48const usd = (n: number) => `$${n.toFixed(2)}`
49const row = (name: string, g: Group) =>
50 g.prompts === 0
51 ? `${name}: no prompts yet`
52 : `${name}: ${g.prompts} prompts, ${usd(g.usd / g.prompts)} and ${k(g.output / g.prompts)} output tokens per prompt, ${(g.window / g.prompts).toFixed(2)}% of the 5-hour window per prompt`
53
54export const register: Register = on => {
55 on('session.start', async ($, e, next) => {
56 await $.command.register({ name: 'savings', description: 'Show what the token-saving mods saved, or reset the ledger', argumentHint: '[reset]' })
57 await load($)
58 return next(e)
59 })
60
61 on('command.run', { command: 'savings' }, async ($, e) => {
62 const l = await load($)
63 if (e.args.trim() === 'reset') {
64 ledger = fresh(new Date(await $.clock.now()).toISOString().slice(0, 10))
65 await save($)
66 return { text: 'Savings ledger cleared.' }
67 }
68 return {
69 text: [
70 `Since ${l.since}, at API prices:`,
71 `output-trimmer: ${k(l.trimmer.tokens)} input tokens not re-read, about ${usd(l.trimmer.usd)}`,
72 `subagent-router: ${k(l.router.tokens)} tokens on a cheaper model, about ${usd(l.router.usd)}`,
73 row('auto-effort on', l.effort.on),
74 row('auto-effort off', l.effort.off),
75 'Prompts differ, so the auto-effort rows only mean something after a few days of each.',
76 ].join('\n'),
77 }
78 })
79
80 on('prompt.submit', async ($, e, next) => {
81 if (e.turnId === undefined && !e.text.startsWith('/')) (await load($)).effort[mode].prompts++
82 return next(e)
83 }).catch(($, e, next) => next(e))
84
85 on('session.measure', async ($, e, next) => {
86 const now = e.rateLimits.find(r => r.kind === 'five_hour')?.percentUsed
87 // a drop means the window reset, which saved nothing
88 if (now !== undefined && window !== undefined && now > window) (await load($)).effort[mode].window += now - window
89 if (now !== undefined) window = now
90 return next(e)
91 })
92
93 on('session.compact', async ($, e, next) => {
94 const result = await next(e)
95 if (e.agentId === undefined) trimmedBase = await read($, trimmedAtom)
96 return result
97 })
98
99 on('turn.step', async function* ($, e, next) {
100 const result = yield* next(e)
101 const u = result.usage
102 if (!u) return result
103 const l = await load($)
104 const model = u.model ?? e.model
105
106 if (e.agentId === undefined) {
107 mainModel = model
108 mode = (await read($, effortAtom)) === null ? 'off' : 'on'
109 const g = l.effort[mode]
110 g.output += u.output_tokens
111 g.usd += cost(model, u) ?? 0
112 // every main request would have re-read the trimmed text, at the cache-read price
113 const trimmed = (await read($, trimmedAtom)) - trimmedBase
114 l.trimmer.tokens += trimmed
115 l.trimmer.usd += (trimmed * (price(model)?.[2] ?? 0)) / 1e6
116 } else if ((await read($, routedAtom)).includes(e.agentId)) {
117 const routed = cost(model, u)
118 const parent = cost(mainModel, u)
119 if (routed !== undefined && parent !== undefined) {
120 l.router.tokens += u.input_tokens + u.cache_creation_input_tokens + u.cache_read_input_tokens + u.output_tokens
121 l.router.usd += parent - routed
122 }
123 }
124 await save($)
125 return result
126 })
127}
128types/index.d.ts 10 lines1// the other mods' own declarations, repeated so the ledger can read what they publish
2declare module 'claude-code' {
3 interface PluginState {
4 'auto-effort': { effort: string | null }
5 'output-trimmer': { trimmed: number }
6 'subagent-router': { routed: string[] }
7 }
8}
9export type Contract = never
10