Right reasoning effort for every prompt without breaking the prompt cache

Two Claude Code mods that stop your plan from being burned by accident.
Built after a real incident: a script that called claude -p every 60 seconds made about 800 Opus calls in 13 hours, emptied two full plan windows, and produced nothing but "hold". Nothing in Claude Code stopped it. These mods would have stopped it after 20 minutes.
| mod | what it does |
|---|---|
| quota-sentry | Caps headless claude -p runs at 20 per hour across your whole machine. After 15 automatic turns in a row with no message from you, it asks before going on. Warns at 80% and 95% of your 5-hour and 7-day windows. Your own prompts are never blocked. |
| effort-router | Sets reasoning effort per prompt: low for "ok, go on", high for real work. It does not switch the main model, because the prompt cache belongs to one model and switching re-reads your whole context. Optionally moves subagents (which start with an empty context) to a cheaper model. Above 80% of your 5-hour window, effort steps down one level. |
Classification is local keyword matching: zero tokens. No network calls, no telemetry.
Requires Claude Code 2.1.287 or later (mods).
claude plugin marketplace add amiralimardani-art/token-guardrails
claude plugin install quota-sentry@token-guardrails
claude plugin install effort-router@token-guardrails
Start a new session, then try /quota-sentry and /effort-router.
| command | effect |
|---|---|
/quota-sentry | plan windows, session cost, loop counters |
/quota-sentry pause · resume | disable the blocks for one hour, or re-enable them |
/effort-router | how your prompts were classified |
/effort-router off · on | turn routing off or on |
/effort-router subagents claude-sonnet-5-5 | move subagents on non-hard tasks to that model (subagents off to undo) |
All commands answer instantly without a model turn, so they cost nothing.
cd quota-sentry && claude plugin test # 4 tests
cd effort-router && claude plugin test # 4 tests
low on a prompt that needed more will give a weaker answer: /effort-router off if it bothers you.Free and MIT licensed. If it saved you a plan window, you can pay what you want on Gumroad.
Need it set up on your team, or a custom guardrail for your workflow? Open an issue.
MIT
hooks/register.js 86 lines1// effort-router — the right reasoning effort for every prompt, without breaking the cache.
2//
3// Why not switch models per prompt? The prompt cache belongs to one model. Jumping between
4// Haiku and Opus re-reads your whole context uncached on every jump, which on a long
5// session costs more than it saves. So the main conversation keeps its model and only the
6// EFFORT changes: low for "ok, go on", high for real work.
7// Subagents start from an empty context, so moving them to a cheaper model costs no cache:
8// opt in with `/effort-router subagents <model-id>` (for example claude-sonnet-5-5).
9// Classification is local keyword matching: zero tokens. When unsure: medium.
10// Above 80% of the 5-hour plan window, effort drops one step.
11
12const HARD = /\b(debug|bug|error|fail|crash|refactor|architect|design|implement|build|write|migrat|optimi[sz]|secur|review|analy[sz]|investigat|why|root cause|performance|test|deploy|production|money|trading|strateg|research|compare|plan)/i
13const EASY = /^\s*(ok(ay)?|yes|yep|no|go( on| ahead)?|continue|proceed|thanks?|thank you|great|perfect|nice|done|sure|status|show|open|next|lgtm)(?=[\s,.!?]|$)/i
14const STEPS = ['low', 'medium', 'high']
15const EFFORT = { easy: 'low', medium: 'medium', hard: 'high' }
16
17export function classify(text) {
18 const t = (text || '').trim()
19 if (HARD.test(t)) return 'hard'
20 if (t.length > 600) return 'hard'
21 if (EASY.test(t) && t.length < 120) return 'easy'
22 return 'medium'
23}
24
25export function stepDown(effort) {
26 const i = STEPS.indexOf(effort)
27 return i > 0 ? STEPS[i - 1] : effort
28}
29
30let level = 'hard'
31let highUsage = false
32const counts = { easy: 0, medium: 0, hard: 0, subagents_moved: 0 }
33
34export function register(on) {
35 on('session.start', async ($, e, next) => {
36 try { await $.command.register({ name: 'effort-router', description: 'Stats and settings. Args: on | off | subagents <model-id> | subagents off', argumentHint: '[on|off|subagents <model-id>|subagents off]', immediate: true }) } catch {}
37 return next(e)
38 })
39
40 on('prompt.submit', async ($, e, next) => {
41 const kind = e.origin ? e.origin.kind : 'unclassified'
42 if (kind === 'composer' || kind === 'bridge') {
43 level = classify(e.text)
44 counts[level] += 1
45 }
46 return next(e)
47 })
48
49 on('session.measure', async ($, e, next) => {
50 const five = (e.rateLimits || []).find((l) => l.kind === 'five_hour')
51 highUsage = !!five && five.percentUsed >= 80
52 return next(e)
53 })
54
55 on('turn.step', async function* ($, e, next) {
56 if ((await $.store.get('off')) === true) return yield* next(e)
57 if (e.agentId) {
58 const target = await $.store.get('subagent_model')
59 if (target && level !== 'hard' && e.model !== target) {
60 counts.subagents_moved += 1
61 return yield* next({ ...e, model: target })
62 }
63 return yield* next(e)
64 }
65 let effort = EFFORT[level]
66 if (highUsage) effort = stepDown(effort)
67 return yield* next({ ...e, effort })
68 })
69
70 on('command.run', { command: 'effort-router' }, async ($, e) => {
71 const args = (e.args || '').trim().split(/\s+/)
72 if (args[0] === 'off') { await $.store.set('off', true); return { text: 'effort-router OFF: model and effort stay as the session sets them.' } }
73 if (args[0] === 'on') { await $.store.set('off', false); return { text: 'effort-router ON.' } }
74 if (args[0] === 'subagents') {
75 if (!args[1] || args[1] === 'off') { await $.store.set('subagent_model', null); return { text: 'Subagents keep their own model.' } }
76 await $.store.set('subagent_model', args[1])
77 return { text: 'Subagents on non-hard tasks will use ' + args[1] + '.' }
78 }
79 const off = (await $.store.get('off')) === true
80 const target = await $.store.get('subagent_model')
81 return { text: 'effort-router ' + (off ? 'OFF' : 'ON') + ' · last prompt: ' + level + ' (effort ' + EFFORT[level] + (highUsage ? ', stepped down: plan above 80%' : '') + ')' +
82 '\nprompts classified: easy ' + counts.easy + ' · medium ' + counts.medium + ' · hard ' + counts.hard +
83 '\nsubagent model: ' + (target || 'unchanged') + ' · requests moved: ' + counts.subagents_moved }
84 })
85}
86