Records which real checks ran and which files changed after them; status line, /evidence, and an evidence tool Claude reads before claiming tested.

Keeps the evidence behind the CLAUDE.md status words: which real checks ran, whether they passed, and which files changed after the last pass.
pnpm test|typecheck|build|lint, vitest, jest, pytest, tsc, cargo/go/swift test/build, eslint, claude plugin test|validate, and similar. Syntax-only checks (bash -n, node --check, json.tool), --help and --version don't count. The outcome comes from the Bash call's own error flag.✓ pnpm run typecheck · 2 files unchecked, or ✗ pnpm test./evidence: the last 10 checks, the files edited after the last passing check, and the strongest word the ledger supports: tested (every edit has a passing check after it) or implemented locally.evidence tool (mcp__check-ledger__evidence): the same report for Claude to read before it says "tested". The tool needs a local listener. Where that is refused (a sandboxed launch) it logs one line, and /evidence and the status line still work.Limits: the ledger lives in the module and restarts on a reload or a new session. It only sees calls made in this session, including its subagents. It observes; it never blocks or changes a call.
Hooks: session.start, tool.call (Bash and edit tools, passed through unchanged), command.run. Calls: $.command.register, $.tool.register, $.ui.status, $.ui.log, $.clock.now.
Test: claude plugin test . (3 tests).
hooks/register.ts 126 lines1import type { Register } from 'claude-code'
2
3// A command counts as a real check when it runs the repo's tests, type-checker, build,
4// linter or validator. Syntax-only checks (bash -n, node --check, json.tool) do not count.
5const CHECK =
6 /\b(?:(?:pnpm|npm|yarn|bun)\s+(?:run\s+)?(?:test|typecheck|type-check|build|lint|check|validate|verify|e2e)(?::\S+)?\b|vitest|jest|playwright\s+test|pytest|python3?\s+-m\s+pytest|tsc\b(?!.*--version)|cargo\s+(?:test|build|clippy)|go\s+(?:test|build|vet)|swift\s+(?:test|build)|xcodebuild\b.*\b(?:test|build)\b|node\s+--test|eslint|ruff\s+check|mypy|claude\s+plugin\s+(?:test|validate))/i
7const NOT_A_CHECK = /\b(?:bash|sh|zsh)\s+-n\b|node\s+--check|json\.tool|--dry-run|--help|--version/i
8const EDIT_TOOLS = new Set(['Edit', 'Write', 'MultiEdit', 'NotebookEdit'])
9
10export type Check = { command: string; outcome: 'pass' | 'fail' | 'denied'; at: number; seq: number }
11export type Change = { file: string; at: number; seq: number }
12
13export type Ledger = { checks: Check[]; changes: Change[]; seq: number }
14
15export function newLedger(): Ledger {
16 return { checks: [], changes: [], seq: 0 }
17}
18
19export function isCheck(command: string): boolean {
20 return CHECK.test(command) && !NOT_A_CHECK.test(command)
21}
22
23export function shortCommand(command: string): string {
24 const one = command.replace(/\s+/g, ' ').trim()
25 return one.length > 60 ? `${one.slice(0, 57)}...` : one
26}
27
28export function recordCheck(l: Ledger, command: string, outcome: Check['outcome'], at: number): void {
29 l.checks.push({ command: shortCommand(command), outcome, at, seq: ++l.seq })
30 if (l.checks.length > 200) l.checks.shift()
31}
32
33export function recordChange(l: Ledger, file: string, at: number): void {
34 l.changes = l.changes.filter(c => c.file !== file)
35 l.changes.push({ file, at, seq: ++l.seq })
36}
37
38function lastPass(l: Ledger): Check | undefined {
39 return [...l.checks].reverse().find(c => c.outcome === 'pass')
40}
41
42// Files changed after the last passing check (all changes when nothing has passed).
43export function unchecked(l: Ledger): Change[] {
44 const pass = lastPass(l)
45 return l.changes.filter(c => !pass || c.seq > pass.seq)
46}
47
48// The strongest word in the CLAUDE.md status vocabulary the ledger alone supports.
49export function supportedStatus(l: Ledger): string {
50 if (l.changes.length === 0) return l.checks.length ? 'no edits this session; checks ran' : 'no edits or checks this session'
51 const last = l.checks.at(-1)
52 if (last && last.outcome === 'fail') return 'implemented locally (latest check failed)'
53 return unchecked(l).length === 0 ? 'tested (every edit has a passing check after it)' : 'implemented locally (edits after the last passing check)'
54}
55
56export function statusText(l: Ledger): string | undefined {
57 if (!l.checks.length && !l.changes.length) return undefined
58 const last = l.checks.at(-1)
59 const open = unchecked(l).length
60 const head = last ? `${last.outcome === 'pass' ? '✓' : '✗'} ${last.command.slice(0, 28)}` : 'no checks yet'
61 return open ? `${head} · ${open} file${open === 1 ? '' : 's'} unchecked` : head
62}
63
64function ago(ms: number): string {
65 const m = Math.round(ms / 60_000)
66 return m < 1 ? 'just now' : m < 60 ? `${m}m ago` : `${Math.floor(m / 60)}h ${m % 60}m ago`
67}
68
69export function reportText(l: Ledger, now: number): string {
70 const lines = [`Check ledger: supports "${supportedStatus(l)}"`]
71 lines.push('', 'Checks (latest last):')
72 if (!l.checks.length) lines.push('- none recorded')
73 for (const c of l.checks.slice(-10)) lines.push(`- ${c.outcome.toUpperCase()} \`${c.command}\` (${ago(now - c.at)})`)
74 const open = unchecked(l)
75 lines.push('', `Files changed after the last passing check: ${open.length}`)
76 for (const c of open.slice(-15)) lines.push(`- ${c.file}`)
77 if (open.length > 15) lines.push(`- ...and ${open.length - 15} more`)
78 lines.push('', 'Recorded since this mod loaded; a check that ran in another session or a subagent shell outside this session is not here.')
79 return lines.join('\n')
80}
81
82// Module state: a reload starts a fresh ledger, which the report says.
83const ledger = newLedger()
84
85export const register: Register = on => {
86 on('session.start', async ($, e, next) => {
87 await $.command.register({ name: 'evidence', description: 'Which checks ran and which edits came after them' })
88 // The tool needs a local listener; where that is refused (a sandboxed launch) the
89 // ledger, status line and /evidence still work.
90 await $.tool
91 .register({
92 name: 'evidence',
93 description:
94 'Read-only. Returns the checks (tests, type-check, build, lint) run in this session with pass/fail, the files edited after the last passing check, and the strongest status word that evidence supports. Call it before reporting a change as tested.',
95 inputSchema: { type: 'object', properties: {} },
96 })
97 .catch(err => $.ui.log(`check-ledger: evidence tool not registered (${String(err)}); /evidence still works`))
98 return next(e)
99 })
100
101 on('tool.call', { tool: 'Bash' }, async ($, e, next) => {
102 if (!isCheck(e.command)) return next(e)
103 const ran = await next(e)
104 const outcome = ran.deny !== undefined ? 'denied' : ran.isError ? 'fail' : 'pass'
105 recordCheck(ledger, e.command, outcome, await $.clock.now())
106 $.ui.status(statusText(ledger))
107 return ran
108 })
109
110 on('tool.call', async ($, e, next) => {
111 if (!EDIT_TOOLS.has(String(e.tool))) return next(e)
112 const ran = await next(e)
113 const input = e as unknown as { file_path?: string; notebook_path?: string }
114 const file = input.file_path ?? input.notebook_path
115 if (file && ran.deny === undefined && !ran.isError) {
116 recordChange(ledger, file, await $.clock.now())
117 $.ui.status(statusText(ledger))
118 }
119 return ran
120 })
121
122 on('tool.call', { tool: 'mcp__check-ledger__evidence' }, async $ => ({ result: reportText(ledger, await $.clock.now()) }))
123
124 on('command.run', { command: 'evidence' }, async $ => ({ text: reportText(ledger, await $.clock.now()) }))
125}
126