Checks Claude's claims (tests pass, verified, fixed, no regression, API returns N) against the tool calls that actually ran. Unsupported claims get a visible…

Fourteen Claude Code mods, packaged as one plugin marketplace (qa-mods). Made for a manual QA workflow: guard rails around risky commands, evidence for what was tested, and a status band above the prompt.
Run in your own Terminal (a sandboxed Claude session cannot write ~/.claude).
claude plugin marketplace add oleksandrkodua/qa-mods
claude plugin install hud@qa-mods
Replace hud with any plugin from the table.
To install everything from a local clone, or to set up a fresh machine, see INSTALL.md and install-pack.sh.
| Plugin | What it does |
|---|---|
sandbox-guard | Denies writes to protected Claude config paths and hands back a Terminal command |
remote-gate | Asks before git push and wrangler --remote, showing folder, branch, remote and the commits that will go out |
blast-radius | Shows what a risky Bash command will touch before it runs |
retry-analyzer | Detects repeated identical failing tool calls and tells Claude to change strategy |
secret-redactor | Masks tokens and keys in tool output |
evidence-saver | Saves screenshots from tool results to a folder you pick |
replay-theater | /replay: step through the session's file edits one by one |
verification-guard | Checks claims like "tests pass" against tool calls that actually ran |
handoff | Offers /handoff as the context fills up; fills the prompt box with a HANDOFF.md prompt |
notify | A toast and a macOS notification when a long command or turn finishes |
quick-actions | Compact and Clear buttons, each asks to confirm first |
hud | Context fill and rate-limit windows above the prompt, a кеш 59:48 chip (m:ss) while the prompt cache is warm, five buttons, and /hud with the figures as text |
next-steps-uk | Fork of next-steps (MIT, Thariq Shihipar): suggested next prompts are always in Ukrainian. Install instead of next-steps |
plan-progress | Fork of plan-progress 0.7.6 (zycck, MIT): live progress bars above the prompt. Install instead of plan-progress@zycck-mods |
claude plugin test hud # from the repo root
cd hud && node --test tests/format.test.mjs tests/handoff-prompt.test.mjs
A bare node --test also picks up register.test.ts, which only runs under claude plugin test.
No license file for the original mods yet. The two forks keep their upstream MIT notices: next-steps-uk/NOTICE.md and plan-progress/NOTICE.md (plus plan-progress/LICENSE.upstream).
hooks/register.ts 99 lines1import type { Register } from 'claude-code'
2
3import { detectClaims } from './claims'
4import { assess, classify, warning } from './ledger'
5import type { Assessment, Entry } from './ledger'
6
7const MAX_ENTRIES = 500
8
9/**
10 * Verification Guard: Claude's claims ("tests pass", "verified", "fixed", "no regression")
11 * are checked against the tool calls that actually ran. Unsupported claims get a visible
12 * warning appended to the answer; `/evidence` lists claims and what backs them.
13 * Warn only: the text is already written when it can be judged, nothing is blocked.
14 */
15export const register: Register = on => {
16 const ledger: Entry[] = []
17 const log: { turn: number; a: Assessment }[] = []
18 let turn = 0
19 let n = 0
20
21 on('session.start', async ($, e, next) => {
22 // a name already taken must not break the mod
23 await $.command
24 .register({
25 name: 'evidence',
26 description: 'Claude’s claims this session and the evidence behind each',
27 })
28 .catch(() => undefined)
29
30 return next(e)
31 })
32
33 on('turn.start', ($, e, next) => {
34 turn += 1
35
36 return next(e)
37 })
38
39 on('tool.call', async ($, e, next) => {
40 const ran = await next(e)
41 const { kind, label } = classify(String(e.tool), e as { command?: unknown; file_path?: unknown })
42
43 ledger.push({ n: ++n, turn, tool: String(e.tool), kind, label, ok: ran.deny === undefined && ran.isError !== true })
44
45 if (ledger.length > MAX_ENTRIES) ledger.shift()
46
47 return ran
48 })
49
50 on('turn.step', async function* ($, e, next) {
51 const stream = next(e)
52 let text = ''
53 let last = -1
54 let sawTool = false
55
56 while (true) {
57 const { value, done } = await stream.next()
58
59 if (done) return value
60
61 if (value.kind === 'text') {
62 text += value.text
63 last = value.index
64 } else if (value.kind === 'tool') {
65 sawTool = true
66 } else if (value.kind === 'stop' && e.agentId === undefined && !sawTool && value.stopReason === 'end_turn' && last >= 0) {
67 // the final answer of the main loop: judge its claims before it is closed
68 const a = assess(detectClaims(text), ledger, turn)
69
70 if (a.rows.length) {
71 log.push({ turn, a })
72
73 if (a.supported < a.rows.length) yield { kind: 'text', index: last, text: warning(a) }
74 }
75 }
76
77 yield value
78 }
79 })
80
81 on('command.run', { command: 'evidence' }, () => {
82 const rows = log.flatMap(x => x.a.rows.map(r => ({ turn: x.turn, r })))
83 const ok = rows.filter(x => x.r.supported).length
84 const count = (k: string, good?: boolean) => ledger.filter(x => x.kind === k && (good === undefined || x.ok === good)).length
85
86 const head = rows.length
87 ? `Verification Guard — ${rows.length} claim(s) checked, ${ok} supported (${Math.round((100 * ok) / rows.length)}%) → ${
88 ok === rows.length ? 'VERIFIED' : ok === 0 ? 'UNVERIFIED' : 'PARTIALLY VERIFIED'
89 }`
90 : 'Verification Guard — no verification/completion claims seen yet in this session.'
91
92 const body = rows.map(x => `${x.r.supported ? '✓' : '✗'} [turn ${x.turn}] «${x.r.claim.quote}»\n ${x.r.note}`).join('\n')
93
94 const tally = `Ledger: ${count('edit')} edit · ${count('test', true)} test ✓ ${count('test', false)} ✗ · ${count('http', true)} http ✓ ${count('http', false)} ✗ · ${count('browser')} browser · ${count('read')} read · ${count('run')} run`
95
96 return { text: [head, body, tally].filter(Boolean).join('\n\n') }
97 })
98}
99hooks/claims.ts 59 lines1export type ClaimKind = 'tests' | 'api' | 'verified' | 'done' | 'safe'
2
3export interface Claim {
4 kind: ClaimKind
5 quote: string
6}
7
8// Cyrillic text: \b does not work there, so those patterns avoid it.
9const PATTERNS: Record<ClaimKind, RegExp[]> = {
10 tests: [
11 /\b(all\s+)?(unit\s+|e2e\s+|integration\s+)?tests?\s+(are\s+|were\s+|now\s+)?(passing|passed|pass|green)\b/i,
12 /\b(test\s+)?suite\s+(is\s+)?(passing|green)\b/i,
13 /тест(ы|ов|и)?\s+(\S+\s+)?(проход|прошл|пройд|зелен|зелён|успешн|успішн|пройшли)/i,
14 ],
15 api: [
16 /\b(returns?|returned|responds?|responded|gets?|got)\s+(an?\s+)?(http\s+)?[245]\d\d\b/i,
17 /\bstatus(\s+code)?\s*(is|=|:|was)?\s*[245]\d\d\b/i,
18 /\bhttp\s+[245]\d\d\b/i,
19 /(возвращает|вернул\S*|отвечает|ответил\S*|повертає|повернув\S*|відповідає|відповів\S*)\s+(http\s+)?[245]\d\d/i,
20 ],
21 verified: [
22 /\b(i\s+)?(have\s+|'ve\s+)?(verified|confirmed|validated|double-checked)\b/i,
23 /(проверил\S*|подтвердил\S*|верифицировал\S*|убедил\S*сь|перевірив\S*|підтвердив\S*|пересвідчи\S*сь)/i,
24 ],
25 done: [
26 /\b(i\s+)?(have\s+|'ve\s+)?(fixed|implemented|completed|resolved)\b/i,
27 /^\s*done\b/im,
28 /(готово|исправил\S*|исправлен\S*|реализовал\S*|сделал\S*|виправив\S*|виправлен\S*|реалізував\S*|зроблено)/i,
29 ],
30 safe: [
31 /\bno\s+regressions?\b/i,
32 /\b(backwards?[- ]compatible|production[- ]ready|safe\s+to\s+(merge|deploy|ship))\b/i,
33 /(без\s+регресс|регрессий\s+нет|обратно\s+совместим|зворотно\s+сумісн|регресій\s+немає)/i,
34 ],
35}
36
37// A sentence that says it did NOT verify is not a claim.
38const NEGATION =
39 /\b(not|n't|never|unable|cannot|can't|couldn't|haven't|didn't|wasn't|unverified|untested|if|once|when|should|would|until)\b|(?:^|\s)(не\s+(проверил|удалось|смог|запуск|выполн|перевір|вдал|підтверд|підтвердж|підтверд)|ещё\s+не|еще\s+не|пока\s+не|поки\s+не|ще\s+не|если|якщо|после\s+того|коли)/i
40
41const stripCode = (s: string) => s.replace(/```[\s\S]*?```/g, ' ').replace(/`[^`\n]*`/g, ' ')
42
43/** Claims about verification/completion in one assistant answer, one per kind, quoted. */
44export function detectClaims(answer: string): Claim[] {
45 const out: Claim[] = []
46 const sentences = stripCode(answer)
47 .split(/(?<=[.!?\n])\s+/)
48 .map(s => s.trim())
49 .filter(Boolean)
50
51 for (const kind of Object.keys(PATTERNS) as ClaimKind[]) {
52 const hit = sentences.find(s => !NEGATION.test(s) && PATTERNS[kind].some(re => re.test(s)))
53
54 if (hit) out.push({ kind, quote: hit.length > 110 ? `${hit.slice(0, 110)}…` : hit })
55 }
56
57 return out
58}
59hooks/ledger.ts 113 lines1import type { Claim, ClaimKind } from './claims'
2
3export type EvKind = 'edit' | 'test' | 'http' | 'browser' | 'read' | 'run' | 'other'
4
5export interface Entry {
6 n: number
7 turn: number
8 tool: string
9 kind: EvKind
10 label: string
11 ok: boolean
12}
13
14const TEST =
15 /(^|[\s;&|(])((npm|pnpm|yarn|bun)\s+(run\s+)?(test|t|e2e|check)\b|(pytest|py\.test|jest|vitest|mocha|rspec|phpunit|tox)\b)|\b(go|cargo|dotnet|swift|mvn|gradlew?)\s+test\b|playwright\s+test|cypress\s+run|node\s+--test|python3?\s+-m\s+(pytest|unittest)|claude\s+plugin\s+test/
16
17const HTTP = /(^|[\s;&|(])(curl|wget|http|https|xh|httpie)\s/
18const TRIVIAL = /^\s*(ls|cat|pwd|echo|which|head|tail|wc|date|true|git\s+(status|diff|log|show|branch)|cd)\b/
19
20const EDIT_TOOLS = ['Edit', 'Write', 'MultiEdit', 'NotebookEdit']
21const READ_TOOLS = ['Read', 'Grep', 'Glob']
22
23/** What kind of evidence one tool call is, from its name and arguments. */
24export function classify(tool: string, args: { command?: unknown; file_path?: unknown }): { kind: EvKind; label: string } {
25 if (EDIT_TOOLS.includes(tool)) return { kind: 'edit', label: String(args.file_path ?? '') }
26
27 if (tool === 'Bash') {
28 const cmd = String(args.command ?? '').replace(/\s+/g, ' ').trim()
29 const label = cmd.length > 80 ? `${cmd.slice(0, 80)}…` : cmd
30
31 if (TEST.test(cmd)) return { kind: 'test', label }
32 if (HTTP.test(cmd)) return { kind: 'http', label }
33
34 return { kind: TRIVIAL.test(cmd) ? 'other' : 'run', label }
35 }
36
37 if (READ_TOOLS.includes(tool)) return { kind: 'read', label: String(args.file_path ?? tool) }
38 if (/screenshot|browser|playwright|computer|navigate|chrome/i.test(tool)) return { kind: 'browser', label: tool }
39
40 return { kind: 'other', label: tool }
41}
42
43export interface Row {
44 claim: Claim
45 supported: boolean
46 /** why it is unsupported, or what supports it */
47 note: string
48}
49
50export interface Assessment {
51 rows: Row[]
52 supported: number
53 status: 'VERIFIED' | 'PARTIALLY VERIFIED' | 'UNVERIFIED'
54}
55
56const WHY: Record<ClaimKind, string> = {
57 tests: 'no passing test run after the last edit',
58 api: 'no HTTP request or browser check after the last edit',
59 verified: 'no successful check (test, request, read or command) in this turn',
60 done: 'files were edited but nothing was run or tested afterwards',
61 safe: 'no passing test run after the last edit',
62}
63
64/** Judge each claim against what actually ran. Evidence must be successful and, for fixes, come after the last edit. */
65export function assess(claims: Claim[], ledger: readonly Entry[], turn: number): Assessment {
66 const lastEdit = Math.max(0, ...ledger.filter(x => x.kind === 'edit').map(x => x.n))
67 const after = ledger.filter(x => x.ok && x.n > lastEdit)
68 const inTurn = ledger.filter(x => x.ok && x.turn === turn)
69 const editedThisTurn = ledger.some(x => x.kind === 'edit' && x.turn === turn)
70 const rows: Row[] = []
71
72 for (const claim of claims) {
73 let via: Entry | undefined
74
75 switch (claim.kind) {
76 case 'tests':
77 case 'safe':
78 via = after.find(x => x.kind === 'test')
79 break
80 case 'api':
81 via = after.find(x => x.kind === 'http' || x.kind === 'browser')
82 break
83 case 'verified':
84 via = inTurn.find(x => ['test', 'http', 'browser', 'read', 'run'].includes(x.kind))
85 break
86 case 'done':
87 if (!editedThisTurn) continue // nothing was changed: "done" is not a code claim
88 via = after.find(x => ['test', 'http', 'browser', 'run'].includes(x.kind))
89 break
90 }
91
92 rows.push({ claim, supported: via !== undefined, note: via ? `${via.kind}: ${via.label}` : WHY[claim.kind] })
93 }
94
95 const supported = rows.filter(r => r.supported).length
96
97 return {
98 rows,
99 supported,
100 status: rows.length === 0 || supported === rows.length ? 'VERIFIED' : supported === 0 ? 'UNVERIFIED' : 'PARTIALLY VERIFIED',
101 }
102}
103
104export function warning(a: Assessment): string {
105 const bad = a.rows.filter(r => !r.supported)
106
107 return (
108 `\n\n⚠ verification-guard: ${a.status} (${a.supported}/${a.rows.length}) — ` +
109 bad.map(r => `«${r.claim.quote}» → ${r.note}`).join('; ') +
110 '. Run /evidence for the full list.'
111 )
112}
113