Blocks edits that make tests pass by weakening them: new skip/only markers, fewer assertions, a gutted test file, or deleting a test file.

Refuses a Write, Edit, MultiEdit or Bash call that would make a failing test "pass" by weakening it. Claude gets the reason and is told to fix the code, or ask you if the test is obsolete.
A tool.call guard that compares old and new text, reads the existing file with $.fs, and fails closed with .catch.
Only files that look like tests: *.test.*, *.spec.*, test_*.py, *_test.py, *_test.go, or anything under __tests__/, tests/, spec/.
| Detection | Rule |
|---|---|
| Skip/only marker | .skip(, .only(, xit(, xdescribe(, it.todo, test.todo, @pytest.mark.skip, @unittest.skip, t.Skip(, pending( appears more often in the new text than the old |
| Fewer assertions | count of expect(, assert, .should, t.Error, t.Fatal, require. drops. Edit: old vs new string. MultiEdit: summed over edits. Write: existing file vs new content |
| Gutted file | a Write leaves fewer than half of the existing lines |
| Deleted test | rm, git rm or git checkout -- naming a test file |
Not flagged: renaming a test, adding tests or assertions, editing non-test files, creating a new test file, an .skip that was already there.
Headless claude -p (Haiku, acceptEdits) in a throwaway dir, asked to change test('adds' to test.skip('adds' in a failing foo.test.js. The edit was refused, the file is unchanged, and Claude relayed:
Test Guard blocked this Edit call: it adds skip/only marker(s) .skip(. Fix the code under test so the existing test passes, do not change the test to fit the code.
One tool.call hook. For a test file it builds the before and after text, runs the checks above, and returns { deny } listing what it found in plain words, or calls next(e). A Write to a path that doesn't exist yet is allowed. If the hook throws (for example the file is over the 4 MiB read cap) the .catch denies.
What claude plugin validate reports:
hooks: tool.call
calls: $.fs.exists (via checkEdit), $.fs.read (via checkEdit)
Requires Claude Code 2.1.289 or later.
claude --plugin-dir ./mods/test-guard # one session
claude plugin marketplace add justmalhar/awesome-claude-mods
claude plugin install test-guard@awesome-claude-mods --scope user
Test it: claude plugin test mods/test-guard. The block was also checked once in a live headless session (see Demo).
Edit only sees the replaced snippet.;, &, | segment: quoted paths with spaces, globs and $(...) aren't understood, and a script that deletes tests isn't seen.sed -i). It is a safety net, not a permission system.None.
hooks/test-guard.mjs 134 lines1// Test Guard: refuses a tool call that makes a failing test "pass" by weakening it.
2//
3// tool.call (Write, Edit, MultiEdit, Bash): for a test file, compare old and new
4// text for new skip/only markers, fewer assertions, or a Write that gutted the
5// file; refuse `rm` / `git rm` / `git checkout --` of a test file.
6//
7// ponytail: substring heuristics, not a parser. It cannot see a test that is
8// weakened semantically (a looser matcher, a mocked-out subject, a commented-out
9// block that keeps the token count equal). Upgrade path: AST diff per language.
10// The host reads `on(...)` and `$.noun.method(...)` from source: spelled literally.
11
12const TOOLS = new Set(["Write", "Edit", "MultiEdit", "Bash"]);
13const SKIPS = [
14 ".skip(", ".only(", "xit(", "xdescribe(", "it.todo", "test.todo",
15 "@pytest.mark.skip", "@unittest.skip", "t.Skip(", "pending(",
16];
17const ASSERTS = ["expect(", "assert", ".should", "t.Error", "t.Fatal", "require."];
18const TEST_DIRS = new Set(["__tests__", "tests", "spec"]);
19const MAX_LINES_LOST = 0.5; // a Write may drop at most half of the existing lines
20
21export function register(on) {
22 on("tool.call", async ($, e, next) => {
23 if (!TOOLS.has(e.tool)) {
24 return next(e);
25 }
26 const found = e.tool === "Bash" ? checkBash(e.command) : await checkEdit($, e);
27 if (found.length === 0) {
28 return next(e);
29 }
30 return {
31 deny:
32 `Test Guard blocked this ${e.tool} call: ${found.join("; ")}. ` +
33 `Fix the code under test so the existing test passes, do not change the test to fit the code. ` +
34 `If the test is genuinely obsolete or wrong, ask the user first and let them decide.`,
35 };
36 }).catch(async () => ({
37 // fail closed: a skipped guard would let the weakened test through
38 deny: "Test Guard failed while checking this call, so it was not run. Do not retry it unless the user asks you to.",
39 }));
40}
41
42/** True for paths that look like a test file or live in a test directory. */
43function isTestPath(path) {
44 const parts = String(path).split("/");
45 const base = parts[parts.length - 1];
46 return (
47 base.includes(".test.") || base.includes(".spec.") ||
48 (base.startsWith("test_") && base.endsWith(".py")) ||
49 base.endsWith("_test.py") || base.endsWith("_test.go") ||
50 parts.slice(0, -1).some((p) => TEST_DIRS.has(p))
51 );
52}
53
54/** Plain-words findings for a Write/Edit/MultiEdit; [] means allow. */
55async function checkEdit($, e) {
56 if (typeof e.file_path !== "string" || !isTestPath(e.file_path)) {
57 return [];
58 }
59 let before;
60 let after;
61 let wholeFile = false;
62 if (e.tool === "Write") {
63 // a new file has nothing to weaken; 4 MiB read cap overflow throws and fails closed
64 if (!(await $.fs.exists(e.file_path))) {
65 return [];
66 }
67 before = await $.fs.read(e.file_path);
68 after = String(e.content ?? "");
69 wholeFile = true;
70 } else if (e.tool === "MultiEdit") {
71 const edits = Array.isArray(e.edits) ? e.edits : [];
72 before = edits.map((x) => String(x.old_string ?? "")).join("\n");
73 after = edits.map((x) => String(x.new_string ?? "")).join("\n");
74 } else {
75 before = String(e.old_string ?? "");
76 after = String(e.new_string ?? "");
77 }
78 return compare(before, after, wholeFile);
79}
80
81/** Findings for before -> after text. Pure. */
82function compare(before, after, wholeFile) {
83 const found = [];
84 const skips = SKIPS.filter((m) => count(after, m) > count(before, m));
85 if (skips.length > 0) {
86 found.push(`it adds skip/only marker(s) ${skips.join(", ")}`);
87 }
88 const was = total(before, ASSERTS);
89 const now = total(after, ASSERTS);
90 // ponytail: zero tolerance, so even a deliberately removed duplicate assertion is flagged
91 if (now < was) {
92 found.push(`it lowers the assertion count from ${was} to ${now}`);
93 }
94 if (wholeFile) {
95 const oldLines = before.split("\n").length;
96 const newLines = after.split("\n").length;
97 if (newLines < oldLines * MAX_LINES_LOST) {
98 found.push(`it shrinks the file from ${oldLines} to ${newLines} lines`);
99 }
100 }
101 return found;
102}
103
104/** Findings for a Bash command: rm / git rm / git checkout -- on a test file. */
105function checkBash(command) {
106 const found = [];
107 // ponytail: whitespace split; quoted paths with spaces and `$(...)` are not understood
108 for (const segment of String(command ?? "").split(/[;&|\n]+/)) {
109 const t = segment.trim().split(/\s+/);
110 const destructive =
111 t[0] === "rm" || (t[0] === "git" && t[1] === "rm") || (t[0] === "git" && t[1] === "checkout" && t.includes("--"));
112 if (destructive) {
113 for (const arg of t.slice(1)) {
114 if (!arg.startsWith("-") && isTestPath(arg)) {
115 found.push(`it deletes or reverts the test file ${arg}`);
116 }
117 }
118 }
119 }
120 return found;
121}
122
123function count(text, token) {
124 let n = 0;
125 for (let i = text.indexOf(token); i !== -1; i = text.indexOf(token, i + token.length)) {
126 n += 1;
127 }
128 return n;
129}
130
131function total(text, tokens) {
132 return tokens.reduce((sum, t) => sum + count(text, t), 0);
133}
134