Read hook payload from fd 0, not /dev/stdin

Claude Code delivers the hook payload on a socket. Opening it by path
('/dev/stdin' -> /proc/self/fd/0) fails with ENXIO, so bash-guard.mjs threw on
every invocation and its bare catch exited 0 silently. The guard looked like it
was never dispatched; it was dying on line 29 each time.

readFileSync(0) is read() on the descriptor with no open(), which works on a
socket. The catch now logs instead of swallowing, so this failure mode can never
again masquerade as non-dispatch.

Adds test-bash-guard.py, which drives the guard over a socketpair. A pipe would
not reproduce the bug, so the socket is load-bearing. Verified failing against
the pre-fix guard (silent, no output) and passing after.

Removes the five diagnostic probes and probe.mjs; they served their purpose.
Bumps to 1.0.2 because the plugin cache is keyed by version and an unchanged
version silently skips reinstall.
This commit is contained in:
2026-07-21 10:24:06 -05:00
parent a651c2cb8c
commit 129354cda8
7 changed files with 335 additions and 89 deletions
+1 -1
View File
@@ -1,6 +1,6 @@
{ {
"name": "guards", "name": "guards",
"version": "1.0.1", "version": "1.0.2",
"description": "Personal enforcement hooks: jj-only version control, no Claude attribution, build+test gate on push, and secret scrubbing on file writes.", "description": "Personal enforcement hooks: jj-only version control, no Claude attribution, build+test gate on push, and secret scrubbing on file writes.",
"author": { "author": {
"name": "Malcolm Roberts" "name": "Malcolm Roberts"
+9 -2
View File
@@ -24,10 +24,17 @@ const deny = (reason) => {
process.exit(0); process.exit(0);
}; };
// Read fd 0 directly. Claude Code delivers the payload on a socket, and opening it by path
// ('/dev/stdin' -> /proc/self/fd/0) fails ENXIO. readFileSync(0) is read() with no open().
let input; let input;
try { try {
input = JSON.parse(readFileSync('/dev/stdin', 'utf8')); input = JSON.parse(readFileSync(0, 'utf8'));
} catch { } catch (e) {
// Never swallow this silently: a guard that dies here is indistinguishable from one that
// never ran, which is exactly what hid the ENXIO bug for three sessions.
try {
appendFileSync('/tmp/bash-guard-mode.log', `stdin-FAILED err=${e?.message ?? e}\n`);
} catch {}
process.exit(0); process.exit(0);
} }
const command = input?.tool_input?.command ?? ''; const command = input?.tool_input?.command ?? '';
-50
View File
@@ -11,46 +11,6 @@
"statusMessage": "Checking jj/attribution policy; gating push on build + tests..." "statusMessage": "Checking jj/attribution policy; gating push on build + tests..."
} }
] ]
},
{
"matcher": "Bash",
"hooks": [
{
"type": "command",
"command": "node \"${CLAUDE_PLUGIN_ROOT}/hooks/probe.mjs\" P1-plain"
}
]
},
{
"matcher": "Bash",
"hooks": [
{
"type": "command",
"command": "node \"${CLAUDE_PLUGIN_ROOT}/hooks/probe.mjs\" P2-timeout-only",
"timeout": 600
}
]
},
{
"matcher": "Bash",
"hooks": [
{
"type": "command",
"command": "node \"${CLAUDE_PLUGIN_ROOT}/hooks/probe.mjs\" P3-status-only",
"statusMessage": "probe P3 status-only"
}
]
},
{
"matcher": "Bash",
"hooks": [
{
"type": "command",
"command": "node \"${CLAUDE_PLUGIN_ROOT}/hooks/probe.mjs\" P4-both",
"timeout": 600,
"statusMessage": "probe P4 both"
}
]
} }
], ],
"PostToolUse": [ "PostToolUse": [
@@ -63,16 +23,6 @@
"timeout": 5 "timeout": 5
} }
] ]
},
{
"matcher": "Write|Edit|MultiEdit",
"hooks": [
{
"type": "command",
"command": "node \"${CLAUDE_PLUGIN_ROOT}/hooks/probe.mjs\" P5-post-status",
"statusMessage": "probe P5 post status"
}
]
} }
] ]
} }
-36
View File
@@ -1,36 +0,0 @@
#!/usr/bin/env node
// Diagnostic probe for hook dispatch. Inert: logs and exits 0, never emits a decision.
//
// Logs BEFORE reading stdin, so three outcomes stay distinguishable:
// no line at all -> entry never dispatched
// phase=spawned only -> dispatched, but the stdin read threw (bash-guard's line 29)
// phase=stdin-ok -> dispatched and payload readable; records permission_mode
//
// Each hooks.json entry passes a distinct label so the transcript's `command` field
// and this log both identify which config variant ran.
import { appendFileSync, readFileSync } from 'fs';
import { homedir } from 'os';
import { join } from 'path';
const label = process.argv[2] ?? '<none>';
const stamp = new Date().toISOString();
// Two sinks: a sandboxed hook child might see a private /tmp but still reach $HOME.
const log = (line) => {
for (const p of ['/tmp/guard-probe.log', join(homedir(), 'guard-probe.log')]) {
try { appendFileSync(p, line); } catch {}
}
};
log(`${stamp} probe=${label} phase=spawned\n`);
try {
const input = JSON.parse(readFileSync('/dev/stdin', 'utf8'));
log(`${stamp} probe=${label} phase=stdin-ok mode=${input?.permission_mode ?? '<absent>'}`
+ ` tool=${input?.tool_name ?? '?'} cmd=${(input?.tool_input?.command ?? '').slice(0, 40)}\n`);
} catch (e) {
log(`${stamp} probe=${label} phase=stdin-FAILED err=${e?.message ?? e}\n`);
}
process.exit(0);
+74
View File
@@ -0,0 +1,74 @@
#!/usr/bin/env python3
"""Regression test for bash-guard.mjs.
The payload MUST be delivered over a socketpair, not a pipe. Claude Code hands hook children
their stdin as a socket, where opening '/dev/stdin' fails ENXIO -- that failure mode is the
entire point of this test and a pipe would not reproduce it.
Run: python3 test-bash-guard.py
"""
import json
import os
import socket
import subprocess
import sys
GUARD = os.path.join(os.path.dirname(os.path.abspath(__file__)), "bash-guard.mjs")
def run_guard(payload):
"""Invoke the guard with fd 0 as a socket. Returns its parsed stdout, or None if silent."""
parent, child = socket.socketpair()
proc = subprocess.Popen(
["node", GUARD], stdin=child.fileno(), stdout=subprocess.PIPE, close_fds=False
)
child.close()
parent.sendall(json.dumps(payload).encode())
parent.shutdown(socket.SHUT_WR)
out, _ = proc.communicate(timeout=30)
parent.close()
return json.loads(out) if out.strip() else None
def bash(command):
return {"tool_name": "Bash", "tool_input": {"command": command}}
def decision(result):
return (result or {}).get("hookSpecificOutput", {}).get("permissionDecision")
def reason(result):
return (result or {}).get("hookSpecificOutput", {}).get("permissionDecisionReason", "")
failures = []
def check(name, condition, detail=""):
if condition:
print(f" PASS {name}")
else:
print(f" FAIL {name} {detail}")
failures.append(name)
print("bash-guard over socket stdin:")
# The regression: before the fd-0 fix this returned None for every input, because the guard
# died on readFileSync('/dev/stdin') and exited 0 silently.
r = run_guard(bash('git commit -m "hook test"'))
check("git commit is denied", decision(r) == "deny", f"got {decision(r)!r}")
check("denial names the jj equivalent", "jj describe" in reason(r), f"got {reason(r)!r}")
r = run_guard(bash("git status"))
check("read-only git is allowed", decision(r) != "deny", f"got {decision(r)!r}")
r = run_guard(bash("ls -la"))
check("unrelated command is allowed", decision(r) != "deny", f"got {decision(r)!r}")
r = run_guard(bash('git commit -m "x\n\nCo-Authored-By: Claude <[email protected]>"'))
check("Claude attribution is denied", decision(r) == "deny", f"got {decision(r)!r}")
sys.exit(1 if failures else 0)
+113
View File
@@ -0,0 +1,113 @@
---
name: review-pr
description: Build a guided review briefing for a GitHub pull request so a human can review it with full-codebase context. Use whenever the user provides a PR number or URL and wants to review it, or says things like "review PR 482", "walk me through this PR", "help me get oriented on this PR", or invokes /review-pr. Checks out the branch, maps the change, and produces a step-by-step reading plan. Does NOT perform the review itself and NEVER posts to the PR.
argument-hint: [pr-number-or-url]
allowed-tools: Bash(gh pr view:*), Bash(gh pr diff:*), Bash(gh pr checkout:*), Bash(gh issue view:*), Bash(git status:*), Bash(git branch:*), Bash(git log:*), Bash(git diff:*), Bash(git fetch:*), Bash(git merge-base:*), Bash(git switch:*), Read, Grep, Glob
---
# PR Review Briefing
You are building a **review map for a human reviewer**, not reviewing the code for them. Your job is to hand them a mental model and a reading order; their job is judgment. Do not render verdicts like "LGTM" or "this is wrong" — describe, locate, and flag what deserves their attention.
Target PR: `$ARGUMENTS`. If no argument was given, run `gh pr view` (it infers the PR from the current branch); if that fails, ask the user which PR to review.
Hard rules, no exceptions:
- Read-only with one exception: `gh pr checkout`. Never edit files, never commit, never push.
- Never post comments, reviews, or approvals to the PR.
- If `git status --porcelain` shows uncommitted changes, STOP and ask before checking anything out.
## Phase 0 — Gather (work silently; only surface problems)
1. `git status --porcelain` — abort per the rule above if dirty.
2. `git branch --show-current` — remember it; you'll tell the user how to get back.
3. `gh pr view $ARGUMENTS --json number,title,body,author,baseRefName,headRefName,additions,deletions,changedFiles,files,labels,url`
4. If the body references issues (`#123`, "closes/fixes #123"), pull each: `gh issue view 123 --json title,body`. The issue is often the real statement of intent.
5. `gh pr checkout $ARGUMENTS` — **unless** a `.jj` directory exists at the repo root (jj-colocated repo). In that case do not move git's HEAD behind jj's back: run `jj git fetch` then `jj new <headRefName>@origin` instead, replace the `git status` dirty-check above with `jj st`, and end the briefing with `jj new trunk()` as the way back rather than `git switch -`.
6. `git fetch origin <baseRefName>`, then:
- `git log --oneline origin/<base>..HEAD` — the commit narrative
- `git diff origin/<base>...HEAD --stat` — the shape of the change
- `git diff origin/<base>...HEAD` — the full diff (for large PRs, read per-file as needed instead)
## Phase 1 — Analyze (internal; do not dump this raw)
**Classify every changed file** into one of:
- **Contract** — interfaces, public types, schemas, migrations, API routes, config formats, feature flags
- **Core** — the substantive logic implementing the intent
- **Ripple** — mechanical fallout: call-site updates, renames, import churn, generated files, lockfiles
- **Tests**
- **Docs/config**
**Find the load-bearing change**: the change that *forces* the others to exist. Usually a contract; sometimes a core algorithm. If you removed it, most of the rest of the diff would be unnecessary — that's the test.
**Expected vs. actual**: From the stated intent alone, list what you'd expect to be touched. Compare with reality. Surprises in *both* directions matter: unexpected files (scope creep? hidden coupling?) and expected-but-untouched files (incomplete change?).
**Blast radius**: For each changed or removed public symbol (function signature, type, endpoint, event, config key), `Grep` the repo for references. Collect call sites that are **not in the diff** — unchanged callers of changed code are where integration bugs live. Also grep for old names/patterns that should now be gone.
**Test mapping**: Which behavior changes have corresponding test changes? Which don't?
**Risk scan** — note only what actually applies here: data migrations and rollback, concurrency/ordering, error and timeout paths, security-sensitive surface (auth, input parsing, secrets), performance-sensitive paths, backward compatibility, feature-flag interactions.
## Phase 2 — The briefing (the only user-visible output)
Produce exactly this structure. Every claim must carry a `path/to/file` (with `:line` where useful) so the reviewer can jump straight there. Total length: about one page. It is a map, not the territory.
```
# Review briefing: <title> (#<number>)
<author> · +<adds>/−<dels> across <n> files · <head> → <base> · <url>
## 0. Before you read my map
Answer these from the PR description/issue alone, then compare with §1–3:
- <2–3 questions that force a hypothesis, e.g. "Which modules would YOU
expect this to touch?" / "Where should the tricky part live?">
## 1. What this change does
<3–6 sentences telling the story: the problem, the approach taken, the
shape of the solution. Plain language. No judgment.>
## 2. Start here
`<path>` — <symbol/section> — <one sentence on why this is the
load-bearing change everything else follows from>
## 3. Reading path
Dependency order, not file order. Check off as you go.
- [ ] 1. `<path>` — <what it is> — verify: <the specific thing to
confirm at this stop>
- [ ] 2. ...
- [ ] N. Ripple skim (one pass): `<paths>` — mechanical <renames/call-site
updates>; confirm nothing substantive is hiding in them.
- [ ] N+1. Tests: `<paths>` — do they pin the new behavior or just
exercise the happy path?
## 4. Blast radius — call sites NOT in this diff
- `<path>:<line>` — uses <changed symbol> — confirm: <what could break>
<or: "None found — every reference to changed symbols is updated in
this PR." Only say this if you actually checked.>
## 5. Things to look for in this PR
<Only risks that apply, each anchored to a location. 3–6 items.>
## 6. Expected but not present
<Missing tests for X; `<path>` untouched though it consumes Y; docs for
Z. If genuinely nothing, say "Nothing notable." — don't invent items.>
## 7. Verify by running
<Exact commands: targeted test invocations, build, a manual poke at the
changed path. Prefer narrow over `run everything`.>
---
You're on `<head-branch>`. When done: `git switch -` returns you to
`<original-branch>`.
```
Additional rules for the briefing:
- Sections 1–3 are strictly descriptive; your observations and concerns belong in §4–6 only. This keeps the reviewer's judgment primary.
- If the diff is very large (roughly 800+ non-generated lines), say so up front and propose staging the review: which subset of the reading path to do first, and what to defer to a second pass.
- If the PR description is empty or useless, note that in §0 and suggest the reviewer ask the author for intent before proceeding — then still build the best map you can from the code.
- If commits are clean and tell a story, mention in §3 that commit-by-commit reading (`git log --oneline origin/<base>..HEAD`) is a viable alternative order.
## Bundled reference
`references/pr-review-field-guide.md` (relative to this skill's directory) is the human-only version of this process — the same phases with no AI in the loop, plus a vim/neovim appendix. If the user asks for "the manual version", "the checklist", or how to review without Claude, display that file or point them to it. Do not paraphrase it from memory; read the file.
@@ -0,0 +1,138 @@
# PR Review Field Guide
A repeatable process for reviewing PRs without losing the big picture. No AI required — just git, `gh` (optional), and an editor with go-to-definition / find-references. The core move: **build a mental model first, then use the diff to verify it** — never the reverse.
---
## Phase 1 — Orient (before opening a single diff) · ~5 min
1. Read the PR description and the linked issue. The issue is usually the truer statement of intent.
2. Write a two-sentence hypothesis: *what should this change touch, and where should the hard part live?* Actually write it down — it's your anchor for the whole review.
3. Skim only the changed-file **list** (not contents). Compare against your hypothesis:
- Files you didn't expect → possible scope creep or coupling you didn't know about. Mark them.
- Files you expected but don't see → possibly incomplete change. Mark those too.
If the description is empty and the file list is confusing, stop and ask the author for a summary. That's a review comment in itself.
## Phase 2 — Get out of the diff viewer · ~2 min
```
gh pr checkout <number> # or: git fetch origin <branch> && git switch <branch>
```
Open the repo in your editor. From here on, the web diff is a *map* you glance at — the territory is the checked-out code, where go-to-definition and find-references work.
## Phase 3 — Find the load-bearing change · ~5 min
One change forces most of the others. Find it. Priority order of suspects:
schema / migration → public interface, type, or API contract → core algorithm → everything else
Test: *if this change were reverted, would most of the rest of the diff become unnecessary?* Read that file **in full, in context** — not the hunk.
## Phase 4 — Read outward, in dependency order · bulk of the review
Contracts → core logic → mechanical ripples → tests. At each stop ask:
- Does this change follow necessarily from the load-bearing one?
- Is anything *extra* hiding here that isn't part of the stated intent?
Batch the mechanical stuff (renames, import churn, call-site updates, lockfiles) into one fast skim pass — just confirm nothing substantive is buried in it. Spending equal attention per file is how reviewers burn out and miss the real issue.
## Phase 5 — Trace the blast radius · ~10 min
The classic integration bug: **unchanged callers of changed code.**
1. For every changed public symbol (signature, type, endpoint, event, config key): find-references. Check each call site that is *not* in the diff — is it still correct under the new behavior?
2. Grep for the old name / old pattern. Anything left over that should be gone?
3. Check the seams the diff doesn't show: serialization boundaries, DB reads of migrated data, consumers in other services.
## Phase 6 — Run it · ~10 min
- Run the targeted tests for the changed area (not just CI-green — read what they assert).
- Exercise the changed path once yourself — app, REPL, or debugger. Step through the load-bearing function with real values.
- Trigger the error path at least once. Error handling is the least-reviewed, most-shipped-broken code there is.
## Phase 7 — Step back (the holistic pass) · ~5 min
Close the diff. Answer from memory:
- Does the *whole* accomplish the stated intent — and nothing meaningfully more or less?
- Is there a simpler design that occurs to you now that you understand it? (Mention it; don't demand it.)
- What breaks at 10× load / 10× data / concurrent use?
- What is *not* tested that worries you?
- Will someone (you) understand this in six months without the PR description?
If you can't answer these, the review isn't done — or the PR is too big, which is feedback in itself.
## Phase 8 — Write the review
1. **Lead with your understanding**: "My read: this does X by changing Y, with Z as the tricky part." If you've misread it, the author corrects the model, not just the comments.
2. Separate **blocking** from **take-it-or-leave-it**, explicitly.
3. Surface Phase 1 surprises and Phase 5 blast-radius findings — those are your highest-value comments, and the ones a hunk-by-hunk reviewer can't make.
---
### Calibration
- Small PR (< ~200 lines): Phases 1, 3, 5, 8 — maybe 15 minutes total.
- Large PR (> ~800 substantive lines): do Phases 1–3, then tell the author what you're reviewing first and what needs a second pass — or ask for a split. Reviewing 2,000 lines in one sitting produces approval, not review.
- Repeat offender friction (huge PRs, empty descriptions, tangled commits): fix upstream — PR templates, stacked PRs, self-review annotations by the author.
---
## Appendix — this workflow in *your* Neovim
Grounded in your LazyVim config after the PR-review upgrades (verified against your pinned plugin sources plus the implemented brief). Two facts shape everything: your `<leader>g*` keys dispatch **git vs. jj** per-repo via `plugins.custom.vcs.is_jj()`, and your localleader is `\`, which all octo.nvim bindings hang off.
### Getting the branch (Phase 2)
- **git repo**: `gh pr checkout <n>` in a terminal (`<c-/>` toggles one at repo root).
- **jj-colocated repo**: stay jj-native — `jj git fetch`, then `jj new <head-branch>@origin`. Your working copy `@` is now an empty child of the PR head, which makes "diff against trunk" mean exactly "the PR."
### The map (Phase 3–4)
`<leader>gR` is the review map: diffview against the resolved base (`origin/HEAD`, falling back to `origin/main`) in git repos, `Jdiff trunk()` in jj repos. Keep the distinction deliberate: `<leader>gd` stays your *local working-tree* diff — mid-review it shows nothing on a clean checkout, and that's correct, not broken. `<leader>gD` gives the same PR-vs-base content as a grouped hunks picker (it now shares `gR`'s base resolution), useful when you want to sample hunks rather than walk files.
### Reading path, blast radius (Phase 4–5)
`<leader>ga` seeds the arglist with the PR's changed files (both VCSs); prune and reorder with `:args` into the briefing's dependency order, then walk with `:n` / `:prev`. For blast radius, `gr` (references) populates the quickfix list — `]q` / `[q` walk it **everywhere now, including inside octo review tabs**, reopen with `<leader>xq`. Any snacks picker sends results to quickfix with `<c-q>`. Project grep is `<leader>sg` (root) / `<leader>sG` (cwd); "is the old pattern really gone" sweeps go through `<leader>sr` (grug-far). History and blame dispatch too: `<leader>gf` / `<leader>gF` (file / repo history), `<leader>gl` / `<leader>gL` (log — `J log` in jj), `<leader>gb` (blame line — jj annotate in jj repos).
### The GitHub layer — octo.nvim (Phase 8)
Octo now uses the snacks picker, and `<localleader>` = `\`. `:Octo pr edit <n>` opens the PR buffer (description, threads — Phase 1 orientation without leaving nvim). `\vs` starts the review, `\vr` resumes a pending one. In the review diff: `\ca` adds a comment (visual mode gives a range comment), `\sa` a suggestion, `]t` / `[t` jump threads, `]u` / `[u` jump unviewed files, `\e` focuses the file panel, `\b` toggles it. Changed-file navigation is `]o` / `[o` (`]O` / `[O` for last/first) — remapped across review diff, thread view, and file panel so the quickfix keys stay yours. `\vs` again opens the submit window: `<C-a>` approve, `<C-m>` comment, `<C-r>` request changes; `<C-c>` closes the review tab.
### The briefing pane
`<leader>aR` prompts for a PR number and launches your right-side `claude` terminal with `/review-pr <n>` already running; `<leader>at` toggles the same pane bare. Keep it open as the map and walk the reading path in the main window — `<leader>ga` then `:args` reorder takes the briefing's file list into the walk.
### Cheat sheet
| Task | Git repo | jj-colocated repo |
|---|---|---|
| Orient (desc + threads) | `:Octo pr edit <n>` | same |
| Get the branch | `gh pr checkout <n>` | `jj git fetch` → `jj new <branch>@origin` |
| PR-vs-base map | `<leader>gR` | same (dispatched → `Jdiff trunk()`) |
| PR hunks picker | `<leader>gD` | same (dispatched) |
| Local working-tree diff | `<leader>gd` | same (dispatched) |
| Changed files → arglist | `<leader>ga` | same (dispatched) |
| Reading path | `:args` reorder → `:n` / `:prev` | same |
| Status | `<leader>gs` | same (dispatched) |
| File / repo history | `<leader>gf` / `<leader>gF` | same (dispatched) |
| Log / blame line | `<leader>gl`, `<leader>gL` / `<leader>gb` | same (→ `J log` / jj annotate) |
| Definition / references | `gd` / `gr` (→ quickfix) | same |
| Implementation / type / hover | `gI` / `gy` / `K` | same |
| Walk quickfix | `]q` / `[q` · list: `<leader>xq` | same |
| Picker → quickfix | `<c-q>` in any snacks picker | same |
| Project grep | `<leader>sg` (root) / `<leader>sG` (cwd) | same |
| Old-pattern sweep | `<leader>sr` (grug-far) | same |
| Start / resume review | `\vs` / `\vr` in PR buffer | same |
| Comment / suggestion at line | `\ca` / `\sa` (visual = range) | same |
| Jump threads / unviewed files | `]t` `[t` / `]u` `[u` | same |
| Changed-file nav (octo) | `]o` `[o` · `]O` `[O` last/first | same |
| Submit | `\vs` → `<C-a>` / `<C-m>` / `<C-r>` | same |
| Close review tab | `<C-c>` | same |
| Launch review briefing (PR #) | `<leader>aR` | same |
| Briefing pane (toggle) | `<leader>at` | same |
*(`\` = your localleader. Octo rows apply inside octo buffers; `]q`/`[q` are quickfix everywhere, including review tabs.)*