@hasna/hooks 0.5.0 → 0.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +81 -8
  2. package/bin/index.js +1756 -382
  3. package/bin/serve.js +5258 -0
  4. package/dist/cf/provision.d.ts +24 -0
  5. package/dist/config.d.ts +18 -0
  6. package/dist/db/legacy-import.d.ts +1 -1
  7. package/dist/db/migrations/004_hooks_table.d.ts +9 -0
  8. package/dist/db/pg-migrations.d.ts +1 -1
  9. package/dist/db/storage-sync.d.ts +26 -6
  10. package/dist/index.d.ts +18 -2
  11. package/dist/index.js +5351 -286
  12. package/dist/lib/custom-install.d.ts +19 -0
  13. package/dist/lib/manifest.d.ts +70 -0
  14. package/dist/lib/resolve.d.ts +19 -0
  15. package/dist/lib/run.d.ts +37 -0
  16. package/dist/lib/store.d.ts +69 -0
  17. package/dist/lib/sync.d.ts +35 -0
  18. package/dist/serve.d.ts +36 -0
  19. package/dist/storage.d.ts +2 -2
  20. package/dist/storage.js +133 -42
  21. package/hooks/codewith-native-common.test.ts +15 -2
  22. package/hooks/hook-scanoutput/README.md +151 -0
  23. package/hooks/hook-scanoutput/package.json +12 -0
  24. package/hooks/hook-scanoutput/src/hook.test.ts +217 -0
  25. package/hooks/hook-scanoutput/src/hook.ts +319 -0
  26. package/hooks/hook-workspace-repos-guard/README.md +63 -0
  27. package/hooks/hook-workspace-repos-guard/package.json +12 -0
  28. package/hooks/hook-workspace-repos-guard/src/hook.test.ts +312 -0
  29. package/hooks/hook-workspace-repos-guard/src/hook.ts +352 -0
  30. package/hooks/hook-workspace-repos-guard/tsconfig.json +21 -0
  31. package/hooks/mention-context/README.md +109 -0
  32. package/hooks/mention-context/package.json +9 -0
  33. package/hooks/mention-context/src/hasna-mention-context.py +1218 -0
  34. package/hooks/mention-context/src/hasna-mention-warm.py +521 -0
  35. package/hooks/mention-context/src/hook.test.ts +68 -0
  36. package/hooks/mention-context/src/test_run_capture.py +345 -0
  37. package/package.json +9 -5
@@ -0,0 +1,151 @@
1
+ # hook-scanoutput
2
+
3
+ Scans tool output for credential shapes and writes a warning. **Detection, not prevention.**
4
+
5
+ ## What it does, and what it cannot do
6
+
7
+ `scanoutput` is a `PostToolUse` hook. It reads the tool result, hands the text to
8
+ `scanInputExposures` from `@hasna/secrets`, and writes a warning to stderr when a
9
+ credential shape is present. It always answers `{ "continue": true }` and never
10
+ blocks anything.
11
+
12
+ **It cannot redact or prevent anything, and it must never be described as though it
13
+ can.** PostToolUse fires after the tool has run and after its result exists.
14
+ Measured across the 48-hook catalog at `@hasna/hooks` 0.5.0, no output-rewriting
15
+ field exists on any event — `continue`, `decision`, `hookSpecificOutput`,
16
+ `additionalContext`, `suppressOutput` and `permissionDecision` are the entire
17
+ surface, and `suppressOutput` hides the hook's own stdout rather than the tool's.
18
+ By the time this hook sees a credential, that credential is already in the
19
+ transcript.
20
+
21
+ It is worth running because today nothing notices at all. A credential emitted by a
22
+ third-party tool into ordinary output is indistinguishable from ordinary output
23
+ until a human happens to look. This turns a silent exposure into a recorded one
24
+ with a named blast radius.
25
+
26
+ A guard advertised as preventing a leak it only reports afterwards retires the
27
+ worry without retiring the exposure. That is worse than no guard, which is why the
28
+ wording above is load-bearing rather than modesty.
29
+
30
+ ## The field name is `tool_response`, and reading the wrong one makes this silent
31
+
32
+ This hook reads `tool_response`, falling back to `tool_output`. That is not defensive
33
+ padding — it is the difference between a working guard and one that reports clean
34
+ forever.
35
+
36
+ Verified against the Claude Code 2.1.226 binary on 2026-08-10. Its embedded schema:
37
+
38
+ ```json
39
+ "tool_input": { "file_path": "/path/to/file.txt", "content": "..." },
40
+ "tool_response": { "success": true } // PostToolUse only
41
+ ```
42
+
43
+ and its implementation builds the hook input as:
44
+
45
+ ```js
46
+ hook_event_name:"PostToolUse", tool_name:e, tool_input:r, tool_response:n, tool_use_id:t, duration_ms:l
47
+ ```
48
+
49
+ Counted in the same binary: `tool_response` 21 occurrences, `tool_output` 2 — and both
50
+ of those are the telemetry event name `tengu_dead_probe_hook_updated_mcp_tool_output`,
51
+ neither adjacent to `PostToolUse`. Control: a deliberately absent string returns 0.
52
+
53
+ **Every other hook in this catalog reads `tool_output` only.** On Claude Code they
54
+ receive `undefined`. For a scanner that means an empty input, and an empty input scores
55
+ as clean — a guard that has never looked at anything and says so in the voice of a
56
+ guard that has. This hook was written that way first and the self-review caught it
57
+ before merge; the demonstration is one line:
58
+
59
+ ```
60
+ PRE-FIX reads tool_output -> undefined => scans NOTHING, reports clean
61
+ POST-FIX reads tool_response -> got 131 chars => scans it
62
+ ```
63
+
64
+ Tests lock both field names and the preference order.
65
+
66
+ ## Install
67
+
68
+ ```bash
69
+ hooks install scanoutput
70
+ ```
71
+
72
+ Never hand-edit `settings.json`. The package owns the wiring; a hand-edit reverts
73
+ on the next managed render and looks done while silently not being.
74
+
75
+ ## Behaviour
76
+
77
+ | Situation | Outcome | Warning |
78
+ |---|---|---|
79
+ | Credential shape found | `found` | detector, severity, location, redacted preview |
80
+ | Output read whole, nothing found | `clean` | silent |
81
+ | Output empty | `empty` | silent |
82
+ | Output truncated, skipped, or 0 bytes read | `unscanned` | says UNSCANNED explicitly |
83
+ | Scanner missing or throws | `error` | says it could not scan |
84
+
85
+ `unscanned` exists so that "I looked at all of it and it was clean" and "I could not
86
+ look" do not share an answer. A guard that reports those identically is the defect
87
+ this hook was built to notice.
88
+
89
+ Findings carry only the scanner's own redacted preview. Values never appear.
90
+
91
+ ## It reports every severity, deliberately
92
+
93
+ The obvious way to cut false positives is to drop `severity: medium` and keep only
94
+ `high`. **Do not.** Measured on station01, 2026-08-10: a `bash -x` trace emitted 25
95
+ live fleet API keys into ordinary tool output, and all 25 were
96
+ `detector=credential_assignment severity=medium`. A high-only filter sees none of
97
+ them. On the same corpora, `severity=high` matched zero findings — true or false.
98
+
99
+ ## Measured false-positive rate
100
+
101
+ 155 streams of real tool output, 1.67 MB, from 78 commands on station01,
102
+ 2026-08-10. Scanned in-process with the same scanner the hook uses.
103
+
104
+ | Corpus | Streams | Bytes | Findings |
105
+ |---|---|---|---|
106
+ | A — ordinary commands (git, ls, npm, gh, Hasna CLIs, curl, source, manifests) | 80 | 771,142 | 0 |
107
+ | B — high-entropy non-credential (lockfiles, minified bundles, base64, UUIDs, hex tokens, SHAs) | 36 | 712,640 | 0 |
108
+ | C — credential-shaped docs (READMEs, `.env.example`, `--help`, shell traces) | 39 | 181,803 | 6 |
109
+
110
+ **6 false positives, all `credential_assignment`, all `severity: medium`:**
111
+
112
+ - 2 × `*_DATABASE_URL` with a connection-string value in a README — documentation
113
+ - 3 × from `.env.example` files — example values, 9 and 23 characters
114
+ - 1 × `PWD=` with an absolute path — `PWD` is in the detector's variable-name list
115
+
116
+ That is 4 of 155 streams firing (2.6%), and zero on the 116 streams of ordinary and
117
+ high-entropy output. The concentration is the useful part: false positives cluster
118
+ in documentation and help text, not in command output.
119
+
120
+ Controls, both directions, on the same harness: three synthetic credential shapes
121
+ injected into three of corpus A's streams were detected on exactly those three and
122
+ nowhere else; the un-injected copy of the same corpus returned zero.
123
+
124
+ The scanner already suppresses obvious placeholders — values starting `$` or `<`,
125
+ containing `...`, all-`*x_-`, or containing `example`/`placeholder`/`redacted`/
126
+ `changeme`/`dummy`. The 6 above are what survives that.
127
+
128
+ **The axis this corpus does not vary:** sustained output from CI systems, container
129
+ runtimes and orchestrators, which print `NAME=value` environment blocks as a matter
130
+ of course. `PWD=` firing once here suggests that population would fire much more
131
+ often, and it is not represented above. Measure the fire rate there before anyone
132
+ proposes escalating this hook from warn to block.
133
+
134
+ ## Cost
135
+
136
+ The scan runs in-process. Measured on station01, 2026-08-10:
137
+
138
+ | Path | Cost |
139
+ |---|---|
140
+ | `secrets scan input` as a CLI spawn | 1.28–1.81 s for 11 KB; 1.35–1.65 s for 2.2 MB |
141
+ | In-process `scanInputExposures` | 25 ms import, then 1.4–7.3 ms for 11 KB, 64–90 ms for 2.2 MB |
142
+
143
+ The CLI cost is process startup and is flat in payload — 198× the input costs the
144
+ same. Removing the spawn removes that term, after which cost becomes roughly linear
145
+ in payload, so the `MAX_SCAN_BYTES` ceiling does buy something in-process where it
146
+ would have bought nothing via the CLI.
147
+
148
+ The dominant remaining cost is not this hook: dispatching any hook through
149
+ `hooks run <name>` measured 2.38–2.99 s on this box. That is a property of hook
150
+ dispatch, paid by every installed hook, and it is the number to look at before
151
+ installing this on a hot matcher.
@@ -0,0 +1,12 @@
1
+ {
2
+ "name": "hook-scanoutput",
3
+ "version": "0.1.0",
4
+ "description": "Claude Code hook that scans tool output for credential shapes and warns — detection after the fact, never prevention",
5
+ "type": "module",
6
+ "main": "./src/hook.ts",
7
+ "scripts": {
8
+ "typecheck": "tsc --noEmit"
9
+ },
10
+ "author": "Hasna",
11
+ "license": "Apache-2.0"
12
+ }
@@ -0,0 +1,217 @@
1
+ import { describe, expect, test } from "bun:test";
2
+ import {
3
+ analyzeToolOutput,
4
+ buildHookOutput,
5
+ extractToolOutputText,
6
+ formatWarning,
7
+ MAX_SCAN_BYTES,
8
+ SCAN_TIMEOUT_MS,
9
+ type ScanFn,
10
+ type ScanOutcome,
11
+ } from "./hook";
12
+
13
+ /**
14
+ * A stand-in for `scanInputExposures` from @hasna/secrets.
15
+ *
16
+ * The scanner itself is already exercised both ways in its own repo; what these
17
+ * tests own is the hook's behaviour around it — that a finding is surfaced, that
18
+ * a clean scan is silent, that an unreadable scan is NOT reported as clean, and
19
+ * that nothing the scanner does can stop the session.
20
+ */
21
+ function fakeScanner(result: Partial<ReturnType<ScanFn>> = {}): ScanFn {
22
+ return () =>
23
+ ({
24
+ schema: "open-secrets.exposure-scan.v1",
25
+ version: 1,
26
+ source: "input",
27
+ root: "<stdin>",
28
+ redacted: true,
29
+ limits: { findings: 50 },
30
+ stats: { filesScanned: 1, filesSkipped: 0, bytesScanned: 100, errors: [], skipped: [] },
31
+ truncated: false,
32
+ findings: [],
33
+ ...result,
34
+ }) as ReturnType<ScanFn>;
35
+ }
36
+
37
+ const finding = (over: Record<string, unknown> = {}) => ({
38
+ id: "f1",
39
+ source: "input",
40
+ detector: "credential_assignment",
41
+ severity: "medium",
42
+ path: "<stdin>",
43
+ line: 12,
44
+ column: 5,
45
+ preview: "HASNA_TODOS_API_KEY=***REDACTED***",
46
+ evidencePath: "<stdin>:12",
47
+ ...over,
48
+ });
49
+
50
+ describe("hook-scanoutput / extractToolOutputText", () => {
51
+ /**
52
+ * The field name is the whole ballgame. Verified against the Claude Code 2.1.226
53
+ * binary on 2026-08-10: its embedded schema reads
54
+ * "tool_response": { "success": true } // PostToolUse only
55
+ * and its implementation builds
56
+ * hook_event_name:"PostToolUse", tool_name:e, tool_input:r, tool_response:n
57
+ * `tool_response` 21 occurrences, `tool_output` 2 — both of those the telemetry
58
+ * name tengu_dead_probe_hook_updated_mcp_tool_output, neither near PostToolUse.
59
+ *
60
+ * Reading only `tool_output` yields undefined on Claude Code, which this hook
61
+ * would score as an empty input and report clean. These tests lock the field.
62
+ */
63
+ test("reads tool_response, the field Claude Code actually sends", () => {
64
+ expect(extractToolOutputText({ tool_response: "hello" })).toBe("hello");
65
+ });
66
+
67
+ test("reads a structured tool_response", () => {
68
+ const text = extractToolOutputText({ tool_response: { stdout: "out-part", stderr: "err-part" } });
69
+ expect(text).toContain("out-part");
70
+ expect(text).toContain("err-part");
71
+ });
72
+
73
+ test("still reads tool_output, the older catalog convention", () => {
74
+ expect(extractToolOutputText({ tool_output: "hello" })).toBe("hello");
75
+ });
76
+
77
+ test("concatenates the usual structured stdout/stderr fields", () => {
78
+ const text = extractToolOutputText({ tool_output: { stdout: "out-part", stderr: "err-part" } });
79
+ expect(text).toContain("out-part");
80
+ expect(text).toContain("err-part");
81
+ });
82
+
83
+ test("prefers tool_response when both are present", () => {
84
+ expect(extractToolOutputText({ tool_response: "from-response", tool_output: "from-output" })).toBe("from-response");
85
+ });
86
+
87
+ test("falls back to serialising an unrecognised output shape rather than skipping it", () => {
88
+ // An output shape this hook does not know must not read as 'no output'.
89
+ const text = extractToolOutputText({ tool_response: { unexpectedField: "SOMETHING-IN-HERE" } });
90
+ expect(text).toContain("SOMETHING-IN-HERE");
91
+ });
92
+
93
+ test("returns empty string when there is no output at all", () => {
94
+ expect(extractToolOutputText({})).toBe("");
95
+ expect(extractToolOutputText({ tool_output: undefined })).toBe("");
96
+ expect(extractToolOutputText({ tool_response: undefined })).toBe("");
97
+ });
98
+ });
99
+
100
+ describe("hook-scanoutput / analyzeToolOutput", () => {
101
+ test("FIRES: a finding in tool output is surfaced with its detector", () => {
102
+ const outcome = analyzeToolOutput("anything", fakeScanner({ findings: [finding()] as never }));
103
+ expect(outcome.status).toBe("found");
104
+ expect(outcome.findingCount).toBe(1);
105
+ expect(outcome.detectors).toEqual(["credential_assignment"]);
106
+ });
107
+
108
+ test("SILENT: ordinary output produces no warning", () => {
109
+ const outcome = analyzeToolOutput("ordinary tool output", fakeScanner());
110
+ expect(outcome.status).toBe("clean");
111
+ expect(outcome.findingCount).toBe(0);
112
+ expect(formatWarning(outcome)).toBe("");
113
+ });
114
+
115
+ test("a clean verdict requires bytes to have actually been read", () => {
116
+ // Zero bytes scanned is not a clean scan; it is a scan that did not happen.
117
+ const outcome = analyzeToolOutput(
118
+ "x",
119
+ fakeScanner({ stats: { filesScanned: 1, filesSkipped: 0, bytesScanned: 0, errors: [], skipped: [] } as never }),
120
+ );
121
+ expect(outcome.status).toBe("unscanned");
122
+ expect(formatWarning(outcome)).toContain("UNSCANNED");
123
+ });
124
+
125
+ test("a truncated scan is reported as unscanned, never as clean", () => {
126
+ const outcome = analyzeToolOutput("x", fakeScanner({ truncated: true, truncatedReason: "max_bytes" } as never));
127
+ expect(outcome.status).toBe("unscanned");
128
+ });
129
+
130
+ test("a skipped input is reported as unscanned, never as clean", () => {
131
+ const outcome = analyzeToolOutput(
132
+ "x",
133
+ fakeScanner({
134
+ stats: {
135
+ filesScanned: 0,
136
+ filesSkipped: 1,
137
+ bytesScanned: 0,
138
+ errors: [],
139
+ skipped: [{ path: "<stdin>", reason: "max_file_bytes", bytes: 9_000_000 }],
140
+ } as never,
141
+ }),
142
+ );
143
+ expect(outcome.status).toBe("unscanned");
144
+ });
145
+
146
+ test("empty tool output is skipped outright and is not called clean", () => {
147
+ const outcome = analyzeToolOutput("", fakeScanner());
148
+ expect(outcome.status).toBe("empty");
149
+ expect(formatWarning(outcome)).toBe("");
150
+ });
151
+
152
+ test("bounds the scan explicitly, because it sits on a tool call's critical path", () => {
153
+ // The Claude installer target writes no `timeout` into settings.json, so the
154
+ // wiring supplies no outer bound and the scanner's own default is 10s.
155
+ let seen: { maxBytes?: number; timeoutMs?: number } | undefined;
156
+ const spy: ScanFn = (opts) => {
157
+ seen = opts;
158
+ return fakeScanner()(opts);
159
+ };
160
+ analyzeToolOutput("x", spy);
161
+ expect(seen?.maxBytes).toBe(MAX_SCAN_BYTES);
162
+ expect(seen?.timeoutMs).toBe(SCAN_TIMEOUT_MS);
163
+ expect(SCAN_TIMEOUT_MS).toBeLessThan(10_000);
164
+ });
165
+
166
+ test("FAIL-OPEN: a scanner that throws degrades to a reported error, never a crash", () => {
167
+ const outcome = analyzeToolOutput("x", (() => {
168
+ throw new Error("scanner exploded");
169
+ }) as ScanFn);
170
+ expect(outcome.status).toBe("error");
171
+ expect(outcome.error).toContain("scanner exploded");
172
+ expect(formatWarning(outcome)).toContain("could not scan");
173
+ });
174
+
175
+ test("reports every severity, because high-only would miss the measured leak class", () => {
176
+ // Measured on 2026-08-10: 25 real fleet API keys in a bash -x trace were ALL
177
+ // detector=credential_assignment severity=medium. A high-only filter sees none of them.
178
+ const outcome = analyzeToolOutput(
179
+ "x",
180
+ fakeScanner({ findings: [finding(), finding({ id: "f2", severity: "high", detector: "aws_access_key_id" })] as never }),
181
+ );
182
+ expect(outcome.findingCount).toBe(2);
183
+ expect(outcome.detectors).toContain("credential_assignment");
184
+ expect(outcome.detectors).toContain("aws_access_key_id");
185
+ });
186
+ });
187
+
188
+ describe("hook-scanoutput / formatWarning", () => {
189
+ test("names the detector and location but never claims prevention", () => {
190
+ const outcome = analyzeToolOutput("x", fakeScanner({ findings: [finding()] as never }));
191
+ const warning = formatWarning(outcome);
192
+ expect(warning).toContain("credential_assignment");
193
+ expect(warning).toContain("<stdin>:12");
194
+ // This guard reports an exposure that has ALREADY been persisted. Wording that
195
+ // implies it stopped anything would retire the worry without retiring the risk.
196
+ expect(warning.toLowerCase()).not.toContain("blocked");
197
+ expect(warning.toLowerCase()).not.toContain("prevented");
198
+ expect(warning.toLowerCase()).toContain("already");
199
+ });
200
+
201
+ test("carries only the scanner's redacted preview, never a raw value", () => {
202
+ const outcome = analyzeToolOutput(
203
+ "x",
204
+ fakeScanner({ findings: [finding({ preview: "KEY=***REDACTED***" })] as never }),
205
+ );
206
+ expect(formatWarning(outcome)).toContain("***REDACTED***");
207
+ });
208
+ });
209
+
210
+ describe("hook-scanoutput / buildHookOutput", () => {
211
+ test("ALWAYS continues, on every outcome, including a scanner error", () => {
212
+ const outcomes: ScanOutcome["status"][] = ["clean", "found", "unscanned", "empty", "error"];
213
+ for (const status of outcomes) {
214
+ expect(buildHookOutput({ status, findingCount: 0, detectors: [], previews: [] }).continue).toBe(true);
215
+ }
216
+ });
217
+ });