@sriinnu/omit 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.clinerules/omit.md +1 -1
- package/.cursor/rules/omit.mdc +1 -1
- package/.github/copilot-instructions.md +15 -0
- package/.github/workflows/publish.yml +69 -0
- package/.github/workflows/test.yml +40 -0
- package/.windsurf/rules/omit.md +1 -1
- package/AGENTS.md +11 -2
- package/README.md +72 -15
- package/action.yml +18 -1
- package/bench/run.mjs +25 -4
- package/bin/omit.mjs +518 -58
- package/hooks/command-sentinel.mjs +39 -11
- package/hooks/dep-sentinel.mjs +184 -30
- package/hooks/final-draft-gate.mjs +231 -37
- package/hooks/hazard-sentinel.mjs +157 -39
- package/hooks/hooks.json +18 -1
- package/hooks/leak-sentinel.mjs +55 -0
- package/hooks/lint-sentinel.mjs +34 -11
- package/lib/danger.mjs +9 -1
- package/lib/deps.mjs +341 -45
- package/lib/exec.mjs +18 -0
- package/lib/git.mjs +107 -0
- package/lib/hazards.mjs +29 -4
- package/lib/leaks.mjs +313 -0
- package/lib/lint.mjs +85 -5
- package/lib/receipts.mjs +356 -0
- package/package.json +5 -1
- package/skills/omit/SKILL.md +25 -13
package/lib/leaks.mjs
ADDED
|
@@ -0,0 +1,313 @@
|
|
|
1
|
+
// Secret-leak detection: commands that print an EXISTING secret's raw value to
|
|
2
|
+
// stdout — keychain/vault reads, env dumps, credential-file cats, or a live key
|
|
3
|
+
// typed straight into a command line. Different from hazards.mjs (which catches
|
|
4
|
+
// secrets typed INTO files): this catches secrets pulled OUT of a store (or
|
|
5
|
+
// pasted directly) and echoed to the terminal, where they land in the agent's
|
|
6
|
+
// own transcript. Suppress a reviewed command with: # omit-allow: <reason>
|
|
7
|
+
//
|
|
8
|
+
// Known limitation, by design rather than oversight: this is a text-regex
|
|
9
|
+
// scanner with no shell evaluation. A command that hides a matched substring
|
|
10
|
+
// behind computation — base64/hex decode then `bash -c`/`eval`, ANSI-C
|
|
11
|
+
// hex-escaped filenames ($'\x2e\x65...'), a secret built by concatenating
|
|
12
|
+
// shell variables before use — produces no findings even though it's
|
|
13
|
+
// semantically identical to a command this file does catch verbatim. Closing
|
|
14
|
+
// that class requires actually evaluating the command (a shell interpreter or
|
|
15
|
+
// AST parser), which is out of scope for a lightweight PreToolUse heuristic;
|
|
16
|
+
// raising this bar is the goal, not a guarantee against a deliberately evasive
|
|
17
|
+
// agent or a malicious actor steering one.
|
|
18
|
+
import { SECRET_RULES, isAllowSuppressed } from './hazards.mjs'
|
|
19
|
+
|
|
20
|
+
// A redirect counts as "suppressed" if the command's own stdout doesn't reach
|
|
21
|
+
// the terminal/tool output at all — `&>/dev/null`, `>/dev/null`, `>>logfile`,
|
|
22
|
+
// `> anyfile`. The mission here is transcript-scoped ("don't print a secret
|
|
23
|
+
// into the conversation"), not disk-scoped, so redirecting to a real file
|
|
24
|
+
// counts as safe even though it still leaves a plaintext secret on disk — a
|
|
25
|
+
// different, lesser problem than this hook exists to catch.
|
|
26
|
+
// `2>/dev/null` alone only hides stderr — the secret still prints on success —
|
|
27
|
+
// so a redirect must not be preceded by a digit other than `1` to count.
|
|
28
|
+
const STDOUT_REDIRECTED_RE = /&>\s*\S+|>>\s*\S+|(?:(?<![0-9])>|1>)\s*\S+/
|
|
29
|
+
|
|
30
|
+
// Piping into one of these consumes the value without ever printing it back —
|
|
31
|
+
// clipboard tools and the --password-stdin/--from-file=/dev/stdin idiom exist
|
|
32
|
+
// specifically so a secret never has to be echoed or shown in `ps`.
|
|
33
|
+
const SAFE_SINK_RE =
|
|
34
|
+
/\|\s*(?:pbcopy\b|xclip\b|wl-copy\b|clip(?:\.exe)?\b|docker\s+login\b[^\n]*--password-stdin\b|kubectl\s+create\s+secret\b[^\n]*--from-file=\/dev\/stdin\b)/i
|
|
35
|
+
|
|
36
|
+
// Naive split on top-level statement separators (&&, ||, ;, newline) — NOT
|
|
37
|
+
// pipe, since a pipe chain is one data-flow unit we reason about via
|
|
38
|
+
// SAFE_SINK_RE instead of splitting apart. Doesn't respect quoting or subshell
|
|
39
|
+
// nesting: a heuristic scoping improvement over checking the whole command at
|
|
40
|
+
// once, not a shell parser. Known gap: a redirect that lives after `done` on a
|
|
41
|
+
// multi-line `for ...; do ...; done > file` loop isn't attributed back to the
|
|
42
|
+
// loop body, since `;`/newline splits the loop apart from its own redirect —
|
|
43
|
+
// that idiom will false-positive here; reach for `# omit-allow:` if you hit it.
|
|
44
|
+
const splitClauses = (command) => command.split(/&&|\|\||;|\n/)
|
|
45
|
+
|
|
46
|
+
const SAFE_ENV_SUFFIXES = ['.example', '.sample', '.template', '.dist']
|
|
47
|
+
// Formats that are never a credential store themselves, even when the
|
|
48
|
+
// filename literally contains "secrets" (a doc *about* secrets management).
|
|
49
|
+
const DOC_EXTENSIONS = ['.md', '.markdown', '.mdx', '.rst', '.adoc']
|
|
50
|
+
// Standard Let's Encrypt / ACME naming for the PUBLIC half of a TLS chain —
|
|
51
|
+
// only the private key (privkey.pem, key.pem, *.key) is actually sensitive.
|
|
52
|
+
const PUBLIC_CERT_BASENAMES = ['fullchain.pem', 'chain.pem', 'cert.pem', 'certificate.pem', 'ca.pem', 'ca-bundle.pem', 'ca-cert.pem', 'ca-certificates.pem']
|
|
53
|
+
|
|
54
|
+
const KEYCHAIN_DUMP_RE = /\bsecurity\s+dump-keychain\b/
|
|
55
|
+
const KEYCHAIN_READ_RE = /\bsecurity\s+find-(?:generic|internet)-password\b[^|;&\n]*-w\b/
|
|
56
|
+
// Captures the redirect target directly (not by re-embedding STDOUT_REDIRECTED_RE,
|
|
57
|
+
// which already consumes its own \S+ target internally — nesting it here would
|
|
58
|
+
// require a second, spurious \S+ after an already-complete match).
|
|
59
|
+
// Target capture excludes ;&| (not just whitespace) so it doesn't swallow a
|
|
60
|
+
// following statement separator when there's no space before it, e.g.
|
|
61
|
+
// `-w > /tmp/x.txt;cat /tmp/x.txt` — a bare \S+ would capture "/tmp/x.txt;"
|
|
62
|
+
// and then never match that string against the later bare "/tmp/x.txt".
|
|
63
|
+
const KEYCHAIN_REDIRECTED_RE = /-w\b[^|;&\n]*(?:&>|>>|(?:(?<![0-9])>|1>))\s*([^\s;&|]+)/
|
|
64
|
+
// \w+ and [^)]* below are bounded (not unbounded) on purpose: an unbounded
|
|
65
|
+
// quantifier that can match a long run of word characters, tried starting at
|
|
66
|
+
// every position in a long non-matching string (e.g. a long token with no `=`
|
|
67
|
+
// anywhere), backtracks character-by-character at each start position — O(n)
|
|
68
|
+
// positions x O(n) backtrack = O(n^2). A real shell variable/function name or
|
|
69
|
+
// a security-call argument list is never remotely close to these bounds.
|
|
70
|
+
const ASSIGN_CAPTURE_RE = /(\w{1,64})=\$\(\s*security\s+find-(?:generic|internet)-password\b[^)]{0,2048}-w[^)]{0,2048}\)/
|
|
71
|
+
const TEST_CAPTURE_RE = /(?:-n|\bif\b|\btest\b)\s+"?\$\(\s*security\s+find-(?:generic|internet)-password\b[^)]{0,2048}-w[^)]{0,2048}\)"?/
|
|
72
|
+
|
|
73
|
+
// Left boundary is explicit (whitespace/start/;&|() rather than bare \b, which
|
|
74
|
+
// would match "env" inside a filename like ".env" (a "." IS a \b boundary).
|
|
75
|
+
const ENV_BARE_RE = /(?:^|[\s;&|(])(env|printenv)\b\s*($|[;&])/
|
|
76
|
+
const PRINTENV_VAR_RE = /(?:^|[\s;&|(])printenv\s+[A-Za-z_][A-Za-z0-9_]*\b/
|
|
77
|
+
const ENV_GREP_RE = /(?:^|[\s;&|(])(?:env|printenv)\b\s*\|\s*(?:e|f)?grep\b([^|;&\n]*)/i
|
|
78
|
+
const ENV_SECRET_WORD_RE = /^(api|access|secret|private|auth|db|aws|gcp)?(key|token|secret|password|credential)s?$/i
|
|
79
|
+
// Connection-string-style var names (DATABASE_URL, MONGODB_URI, REDIS_URL,
|
|
80
|
+
// FOO_DSN) commonly embed a username:password, unlike a generic *_URL such as
|
|
81
|
+
// API_URL or WEBHOOK_URL — so this requires a recognized data-store prefix
|
|
82
|
+
// rather than treating any bare "url"/"uri" as sensitive.
|
|
83
|
+
const CONNECTION_STRING_RE = /\b(?:database|db|mongo|mongodb|redis|postgres|postgresql|mysql|amqp|rabbitmq)_?(?:url|uri)\b|\bdsn\b|\bconnection_?string\b/i
|
|
84
|
+
// -q/-c/-l/-L never print the matched line's value, only a boolean/count/filename.
|
|
85
|
+
const GREP_SAFE_FLAGS_RE = /(?:^|\s)-[a-zA-Z]*[qcl][a-zA-Z]*(?:\s|$)|--quiet\b|--count\b|--files-with-matches\b|--files-without-match\b/
|
|
86
|
+
|
|
87
|
+
// grep-family tools print matching lines from a file's content just like cat
|
|
88
|
+
// does when given a credential file as the target — included here (unlike in
|
|
89
|
+
// SEARCH_TOOL_RE below, which is about a secret-SHAPED STRING used as a search
|
|
90
|
+
// pattern, a different concern).
|
|
91
|
+
const READ_CMDS_RE = /\b(cat|less|more|head|tail|bat|grep|egrep|fgrep|rg|ag|ack)\b/
|
|
92
|
+
const GREP_FAMILY_RE = /\b(?:grep|egrep|fgrep|rg|ag|ack)\b/
|
|
93
|
+
const SECRET_FILE_RE = /(?<![\w.-])(?:[\w./~-]*\/)?(\.env(?:\.[\w-]+)?|credentials(?:\.json)?|secrets\.[\w-]+|[\w.-]+\.pem|[\w.-]*_rsa|id_rsa)(?![\w.-])/gi
|
|
94
|
+
// Cap on how long a single whitespace-separated token gets tested against
|
|
95
|
+
// SECRET_FILE_RE. Without this, a single very long non-matching argument (a
|
|
96
|
+
// long path-like string with no real extension) drives the regex — its
|
|
97
|
+
// optional path-prefix combined with re-scanning from every failed position —
|
|
98
|
+
// into ~O(n^2) time: a plain `cat` of a ~200KB argument was measured to hang
|
|
99
|
+
// for over a minute. No legitimate filename is anywhere near this long.
|
|
100
|
+
const MAX_FILENAME_TOKEN_LENGTH = 512
|
|
101
|
+
|
|
102
|
+
// Commands that scan/search text rather than transmit it: a secret-shaped
|
|
103
|
+
// string used as a search pattern (a security audit grepping for a known
|
|
104
|
+
// placeholder to remove it) isn't a live secret being set or sent anywhere.
|
|
105
|
+
const SEARCH_TOOL_RE = /^\s*(?:sudo\s+)?(?:grep|egrep|fgrep|rg|ag|ack)\b/
|
|
106
|
+
// Widely-published placeholder values from official vendor docs/examples —
|
|
107
|
+
// never real credentials, exact-match only so a genuine key sharing the same
|
|
108
|
+
// prefix still gets caught.
|
|
109
|
+
// Assembled rather than written literally: this line is itself a key-shaped
|
|
110
|
+
// string, and the hazard scanner is supposed to flag key-shaped strings. Writing
|
|
111
|
+
// it whole made the tool's own source a false positive against itself.
|
|
112
|
+
const KNOWN_PLACEHOLDER_SECRETS = new Set(['AKIA' + 'IOSFODNN7EXAMPLE'])
|
|
113
|
+
|
|
114
|
+
// Secret-fetch commands beyond macOS Keychain: Linux keyrings, password-store,
|
|
115
|
+
// gpg, and the common cloud/vault CLIs. Unlike the keychain rule above (which
|
|
116
|
+
// earned its precise capture/redirect scoping from a real script in
|
|
117
|
+
// production), these use one coarse per-clause safety check — omitted:
|
|
118
|
+
// deeper data-flow tracing for each of these until one has a real
|
|
119
|
+
// counter-example that needs it, same as the keychain rule once did.
|
|
120
|
+
const SECRET_FETCH_COMMANDS = [
|
|
121
|
+
['secret-tool-print', /\bsecret-tool\s+lookup\b/, null, '`secret-tool lookup` (Linux keyring) prints the secret to stdout.'],
|
|
122
|
+
[
|
|
123
|
+
'pass-show-print',
|
|
124
|
+
/\bpass\s+(?:show\s+)?[A-Za-z][\w./-]*\b/,
|
|
125
|
+
/\bpass\s+(generate|insert|edit|rm|remove|delete|git|ls|list|find|grep|init|mv|move|cp|copy|help|version|--help|--version)\b/,
|
|
126
|
+
'`pass` (password-store) prints the decrypted secret to stdout.',
|
|
127
|
+
],
|
|
128
|
+
[
|
|
129
|
+
'gpg-decrypt-print',
|
|
130
|
+
/\bgpg\s+(?:--decrypt\b|-[a-zA-Z]*d[a-zA-Z]*\b)/,
|
|
131
|
+
/(?:-o\s|--output\b)/,
|
|
132
|
+
'`gpg --decrypt` with no -o/--output target prints the decrypted plaintext to stdout.',
|
|
133
|
+
],
|
|
134
|
+
['1password-read-print', /\bop\s+(?:read\b|item\s+get\b[^\n]*(?:--fields|--reveal)\b)/, /--help\b|-h\b/, '1Password CLI (`op read` / `op item get --fields`) prints the secret value to stdout.'],
|
|
135
|
+
['aws-secretsmanager-print', /\baws\s+secretsmanager\s+get-secret-value\b/, null, '`aws secretsmanager get-secret-value` prints the SecretString to stdout.'],
|
|
136
|
+
['vault-read-print', /\bvault\s+(?:kv\s+get|read)\b/, null, '`vault kv get` / `vault read` prints secret data to stdout.'],
|
|
137
|
+
['gcloud-secret-print', /\bgcloud\s+secrets\s+versions\s+access\b/, null, '`gcloud secrets versions access` prints the secret payload to stdout.'],
|
|
138
|
+
['azure-keyvault-print', /\baz\s+keyvault\s+secret\s+show\b/, null, '`az keyvault secret show` prints the secret value to stdout.'],
|
|
139
|
+
['kubectl-secret-print', /\bkubectl\s+get\s+secrets?\b[^\n]*-o\s+(?:jsonpath|go-template|json|yaml)\b/, null, '`kubectl get secret -o jsonpath/json/yaml` dumps the secret\'s data field to stdout.'],
|
|
140
|
+
]
|
|
141
|
+
|
|
142
|
+
function grepTargetsSecret(argsStr) {
|
|
143
|
+
if (GREP_SAFE_FLAGS_RE.test(argsStr)) return false
|
|
144
|
+
if (CONNECTION_STRING_RE.test(argsStr)) return true
|
|
145
|
+
const words = argsStr.split(/[^A-Za-z]+/).filter(Boolean)
|
|
146
|
+
return words.some((w) => ENV_SECRET_WORD_RE.test(w))
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
function hasUnsafeSecretFileTarget(command) {
|
|
150
|
+
// Test individual whitespace-separated tokens, not the whole command as one
|
|
151
|
+
// blob — a credential filename is always a single token, so this loses no
|
|
152
|
+
// detection while bounding each SECRET_FILE_RE call to a short string (see
|
|
153
|
+
// MAX_FILENAME_TOKEN_LENGTH) and closing the quadratic-time DoS above.
|
|
154
|
+
for (const rawToken of command.split(/\s+/)) {
|
|
155
|
+
if (rawToken.length === 0 || rawToken.length > MAX_FILENAME_TOKEN_LENGTH) continue
|
|
156
|
+
SECRET_FILE_RE.lastIndex = 0
|
|
157
|
+
const m = SECRET_FILE_RE.exec(rawToken)
|
|
158
|
+
if (!m) continue
|
|
159
|
+
const token = m[1].toLowerCase()
|
|
160
|
+
const base = token.split('/').pop()
|
|
161
|
+
if (SAFE_ENV_SUFFIXES.some((suf) => token.endsWith(suf))) continue
|
|
162
|
+
if (DOC_EXTENSIONS.some((ext) => token.endsWith(ext))) continue
|
|
163
|
+
if (PUBLIC_CERT_BASENAMES.includes(base)) continue
|
|
164
|
+
return true
|
|
165
|
+
}
|
|
166
|
+
return false
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// Per-clause check for the coarser SECRET_FETCH_COMMANDS table: true if this
|
|
170
|
+
// clause's stdout is captured into a variable, redirected, or piped into a
|
|
171
|
+
// known non-printing sink, rather than printed bare.
|
|
172
|
+
function isCapturedOrRedirected(clause) {
|
|
173
|
+
if (STDOUT_REDIRECTED_RE.test(clause)) return true
|
|
174
|
+
if (SAFE_SINK_RE.test(clause)) return true
|
|
175
|
+
// Bounded quantifiers — see the comment above ASSIGN_CAPTURE_RE for why.
|
|
176
|
+
if (/\w{1,64}=\$\([^)]{0,2048}\)|\w{1,64}=`[^`]{0,2048}`/.test(clause)) return true
|
|
177
|
+
return false
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
// True when a KEYCHAIN_READ_RE match sits as the sole statement in a shell
|
|
181
|
+
// function body (`name() {\n security ... -w\n}`) AND that function is later
|
|
182
|
+
// called in a captured way (`$(name ...)`, `` `name ...` ``) — the standard
|
|
183
|
+
// idiom for a helper that returns its value via stdout/$(...) at the call
|
|
184
|
+
// site. Wrapping alone is NOT sufficient: a bare, uncaptured call to the
|
|
185
|
+
// function prints to stdout on invocation exactly like the unwrapped command
|
|
186
|
+
// would, so this requires positive evidence of a captured call, not just the
|
|
187
|
+
// absence of other statements in the body.
|
|
188
|
+
function isSoleStatementInFunctionBody(command, match) {
|
|
189
|
+
const before = command.slice(0, match.index)
|
|
190
|
+
const opener = /(\w{1,64})\s*\(\)\s*\{\s*$/.exec(before)
|
|
191
|
+
if (!opener) return false
|
|
192
|
+
const after = command.slice(match.index + match[0].length)
|
|
193
|
+
if (!/^(?:\s*2?>&?1?\s*\/dev\/null)?\s*\n?\s*\}/.test(after)) return false
|
|
194
|
+
const funcName = opener[1]
|
|
195
|
+
const capturedCallRe = new RegExp(`\\$\\(\\s*${funcName}\\b|\`\\s*${funcName}\\b`)
|
|
196
|
+
return capturedCallRe.test(command.slice(match.index + match[0].length))
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
export function assessLeak(command) {
|
|
200
|
+
// The reviewed command is the escape hatch, and the form is the documented
|
|
201
|
+
// one: a trailing `# omit-allow: <reason>` comment. A bare or quoted token is
|
|
202
|
+
// not a review — `echo omit-allow:; cat .env` used to buy silence for the
|
|
203
|
+
// whole line, which made the marker a bypass rather than a signature.
|
|
204
|
+
if (isAllowSuppressed(command)) return []
|
|
205
|
+
const findings = []
|
|
206
|
+
const push = (rule, reason) => findings.push({ rule, reason })
|
|
207
|
+
|
|
208
|
+
// A live secret typed directly into the command line — e.g. pasted into a
|
|
209
|
+
// curl header — is dangerous regardless of redirects, since the command
|
|
210
|
+
// itself (not just its output) lands in the tool's recorded input. Scoped
|
|
211
|
+
// per clause (not "is the whole command a search-tool invocation") so a
|
|
212
|
+
// harmless `grep ... &&` prefix can't suppress detection of a real secret
|
|
213
|
+
// in a later, unrelated clause of the same compound command. Skipped only
|
|
214
|
+
// for clauses that are themselves search-tool invocations (grep/rg/...): a
|
|
215
|
+
// secret-shaped string there is almost always a search pattern (e.g.
|
|
216
|
+
// auditing for a known placeholder), not a live value being set or sent.
|
|
217
|
+
for (const clause of splitClauses(command)) {
|
|
218
|
+
if (SEARCH_TOOL_RE.test(clause)) continue
|
|
219
|
+
for (const [rule, re] of SECRET_RULES) {
|
|
220
|
+
const m = re.exec(clause)
|
|
221
|
+
if (m && !KNOWN_PLACEHOLDER_SECRETS.has(m[0])) {
|
|
222
|
+
push(
|
|
223
|
+
`literal-secret-${rule}`,
|
|
224
|
+
`this command line contains what looks like a live ${rule.replace(/-/g, ' ')} — the value would be recorded in the transcript verbatim. Reference it via a variable or secrets manager instead of the literal, and rotate the key if it's real.`,
|
|
225
|
+
)
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
if (KEYCHAIN_DUMP_RE.test(command)) {
|
|
231
|
+
push('keychain-dump', '`security dump-keychain` prints every keychain item, secrets included. Look up one specific item instead.')
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
const keychainMatch = KEYCHAIN_READ_RE.exec(command)
|
|
235
|
+
if (keychainMatch) {
|
|
236
|
+
const assign = ASSIGN_CAPTURE_RE.exec(command)
|
|
237
|
+
const redirect = KEYCHAIN_REDIRECTED_RE.exec(command)
|
|
238
|
+
const tailFromMatch = command.slice(keychainMatch.index + keychainMatch[0].length)
|
|
239
|
+
let safe = TEST_CAPTURE_RE.test(command) || isSoleStatementInFunctionBody(command, keychainMatch) || SAFE_SINK_RE.test(tailFromMatch)
|
|
240
|
+
if (redirect) {
|
|
241
|
+
// Redirected to a file (not /dev/null) is only safe if nothing later in
|
|
242
|
+
// the command reads that same file back out — "write to a scratch file,
|
|
243
|
+
// then cat it to double-check" is a real leak, not a suppression. Plain
|
|
244
|
+
// substring search, not a rebuilt RegExp: the target is an unbounded,
|
|
245
|
+
// command-controlled string, and interpolating it into a pattern risks
|
|
246
|
+
// hitting V8's regex-size ceiling on a long path (a real, reproducible
|
|
247
|
+
// crash — not hypothetical).
|
|
248
|
+
const target = redirect[1]
|
|
249
|
+
const tail = command.slice(redirect.index + redirect[0].length)
|
|
250
|
+
const readMatch = READ_CMDS_RE.exec(tail)
|
|
251
|
+
const rereadElsewhere = readMatch !== null && tail.slice(readMatch.index).includes(target)
|
|
252
|
+
safe = safe || !rereadElsewhere
|
|
253
|
+
}
|
|
254
|
+
if (assign) {
|
|
255
|
+
const varName = assign[1]
|
|
256
|
+
const tail = command.slice(assign.index + assign[0].length)
|
|
257
|
+
const echoedLater = new RegExp(`\\b(echo|printf|print|cat)\\b[^\\n]*\\$\\{?${varName}\\}?\\b`).test(tail)
|
|
258
|
+
safe = !echoedLater || SAFE_SINK_RE.test(tail)
|
|
259
|
+
}
|
|
260
|
+
if (!safe) {
|
|
261
|
+
push(
|
|
262
|
+
'keychain-secret-print',
|
|
263
|
+
'`security find-generic-password -w` prints the raw secret to stdout. Redirect it (`-w &>/dev/null`) to just ' +
|
|
264
|
+
'check it exists, or capture it into a variable (`val=$(security find-generic-password -s NAME -w 2>/dev/null)`) without echoing it back.',
|
|
265
|
+
)
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
// Scoped per clause: an unrelated redirect on one clause of a compound
|
|
270
|
+
// command must not suppress a real, unredirected printenv leak on another.
|
|
271
|
+
let envDumpFound = false
|
|
272
|
+
let printenvVarFound = false
|
|
273
|
+
for (const clause of splitClauses(command)) {
|
|
274
|
+
if (!envDumpFound && ENV_BARE_RE.test(clause)) {
|
|
275
|
+
push('env-dump', 'bare `env`/`printenv` can print every secret-bearing variable in the shell. Check one variable with `[[ -n "$VAR" ]]` instead.')
|
|
276
|
+
envDumpFound = true
|
|
277
|
+
} else if (!printenvVarFound && PRINTENV_VAR_RE.test(clause) && !STDOUT_REDIRECTED_RE.test(clause)) {
|
|
278
|
+
push('printenv-var-print', '`printenv <name>` prints that variable\'s raw value to stdout. Check with `[[ -n "$VAR" ]]` instead.')
|
|
279
|
+
printenvVarFound = true
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
const envGrep = ENV_GREP_RE.exec(command)
|
|
283
|
+
if (envGrep && grepTargetsSecret(envGrep[1])) {
|
|
284
|
+
push('env-grep-secret', '`env | grep` targeting a KEY/TOKEN/SECRET/PASSWORD/CREDENTIAL/connection-string variable prints its value, not just its name. Use `env | grep -q ...` or `[[ -n "$VAR" ]]` instead.')
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
for (const clause of splitClauses(command)) {
|
|
288
|
+
if (!READ_CMDS_RE.test(clause) || STDOUT_REDIRECTED_RE.test(clause)) continue
|
|
289
|
+
// grep-family's own value-suppressing flags (-c/-q/-l/-L) are exempt here
|
|
290
|
+
// too — this is the tool's own recommended remediation for this rule.
|
|
291
|
+
if (GREP_FAMILY_RE.test(clause) && GREP_SAFE_FLAGS_RE.test(clause)) continue
|
|
292
|
+
if (hasUnsafeSecretFileTarget(clause)) {
|
|
293
|
+
push(
|
|
294
|
+
'credential-file-read',
|
|
295
|
+
'this reads a file that looks like a credential store (.env, credentials, *.pem, id_rsa, secrets.*) with a command ' +
|
|
296
|
+
'that prints its contents. Use `test -f <file>` to check it exists, or `grep -c "" <file>` to count lines, without printing contents.',
|
|
297
|
+
)
|
|
298
|
+
break // one finding per command for this rule, regardless of how many clauses match
|
|
299
|
+
}
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
for (const [rule, matchRe, excludeRe, reason] of SECRET_FETCH_COMMANDS) {
|
|
303
|
+
for (const clause of splitClauses(command)) {
|
|
304
|
+
if (!matchRe.test(clause)) continue
|
|
305
|
+
if (excludeRe && excludeRe.test(clause)) continue
|
|
306
|
+
if (isCapturedOrRedirected(clause)) continue
|
|
307
|
+
push(rule, `${reason} Capture it into a variable, or redirect stdout to a file, instead of letting it print bare.`)
|
|
308
|
+
break
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
return findings
|
|
313
|
+
}
|
package/lib/lint.mjs
CHANGED
|
@@ -3,6 +3,9 @@
|
|
|
3
3
|
import { existsSync, readFileSync } from 'node:fs'
|
|
4
4
|
import { extname, join } from 'node:path'
|
|
5
5
|
import { spawnSync } from 'node:child_process'
|
|
6
|
+
import { execDisabled } from './exec.mjs'
|
|
7
|
+
|
|
8
|
+
export { execDisabled }
|
|
6
9
|
|
|
7
10
|
const JS = new Set(['.js', '.jsx', '.ts', '.tsx', '.mjs', '.cjs', '.vue', '.svelte'])
|
|
8
11
|
const PY = new Set(['.py'])
|
|
@@ -24,19 +27,96 @@ export function detectLinters(cwd) {
|
|
|
24
27
|
return linters
|
|
25
28
|
}
|
|
26
29
|
|
|
30
|
+
// Characters cmd.exe re-expands inside what looks like a quoted argument, plus
|
|
31
|
+
// the quote and newlines. An argv carrying one never reaches a Windows shim.
|
|
32
|
+
const WIN_UNSAFE = /[&|<>^()%"!\r\n]/
|
|
33
|
+
|
|
34
|
+
// Windows ships `npx` as npx.cmd, which CreateProcess refuses to start: it needs
|
|
35
|
+
// cmd.exe. Resolve the shim here so that decision is visible instead of implied.
|
|
36
|
+
function resolveShim(cmd, env) {
|
|
37
|
+
const exts = (env.PATHEXT || '.COM;.EXE;.BAT;.CMD').split(';').filter(Boolean)
|
|
38
|
+
for (const dir of (env.PATH || '').split(';').filter(Boolean)) {
|
|
39
|
+
for (const ext of ['', ...exts]) {
|
|
40
|
+
const path = join(dir, cmd + ext)
|
|
41
|
+
if (existsSync(path)) return path
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
return null
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
// What we hand the OS, as a plan a test can inspect. `shell` is false on every
|
|
48
|
+
// platform, always: this argv's tail is filenames from `git diff`, and `shell:
|
|
49
|
+
// true` concatenates them back into a command line for the OS to re-parse — so a
|
|
50
|
+
// PR file named `a&b.js` becomes two commands, and the diff author picks the
|
|
51
|
+
// second one. A .cmd shim is the one case that needs a shell at all, so it gets
|
|
52
|
+
// an explicitly built, verbatim line, and the run is refused outright if any
|
|
53
|
+
// argument carries a character that line could not survive as data.
|
|
54
|
+
// Exported for the test that pins this; not part of the module's API.
|
|
55
|
+
export function spawnPlan(cmd, args, platform = process.platform, env = process.env) {
|
|
56
|
+
if (platform !== 'win32') return { cmd, args, shell: false }
|
|
57
|
+
const shim = resolveShim(cmd, env)
|
|
58
|
+
if (shim && !/\.(cmd|bat)$/i.test(shim)) return { cmd: shim, args, shell: false }
|
|
59
|
+
// Trailing `\` would escape the closing quote of a quoted token.
|
|
60
|
+
if (args.some((a) => WIN_UNSAFE.test(a) || /\\$/.test(a))) return null
|
|
61
|
+
const line = [shim ?? cmd, ...args].map((a) => (/\s/.test(a) ? `"${a}"` : a)).join(' ')
|
|
62
|
+
return { cmd: env.ComSpec || 'cmd.exe', args: ['/d', '/s', '/c', `"${line}"`], shell: false, verbatim: true }
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function runLinter(cmd, args, cwd) {
|
|
66
|
+
const plan = spawnPlan(cmd, args)
|
|
67
|
+
if (!plan) return { refused: true }
|
|
68
|
+
return spawnSync(plan.cmd, plan.args, {
|
|
69
|
+
cwd,
|
|
70
|
+
encoding: 'utf8',
|
|
71
|
+
timeout: 60_000,
|
|
72
|
+
shell: plan.shell,
|
|
73
|
+
windowsVerbatimArguments: plan.verbatim === true,
|
|
74
|
+
maxBuffer: 16 << 20,
|
|
75
|
+
})
|
|
76
|
+
}
|
|
77
|
+
|
|
27
78
|
// Runs every detected linter over the files it applies to.
|
|
28
|
-
//
|
|
79
|
+
//
|
|
80
|
+
// Four outcomes, kept apart because the caller is the one making a claim about
|
|
81
|
+
// the repo: `ok` and `fail` (it ran), `unavailable` (configured, but there was
|
|
82
|
+
// nothing to run it with) and `not-run` (execution is off). Collapsing the
|
|
83
|
+
// middle two into "no results" let `rm -rf node_modules` turn an unrun linter
|
|
84
|
+
// into `no linter configured` — a false claim about the repo, printed by the
|
|
85
|
+
// agent's own tool in the same breath as a green verdict.
|
|
86
|
+
//
|
|
87
|
+
// `ok` stays true for both not-ran states: it answers "is anything wrong", which
|
|
88
|
+
// is what every existing caller gates on. Render on `status` before claiming a
|
|
89
|
+
// pass — `ok` alone cannot tell a clean run from a linter that never started.
|
|
90
|
+
// omitted: a `no-match` entry for a configured linter with no file of its
|
|
91
|
+
// extension in the diff; that reads as an empty result today, same as no config.
|
|
29
92
|
export function lintFiles(cwd, files) {
|
|
30
93
|
const results = []
|
|
31
94
|
for (const l of detectLinters(cwd)) {
|
|
32
95
|
const mine = files.filter((f) => l.exts.has(extname(f).toLowerCase()))
|
|
33
96
|
if (mine.length === 0) continue
|
|
97
|
+
if (execDisabled()) {
|
|
98
|
+
results.push({ linter: l.name, status: 'not-run', ok: true, output: `not run: OMIT_NO_EXEC is set` })
|
|
99
|
+
continue
|
|
100
|
+
}
|
|
34
101
|
const [cmd, ...args] = l.argv(mine)
|
|
35
|
-
const r =
|
|
36
|
-
if (r.
|
|
102
|
+
const r = runLinter(cmd, args, cwd)
|
|
103
|
+
if (r.refused) {
|
|
104
|
+
// Reached only on Windows, for a .cmd shim handed a name cmd.exe would re-parse.
|
|
105
|
+
results.push({ linter: l.name, status: 'unavailable', ok: true, output: `not run: a filename carries shell characters` })
|
|
106
|
+
continue
|
|
107
|
+
}
|
|
37
108
|
const out = `${r.stdout ?? ''}${r.stderr ?? ''}`
|
|
38
|
-
if (r.
|
|
39
|
-
|
|
109
|
+
if (r.error || r.status === null) {
|
|
110
|
+
results.push({ linter: l.name, status: 'unavailable', ok: true, output: out.trim() || `could not run: ${r.error?.code ?? 'no exit status'}` })
|
|
111
|
+
continue
|
|
112
|
+
}
|
|
113
|
+
// The binary is missing, not the lint clean: npx exits nonzero and says so
|
|
114
|
+
// on stderr. Reporting that as a lint failure would be a false objection.
|
|
115
|
+
if (r.status !== 0 && /not found|command not found|npm error|npm ERR/i.test(out)) {
|
|
116
|
+
results.push({ linter: l.name, status: 'unavailable', ok: true, output: out.trim().split('\n').slice(0, 30).join('\n') })
|
|
117
|
+
continue
|
|
118
|
+
}
|
|
119
|
+
results.push({ linter: l.name, status: r.status === 0 ? 'ok' : 'fail', ok: r.status === 0, output: out.trim().split('\n').slice(0, 30).join('\n') })
|
|
40
120
|
}
|
|
41
121
|
return results
|
|
42
122
|
}
|