easyvibegate 0.4.4 → 0.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,260 @@
1
+ /**
2
+ * A small per-language lexer for source code, chosen by file extension.
3
+ *
4
+ * Why this exists: the old masker scanned characters with no notion of which
5
+ * language it was reading. `//` inside a URL string became a line comment and
6
+ * erased the real code after it; a JS `#private` field or a Python `4 // 2`
7
+ * was taken for a comment; a quoted string was not atomic, so anything inside
8
+ * it could open a "comment" that swallowed the rest of the line. Every such
9
+ * corner was a silently missed finding — a clean PASS on dangerous code.
10
+ *
11
+ * Lexing once, per language, makes that family of bugs impossible by
12
+ * construction: a string is one token whatever it contains, a comment is one
13
+ * token, and the rules for what starts either come from the language, not
14
+ * from a guess. Same approach as sql-lex.ts, which SQL already relies on.
15
+ *
16
+ * Coverage is deliberately small: comments, strings, template literals and
17
+ * regex literals. Everything else is `word` or `punct`. The grammars are
18
+ * approximations for languages we only *mask* (Go/Java/Rust share the C-family
19
+ * rules; shell/YAML/TOML share `#` rules), documented in `langForFile`.
20
+ */
21
+ import { lexSql } from './sql-lex.js';
22
+ const NONE = { slash: false, hash: false, html: false, quotes: true, triple: false, template: false, regex: false, privateName: false };
23
+ const JS = { ...NONE, slash: true, template: true, regex: true, privateName: true };
24
+ const PROFILES = {
25
+ js: JS,
26
+ html: { ...JS, html: true },
27
+ python: { ...NONE, hash: true, triple: true },
28
+ hash: { ...NONE, hash: true },
29
+ markup: { ...NONE, html: true, quotes: false },
30
+ mixed: { ...NONE, slash: true, hash: true },
31
+ none: NONE,
32
+ };
33
+ const EXT_LANG = {
34
+ '.js': 'js', '.jsx': 'js', '.ts': 'js', '.tsx': 'js', '.mjs': 'js', '.cjs': 'js', '.mts': 'js', '.cts': 'js',
35
+ // C-family languages share the comment/string shapes we care about.
36
+ '.go': 'js', '.java': 'js', '.kt': 'js', '.kts': 'js', '.rs': 'js', '.cs': 'js', '.swift': 'js', '.dart': 'js',
37
+ '.scala': 'js', '.groovy': 'js', '.gradle': 'js', '.php': 'js', '.c': 'js', '.h': 'js', '.cc': 'js', '.cpp': 'js',
38
+ '.hpp': 'js', '.css': 'js', '.scss': 'js', '.less': 'js', '.jsonc': 'js', '.json5': 'js',
39
+ '.vue': 'html', '.svelte': 'html', '.astro': 'html', '.html': 'html', '.htm': 'html',
40
+ '.py': 'python', '.pyi': 'python', '.pyw': 'python',
41
+ '.rb': 'hash', '.sh': 'hash', '.bash': 'hash', '.zsh': 'hash', '.fish': 'hash', '.yml': 'hash', '.yaml': 'hash',
42
+ '.toml': 'hash', '.ini': 'hash', '.cfg': 'hash', '.conf': 'hash', '.env': 'hash', '.properties': 'hash',
43
+ '.pl': 'hash', '.pm': 'hash', '.r': 'hash', '.ex': 'hash', '.exs': 'hash', '.nix': 'hash', '.tfvars': 'hash',
44
+ '.sql': 'sql', '.psql': 'sql', '.pgsql': 'sql',
45
+ '.xml': 'markup', '.plist': 'markup', '.svg': 'markup',
46
+ // Prose keeps the legacy rules so `// easyvibegate-ignore` in a README still counts as a comment.
47
+ '.md': 'mixed', '.mdx': 'mixed', '.txt': 'mixed',
48
+ '.json': 'none', '.ipynb': 'none', '.csv': 'none', '.lock': 'none',
49
+ '.pem': 'none', '.key': 'none', '.crt': 'none', '.cert': 'none', '.pkcs8': 'none',
50
+ };
51
+ const HASH_NAMES = new Set(['Dockerfile', 'Makefile', 'Gemfile', 'Procfile', '.gitignore', '.dockerignore', '.npmrc', '.yarnrc']);
52
+ /** Pick the lexing rules for a path. Unknown extensions get the legacy `mixed` rules. */
53
+ export function langForFile(rel) {
54
+ const base = rel.slice(Math.max(rel.lastIndexOf('/'), rel.lastIndexOf('\\')) + 1);
55
+ if (base.startsWith('.env') || base.startsWith('docker-compose') || HASH_NAMES.has(base))
56
+ return 'hash';
57
+ const dot = base.lastIndexOf('.');
58
+ const ext = dot <= 0 ? '' : base.slice(dot).toLowerCase();
59
+ return EXT_LANG[ext] ?? 'mixed';
60
+ }
61
+ const isIdentStart = (c) => /[A-Za-z_$€-￿]/.test(c);
62
+ const isIdentPart = (c) => /[A-Za-z0-9_$€-￿]/.test(c);
63
+ const isSpace = (c) => c === ' ' || c === '\t' || c === '\n' || c === '\r' || c === '\f' || c === '\v';
64
+ /** After these words a `/` starts a regex literal, not a division. */
65
+ const REGEX_AFTER_WORD = new Set([
66
+ 'return', 'typeof', 'instanceof', 'in', 'of', 'new', 'delete', 'void', 'throw', 'case', 'do', 'else', 'yield', 'await',
67
+ ]);
68
+ /** Tokenize `src` with the rules of `lang`. Offsets are absolute into `src`. */
69
+ export function lexCode(src, lang = 'mixed') {
70
+ if (lang === 'sql') {
71
+ // SQL already has a real lexer; only the token vocabulary differs.
72
+ return lexSql(src).map((t) => ({
73
+ type: t.type === 'comment' ? 'comment' : t.type === 'word' || t.type === 'punct' ? t.type : 'string',
74
+ value: src.slice(t.start, t.end),
75
+ start: t.start,
76
+ end: t.end,
77
+ }));
78
+ }
79
+ const out = [];
80
+ lexInto(src, 0, PROFILES[lang], out, false);
81
+ return out;
82
+ }
83
+ /**
84
+ * Lex from `from` to the end, or — inside a template `${…}` — up to the `}`
85
+ * that closes it, whose index is returned. Nested braces, strings, comments
86
+ * and templates inside the expression are all handled by the same loop.
87
+ */
88
+ function lexInto(src, from, p, out, inTemplateExpr) {
89
+ const n = src.length;
90
+ const push = (type, start, end) => out.push({ type, value: src.slice(start, end), start, end });
91
+ let depth = 0;
92
+ let i = from;
93
+ while (i < n) {
94
+ const ch = src[i];
95
+ const next = src[i + 1];
96
+ if (isSpace(ch)) {
97
+ i++;
98
+ continue;
99
+ }
100
+ if (inTemplateExpr) {
101
+ if (ch === '{')
102
+ depth++;
103
+ else if (ch === '}') {
104
+ if (depth === 0)
105
+ return i;
106
+ depth--;
107
+ }
108
+ }
109
+ if (p.slash && ch === '/' && next === '/') {
110
+ let j = i;
111
+ while (j < n && src[j] !== '\n')
112
+ j++;
113
+ push('comment', i, j);
114
+ i = j;
115
+ continue;
116
+ }
117
+ if (p.slash && ch === '/' && next === '*') {
118
+ const c = src.indexOf('*/', i + 2);
119
+ const end = c === -1 ? n : c + 2;
120
+ push('comment', i, end);
121
+ i = end;
122
+ continue;
123
+ }
124
+ if (p.hash && ch === '#' && src[i - 1] !== '$') {
125
+ let j = i;
126
+ while (j < n && src[j] !== '\n')
127
+ j++;
128
+ push('comment', i, j);
129
+ i = j;
130
+ continue;
131
+ }
132
+ if (p.html && ch === '<' && src.startsWith('<!--', i)) {
133
+ const c = src.indexOf('-->', i + 4);
134
+ const end = c === -1 ? n : c + 3;
135
+ push('comment', i, end);
136
+ i = end;
137
+ continue;
138
+ }
139
+ if (p.quotes && (ch === '"' || ch === "'")) {
140
+ if (p.triple && next === ch && src[i + 2] === ch) {
141
+ // '''…''' spans lines; a backslash still escapes the next character.
142
+ const q = ch + ch + ch;
143
+ let j = i + 3;
144
+ while (j < n && !src.startsWith(q, j))
145
+ j += src[j] === '\\' ? 2 : 1;
146
+ const end = Math.min(j + 3, n);
147
+ push('string', i, end);
148
+ i = end;
149
+ continue;
150
+ }
151
+ // One-line string. An unterminated one ends at the newline so damage from
152
+ // a stray quote (JSX text, prose) never spreads past its own line.
153
+ let j = i + 1;
154
+ while (j < n && src[j] !== ch && src[j] !== '\n')
155
+ j += src[j] === '\\' ? 2 : 1;
156
+ const end = Math.min(src[j] === ch ? j + 1 : j, n);
157
+ push('string', i, end);
158
+ i = end;
159
+ continue;
160
+ }
161
+ if (p.template && ch === '`') {
162
+ let chunk = i;
163
+ let j = i + 1;
164
+ while (j < n) {
165
+ const c = src[j];
166
+ if (c === '\\') {
167
+ j += 2;
168
+ continue;
169
+ }
170
+ if (c === '`') {
171
+ j++;
172
+ break;
173
+ }
174
+ if (c === '$' && src[j + 1] === '{') {
175
+ push('template', chunk, j + 2);
176
+ // The expression is code again: lex it in place, resume after its `}`.
177
+ const close = lexInto(src, j + 2, p, out, true);
178
+ chunk = close;
179
+ j = close + 1;
180
+ continue;
181
+ }
182
+ j++;
183
+ }
184
+ const end = Math.min(j, n);
185
+ if (end > chunk)
186
+ push('template', chunk, end);
187
+ i = end;
188
+ continue;
189
+ }
190
+ if (p.regex && ch === '/' && regexAllowed(out)) {
191
+ const end = scanRegex(src, i);
192
+ if (end !== -1) {
193
+ push('regex', i, end);
194
+ i = end;
195
+ continue;
196
+ }
197
+ }
198
+ if (isIdentStart(ch) || (p.privateName && ch === '#' && next !== undefined && isIdentStart(next))) {
199
+ let j = i + 1;
200
+ while (j < n && isIdentPart(src[j]))
201
+ j++;
202
+ push('word', i, j);
203
+ i = j;
204
+ continue;
205
+ }
206
+ if (ch >= '0' && ch <= '9') {
207
+ let j = i + 1;
208
+ while (j < n && /[A-Za-z0-9_.]/.test(src[j]))
209
+ j++;
210
+ push('word', i, j);
211
+ i = j;
212
+ continue;
213
+ }
214
+ push('punct', i, i + 1);
215
+ i++;
216
+ }
217
+ return n;
218
+ }
219
+ /** A `/` after a value (identifier, number, `)`, `]`, `}`) divides; elsewhere it opens a regex. */
220
+ function regexAllowed(out) {
221
+ let k = out.length - 1;
222
+ while (k >= 0 && out[k]?.type === 'comment')
223
+ k--;
224
+ const prev = out[k];
225
+ if (!prev)
226
+ return true;
227
+ if (prev.type === 'punct')
228
+ return !')]}'.includes(prev.value);
229
+ if (prev.type === 'word')
230
+ return REGEX_AFTER_WORD.has(prev.value);
231
+ return false;
232
+ }
233
+ /** End offset of a regex literal starting at `i`, or -1 if the line ends first (so it was a division). */
234
+ function scanRegex(src, i) {
235
+ let j = i + 1;
236
+ let inClass = false;
237
+ while (j < src.length) {
238
+ const c = src[j];
239
+ if (c === '\n')
240
+ return -1;
241
+ if (c === '\\') {
242
+ j += 2;
243
+ continue;
244
+ }
245
+ if (inClass) {
246
+ if (c === ']')
247
+ inClass = false;
248
+ }
249
+ else if (c === '[')
250
+ inClass = true;
251
+ else if (c === '/') {
252
+ j++;
253
+ while (j < src.length && /[a-z]/i.test(src[j]))
254
+ j++;
255
+ return j;
256
+ }
257
+ j++;
258
+ }
259
+ return -1;
260
+ }
@@ -0,0 +1,33 @@
1
+ import { execFileSync } from 'node:child_process';
2
+ function git(root, args) {
3
+ try {
4
+ return execFileSync('git', ['-C', root, ...args], { encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'], timeout: 5000 });
5
+ }
6
+ catch {
7
+ return null;
8
+ }
9
+ }
10
+ /**
11
+ * Build a per-root resolver. Tracked files are listed once; ignore checks run
12
+ * lazily and are cached, so a project with many findings still costs a few
13
+ * git calls, not one per finding.
14
+ */
15
+ export function createExposure(root) {
16
+ const isRepo = git(root, ['rev-parse', '--git-dir']) !== null;
17
+ if (!isRepo)
18
+ return () => 'no-git';
19
+ const tracked = new Set((git(root, ['ls-files', '-z']) ?? '').split('\0').filter(Boolean));
20
+ const cache = new Map();
21
+ return (rel) => {
22
+ const hit = cache.get(rel);
23
+ if (hit)
24
+ return hit;
25
+ let ex;
26
+ if (tracked.has(rel))
27
+ ex = 'committed';
28
+ else
29
+ ex = git(root, ['check-ignore', '-q', '--', rel]) !== null ? 'ignored' : 'untracked';
30
+ cache.set(rel, ex);
31
+ return ex;
32
+ };
33
+ }
@@ -2,8 +2,21 @@
2
2
  * Blank out comments (and, optionally, quoted strings) while preserving length
3
3
  * and newlines, so match offsets and line numbers stay correct. Used so that a
4
4
  * comment like `// never use eval()` is not reported as a finding.
5
+ *
6
+ * What counts as a comment or a string depends on the language, so pass the
7
+ * file name (or a `lang`) — see code-lex.ts. Without either, the legacy
8
+ * `mixed` rules apply: both `//` and `#` comments, strings atomic.
5
9
  */
10
+ import { lexCode, langForFile } from './code-lex.js';
6
11
  export function maskCode(src, opts = {}) {
12
+ const lang = opts.lang ?? (opts.file ? langForFile(opts.file) : 'mixed');
13
+ return maskTokens(src, lexCode(src, lang), opts);
14
+ }
15
+ /**
16
+ * Same as maskCode but over tokens lexed once — a checker that needs both the
17
+ * "no comments" and the "no strings" view of a file pays for lexing only once.
18
+ */
19
+ export function maskTokens(src, tokens, opts = {}) {
7
20
  const out = src.split('');
8
21
  const n = src.length;
9
22
  const blank = (from, to) => {
@@ -11,54 +24,29 @@ export function maskCode(src, opts = {}) {
11
24
  if (out[k] !== '\n')
12
25
  out[k] = ' ';
13
26
  };
14
- let i = 0;
15
- while (i < n) {
16
- const ch = src[i];
17
- const next = src[i + 1];
18
- if (ch === '/' && next === '/') {
19
- let j = i;
20
- while (j < n && src[j] !== '\n')
21
- j++;
22
- blank(i, j);
23
- i = j;
24
- continue;
25
- }
26
- if (ch === '#' && src[i - 1] !== '$') {
27
- let j = i;
28
- while (j < n && src[j] !== '\n')
29
- j++;
30
- blank(i, j);
31
- i = j;
32
- continue;
33
- }
34
- if (ch === '/' && next === '*') {
35
- const c = src.indexOf('*/', i + 2);
36
- const end = c === -1 ? n : c + 2;
37
- blank(i, end);
38
- i = end;
39
- continue;
40
- }
41
- if (opts.strings && (ch === '"' || ch === "'")) {
42
- let j = i + 1;
43
- while (j < n && src[j] !== ch) {
44
- if (src[j] === '\\')
45
- j++;
46
- if (src[j] === '\n')
47
- break;
48
- j++;
49
- }
50
- const end = Math.min(j + 1, n);
51
- blank(i, end);
52
- i = end;
53
- continue;
54
- }
55
- i++;
27
+ for (const t of tokens) {
28
+ if (t.type === 'comment')
29
+ blank(t.start, t.end);
30
+ else if (opts.strings && (t.type === 'string' || t.type === 'template' || t.type === 'regex'))
31
+ blank(t.start, t.end);
56
32
  }
57
33
  return out.join('');
58
34
  }
59
- /** Minified/bundled output is not source a human wrote — reviewing it is noise. */
35
+ /**
36
+ * Paths that hold generated output, not source a human wrote: `.min.js`,
37
+ * `dist/`, `build/`, `vendor/`, `bundle/`. A checker that skips these must
38
+ * say so (CheckerResult.partial) — skipped is not clean.
39
+ */
40
+ export function looksBundledPath(rel) {
41
+ return /\.min\.(js|css|mjs|cjs)$/i.test(rel) || /(^|\/)(dist|build|vendor|bundle)\//.test(rel);
42
+ }
43
+ /**
44
+ * Minified/bundled output is not source a human wrote — reviewing it is noise.
45
+ * The long-line heuristic is only a guess: one long data line in a real source
46
+ * file also trips it, so callers must never use it to skip a file silently.
47
+ */
60
48
  export function looksMinified(rel, content) {
61
- if (/\.min\.(js|css|mjs|cjs)$/i.test(rel) || /(^|\/)(dist|build|vendor|bundle)\//.test(rel))
49
+ if (looksBundledPath(rel))
62
50
  return true;
63
51
  const lines = content.split('\n');
64
52
  const longest = lines.reduce((m, l) => Math.max(m, l.length), 0);
@@ -0,0 +1,172 @@
1
+ /**
2
+ * A small SQL lexer.
3
+ *
4
+ * Why this exists: the RLS checker used to look for statements with regexes run
5
+ * over a "masked" copy of the file, where comments and strings had been blanked
6
+ * out by a second, separate scanner. Every syntax corner that the masker got
7
+ * wrong became a silently missed (or invented) statement — `E'it\'s'`, nested
8
+ * block comments, `$tag$` bodies, a column alias in double quotes that happens
9
+ * to contain SQL keywords. Approximating a real grammar with regexes has no end.
10
+ *
11
+ * Lexing once, here, makes that whole family of bugs impossible by construction:
12
+ * a string is a token, a comment is a token, a quoted identifier is a token, and
13
+ * none of them can ever be read as executable keywords.
14
+ */
15
+ const isIdentStart = (c) => /[A-Za-z_€-￿]/.test(c);
16
+ const isIdentPart = (c) => /[A-Za-z0-9_$€-￿]/.test(c);
17
+ /** Read a `$tag$` opener at `i`, or null if this `$` is not a dollar-quote. */
18
+ function dollarTagAt(sql, i) {
19
+ if (sql[i] !== '$')
20
+ return null;
21
+ let j = i + 1;
22
+ while (j < sql.length && sql[j] !== '$') {
23
+ const c = sql[j];
24
+ // A tag is an identifier; `$1` (a bind parameter) is not a dollar-quote.
25
+ if (!(j === i + 1 ? isIdentStart(c) : isIdentPart(c)))
26
+ return null;
27
+ j++;
28
+ }
29
+ return j < sql.length ? sql.slice(i, j + 1) : null;
30
+ }
31
+ /** Tokenize `sql`. Offsets are absolute; add `offset` when lexing a fragment. */
32
+ export function lexSql(sql, offset = 0) {
33
+ const out = [];
34
+ const n = sql.length;
35
+ let i = 0;
36
+ while (i < n) {
37
+ const ch = sql[i];
38
+ if (/\s/.test(ch)) {
39
+ i++;
40
+ continue;
41
+ }
42
+ // -- line comment
43
+ if (ch === '-' && sql[i + 1] === '-') {
44
+ let j = i;
45
+ while (j < n && sql[j] !== '\n')
46
+ j++;
47
+ out.push({ type: 'comment', value: sql.slice(i, j), start: offset + i, end: offset + j });
48
+ i = j;
49
+ continue;
50
+ }
51
+ // /* block comment */ — PostgreSQL nests these.
52
+ if (ch === '/' && sql[i + 1] === '*') {
53
+ let depth = 0;
54
+ let j = i;
55
+ while (j < n) {
56
+ if (sql[j] === '/' && sql[j + 1] === '*') {
57
+ depth++;
58
+ j += 2;
59
+ continue;
60
+ }
61
+ if (sql[j] === '*' && sql[j + 1] === '/') {
62
+ depth--;
63
+ j += 2;
64
+ if (depth === 0)
65
+ break;
66
+ continue;
67
+ }
68
+ j++;
69
+ }
70
+ const end = depth === 0 ? j : n;
71
+ out.push({ type: 'comment', value: sql.slice(i, end), start: offset + i, end: offset + end });
72
+ i = end;
73
+ continue;
74
+ }
75
+ // '...' string. Doubling ('') always escapes; a backslash escapes only in an
76
+ // E'' string, where PostgreSQL enables C-style escapes.
77
+ if (ch === "'") {
78
+ const prev = out[out.length - 1];
79
+ const eString = !!prev && prev.type === 'word' && /^e$/i.test(prev.value) && prev.end === offset + i;
80
+ let j = i + 1;
81
+ while (j < n) {
82
+ if (eString && sql[j] === '\\') {
83
+ j += 2;
84
+ continue;
85
+ }
86
+ if (sql[j] === "'") {
87
+ if (sql[j + 1] === "'") {
88
+ j += 2;
89
+ continue;
90
+ }
91
+ break;
92
+ }
93
+ j++;
94
+ }
95
+ const end = Math.min(j + 1, n);
96
+ out.push({
97
+ type: 'string',
98
+ value: sql.slice(i + 1, Math.max(i + 1, j)),
99
+ start: offset + i,
100
+ end: offset + end,
101
+ bodyStart: offset + i + 1,
102
+ });
103
+ i = end;
104
+ continue;
105
+ }
106
+ // "..." quoted identifier. Its contents are a NAME, never statements.
107
+ if (ch === '"') {
108
+ let j = i + 1;
109
+ while (j < n) {
110
+ if (sql[j] === '"') {
111
+ if (sql[j + 1] === '"') {
112
+ j += 2;
113
+ continue;
114
+ }
115
+ break;
116
+ }
117
+ j++;
118
+ }
119
+ const end = Math.min(j + 1, n);
120
+ out.push({
121
+ type: 'quotedIdent',
122
+ value: sql.slice(i + 1, Math.max(i + 1, j)).replace(/""/g, '"'),
123
+ start: offset + i,
124
+ end: offset + end,
125
+ });
126
+ i = end;
127
+ continue;
128
+ }
129
+ // $tag$ ... $tag$
130
+ const tag = dollarTagAt(sql, i);
131
+ if (tag) {
132
+ const bodyStart = i + tag.length;
133
+ const close = sql.indexOf(tag, bodyStart);
134
+ const bodyEnd = close === -1 ? n : close;
135
+ const end = close === -1 ? n : close + tag.length;
136
+ out.push({
137
+ type: 'dollarString',
138
+ value: sql.slice(bodyStart, bodyEnd),
139
+ start: offset + i,
140
+ end: offset + end,
141
+ bodyStart: offset + bodyStart,
142
+ });
143
+ i = end;
144
+ continue;
145
+ }
146
+ // word / unquoted identifier
147
+ if (isIdentStart(ch)) {
148
+ let j = i + 1;
149
+ while (j < n && isIdentPart(sql[j]))
150
+ j++;
151
+ out.push({ type: 'word', value: sql.slice(i, j), start: offset + i, end: offset + j });
152
+ i = j;
153
+ continue;
154
+ }
155
+ // number — lexed as a word so it can never be mistaken for a keyword
156
+ if (/[0-9]/.test(ch)) {
157
+ let j = i + 1;
158
+ while (j < n && /[0-9.]/.test(sql[j]))
159
+ j++;
160
+ out.push({ type: 'word', value: sql.slice(i, j), start: offset + i, end: offset + j });
161
+ i = j;
162
+ continue;
163
+ }
164
+ out.push({ type: 'punct', value: ch, start: offset + i, end: offset + i + 1 });
165
+ i++;
166
+ }
167
+ return out;
168
+ }
169
+ /** Tokens that carry executable SQL: comments and literals are dropped. */
170
+ export function codeTokens(tokens) {
171
+ return tokens.filter((t) => t.type !== 'comment');
172
+ }
@@ -48,3 +48,15 @@ export function decodeJwtPayload(token) {
48
48
  return null;
49
49
  }
50
50
  }
51
+ /**
52
+ * A DNS label, bounded to its real maximum (RFC 1035, 63 octets) — never `+`.
53
+ * Used right before a literal host suffix (`.firebaseapp.com`, `.supabase.co`)
54
+ * in a regex run against raw, unbounded file content. An open `[a-z0-9-]+` in
55
+ * that position is a classic quadratic-time regex: at every position inside a
56
+ * long run of matching characters (a minified bundle, a base64 blob, a lockfile
57
+ * hash) the engine greedily consumes to the end and backtracks one character at
58
+ * a time looking for a literal that never comes. Measured: a 500KB matching run
59
+ * took over two minutes unbounded; bounded, the same file scans in single-digit
60
+ * milliseconds — and no real hostname label is longer than this anyway.
61
+ */
62
+ export const DNS_LABEL = '[a-z0-9-]{1,63}';
@@ -1,10 +1,32 @@
1
1
  import { readdirSync, statSync, readFileSync } from 'node:fs';
2
- import { join, relative, extname, sep } from 'node:path';
2
+ import { join, relative, extname, sep, resolve } from 'node:path';
3
3
  const SKIP_DIRS = new Set([
4
4
  '.git', 'node_modules', '.next', 'dist', 'build', 'out', '.venv', 'venv',
5
5
  '__pycache__', 'coverage', '.turbo', '.cache', 'vendor', '.svelte-kit',
6
6
  '.nuxt', '.output', 'target', '.idea', '.vscode', 'easyvibegate-report',
7
+ // Installed third-party code, not the user's own. Matching only the venv
8
+ // folder names above misses a venv called anything else (tools/ytenv/...),
9
+ // and then every key inside a vendored library is reported as the user's leak.
10
+ 'site-packages', '__pypackages__', 'bower_components', 'Pods',
11
+ // Yarn Berry vendors its own release script and zips dependencies here —
12
+ // tool-managed, not the user's code (and often full of high-entropy blobs
13
+ // that would otherwise read as secrets).
14
+ '.yarn',
7
15
  ]);
16
+ /**
17
+ * Names ambiguous enough that they are sometimes real source (a module named
18
+ * "cache", a package called "tmp") and sometimes pure data. Skipped only when
19
+ * `relDir` (the ambiguous directory's own path, relative to the project root,
20
+ * e.g. "cache" or "data/cache") has at most 2 path segments — i.e. the
21
+ * directory IS the root's own child, or is nested exactly one level below it.
22
+ * The real-world evidence for this was `data/cache/*.json` full of API
23
+ * pagination tokens, 559 of 569 "generic secrets" in one real project. Deeper
24
+ * nesting (`apps/api/src/lib/cache/`, 4 segments) is ordinary source and stays
25
+ * scanned; blanket name-matching at any depth once made a source directory
26
+ * invisible to every check with no visible coverage gap.
27
+ */
28
+ const AMBIGUOUS_DATA_DIRS = new Set(['cache', 'caches', 'tmp', 'temp', '.tmp']);
29
+ const isShallowDataDir = (name, relDir) => AMBIGUOUS_DATA_DIRS.has(name) && relDir.split('/').length <= 2;
8
30
  const TEXT_EXT = new Set([
9
31
  '.js', '.jsx', '.ts', '.tsx', '.mjs', '.cjs', '.vue', '.svelte',
10
32
  '.py', '.rb', '.php', '.go', '.rs', '.java', '.kt', '.cs',
@@ -27,11 +49,13 @@ function isScannable(name) {
27
49
  return true;
28
50
  return TEXT_EXT.has(extname(name).toLowerCase());
29
51
  }
30
- /** Recursively collect scannable text files under `root`, skipping noise. */
31
- export function walk(root) {
52
+ export function walk(root, opts = {}) {
53
+ const excluded = new Set((opts.excludeAbs ?? []).map((d) => resolve(d)));
32
54
  const out = [];
33
55
  let skippedOversized = 0;
34
56
  let skippedUnreadable = 0;
57
+ let skippedDirs = 0;
58
+ let skippedSymlinks = 0;
35
59
  const stack = [root];
36
60
  while (stack.length > 0) {
37
61
  const dir = stack.pop();
@@ -40,14 +64,35 @@ export function walk(root) {
40
64
  entries = readdirSync(dir, { withFileTypes: true });
41
65
  }
42
66
  catch {
67
+ // A directory we cannot list is an unchecked subtree, not an empty one.
68
+ // Swallowing this silently let a project with an unreadable folder report
69
+ // full coverage and a clean PASS.
70
+ skippedDirs++;
43
71
  continue;
44
72
  }
45
73
  for (const ent of entries) {
46
- if (ent.isSymbolicLink())
47
- continue;
48
74
  const full = join(dir, ent.name);
75
+ if (ent.isSymbolicLink()) {
76
+ // Never followed: a link can loop or escape the project root. But a
77
+ // link to a directory or a scannable file is unchecked content, and
78
+ // silently dropping it let a project whose `migrations` was a symlink
79
+ // PASS with full coverage. Count it so the walk is reported partial.
80
+ const relLink = relative(root, full).split(sep).join('/');
81
+ if (SKIP_DIRS.has(ent.name) || isShallowDataDir(ent.name, relLink))
82
+ continue;
83
+ try {
84
+ const target = statSync(full); // follows the link
85
+ if (target.isDirectory() || (target.isFile() && isScannable(ent.name)))
86
+ skippedSymlinks++;
87
+ }
88
+ catch {
89
+ /* dangling link — nothing behind it to scan */
90
+ }
91
+ continue;
92
+ }
49
93
  if (ent.isDirectory()) {
50
- if (!SKIP_DIRS.has(ent.name))
94
+ const relDir = relative(root, full).split(sep).join('/');
95
+ if (!SKIP_DIRS.has(ent.name) && !isShallowDataDir(ent.name, relDir) && !excluded.has(resolve(full)))
51
96
  stack.push(full);
52
97
  continue;
53
98
  }
@@ -82,5 +127,5 @@ export function walk(root) {
82
127
  });
83
128
  }
84
129
  }
85
- return { files: out, skippedOversized, skippedUnreadable };
130
+ return { files: out, skippedOversized, skippedUnreadable, skippedDirs, skippedSymlinks };
86
131
  }