jules-orchestrator-kit 0.72.2 → 0.73.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/.agent/prompts/{Overseer.md → Auditor.md} +3 -3
  2. package/.agent/prompts/{Alchemist.md → Database.md} +1 -1
  3. package/.agent/prompts/Debugger.md +25 -0
  4. package/.agent/prompts/{Scribe.md → Docs.md} +8 -5
  5. package/.agent/prompts/{Spectator.md → E2E.md} +9 -6
  6. package/.agent/prompts/{Janitor.md → Hygiene.md} +2 -2
  7. package/.agent/prompts/{Bolt.md → Performance.md} +1 -1
  8. package/.agent/prompts/Resilience.md +20 -0
  9. package/.agent/prompts/Security.md +21 -0
  10. package/.agent/prompts/Testing.md +30 -0
  11. package/.agent/prompts/Types.md +19 -0
  12. package/.agent/rules/jules-protocol.md +4 -3
  13. package/AGENTS.md +78 -96
  14. package/CHANGELOG.md +193 -0
  15. package/JULES_RULES_TEMPLATE.md +83 -96
  16. package/LICENSE +1 -1
  17. package/README.md +77 -439
  18. package/ROADMAP_V1.md +22 -132
  19. package/bin/agentctl.mjs +443 -144
  20. package/bin/init.js +6 -3
  21. package/index.mjs +9 -6
  22. package/package.json +1 -1
  23. package/scripts/asset-integrity-check.mjs +1 -1
  24. package/scripts/doc-sync-check.mjs +47 -0
  25. package/scripts/generate-command-reference.mjs +39 -0
  26. package/scripts/jules-dispatch.mjs +12 -113
  27. package/scripts/jules-merge-swarm.mjs +8 -196
  28. package/scripts/jules-patch.mjs +7 -8
  29. package/scripts/jules-queue-runner.mjs +6 -8
  30. package/scripts/jules-scan-todos.mjs +10 -38
  31. package/scripts/jules-self-audit.mjs +8 -139
  32. package/scripts/jules-status.mjs +32 -38
  33. package/scripts/jules-webhook-receiver.mjs +1 -1
  34. package/src/assertions.mjs +5 -50
  35. package/src/bidi-guard.mjs +36 -0
  36. package/src/budget.mjs +3 -14
  37. package/src/config.mjs +2 -6
  38. package/src/dashboard.mjs +7 -9
  39. package/src/dispatch.mjs +212 -0
  40. package/src/engine.mjs +40 -16
  41. package/src/evidence.mjs +10 -41
  42. package/src/execution-envelope.mjs +13 -1
  43. package/src/flaky-ledger.mjs +1 -1
  44. package/src/fs-atomic.mjs +72 -0
  45. package/src/git.mjs +298 -27
  46. package/src/mcp.mjs +296 -7
  47. package/src/memory.mjs +0 -0
  48. package/src/merge-swarm.mjs +202 -0
  49. package/src/ops/cli-intent.mjs +1 -0
  50. package/src/ops/command-registry.mjs +796 -70
  51. package/src/ops/doctor-registry.mjs +134 -47
  52. package/src/ops/handover.mjs +3 -27
  53. package/src/ops/pr-harvest.mjs +1 -1
  54. package/src/prompt-guard.mjs +33 -3
  55. package/src/provider.mjs +51 -7
  56. package/src/remediation.mjs +2 -2
  57. package/src/role-resolver.mjs +113 -3
  58. package/src/router.mjs +19 -11
  59. package/src/runtime-env.mjs +67 -0
  60. package/src/scaffold.mjs +3 -1
  61. package/src/scope-guard.mjs +249 -0
  62. package/src/secret-scanner.mjs +530 -0
  63. package/src/security.mjs +81 -2972
  64. package/src/self-audit.mjs +140 -0
  65. package/src/session-ops.mjs +29 -0
  66. package/src/stability.mjs +8 -1
  67. package/src/stack-detector.mjs +5 -2
  68. package/src/state.mjs +45 -0
  69. package/src/swarm.mjs +76 -0
  70. package/src/task-optimizer.mjs +34 -7
  71. package/src/telemetry.mjs +23 -0
  72. package/src/test-tamper-guard.mjs +2173 -0
  73. package/src/todo-scanner.mjs +129 -0
  74. package/src/web-templates.mjs +3 -3
  75. package/src/webhook.mjs +10 -3
  76. package/src/wizard-init.mjs +12 -20
  77. package/src/wizard-oracle.mjs +4 -3
  78. package/src/wizard-task.mjs +37 -9
  79. package/.agent/prompts/Sentinel.md +0 -18
  80. package/scripts/utils.mjs +0 -241
@@ -0,0 +1,530 @@
1
+ /**
2
+ * Credential and PII detection over arbitrary text.
3
+ *
4
+ * Split out of src/security.mjs (P05). Everything here answers one of two
5
+ * questions about a blob of text: "does it contain a secret" and "give me a copy
6
+ * with the secret removed". It is the scanner half of the old bundle; deciding
7
+ * what to do with a finding (scanDiff, binary payload scanning, the diff parser)
8
+ * stayed in the facade so this module never has to know what a unified diff is.
9
+ *
10
+ * Evasion hardening lives here: invisible-character stripping, NFKD plus a
11
+ * homoglyph table, string-concatenation rejoining, and hex/percent/base64
12
+ * decoding — all of it so a credential cannot hide behind a substituted glyph,
13
+ * a line wrap, or an encoding.
14
+ */
15
+
16
+ export const HIGH_CONFIDENCE_PATTERNS = [
17
+ /\bghp_[A-Za-z0-9_]{36,255}\b/g,
18
+ /\bgho_[A-Za-z0-9_]{36,255}\b/g,
19
+ /\bghu_[A-Za-z0-9_]{36,255}\b/g,
20
+ /\bghs_[A-Za-z0-9_]{36,255}\b/g,
21
+ /\bghr_[A-Za-z0-9_]{36,255}\b/g,
22
+ /\bgithub_pat_[A-Za-z0-9_]{22,255}\b/g,
23
+
24
+ /\bAKIA[0-9A-Z]{16}\b/g,
25
+ /\bASIA[0-9A-Z]{16}\b/g,
26
+ /\baws_secret_access_key\s*[:=]\s*['"]?[A-Za-z0-9\/+=]{40}['"]?/gi,
27
+
28
+ /-----BEGIN (?:RSA|DSA|EC|OPENSSH|ENCRYPTED|PRIVATE)(?:\s+PRIVATE)? KEY-----[\s\S]*?-----END (?:RSA|DSA|EC|OPENSSH|ENCRYPTED|PRIVATE)(?:\s+PRIVATE)? KEY-----/g,
29
+ /PuTTY-User-Key-File-[0-9]:[^\n]+/g,
30
+
31
+ /\bsk_live_[0-9a-zA-Z]{24,99}\b/g,
32
+ /\brk_live_[0-9a-zA-Z]{24,99}\b/g,
33
+ /\bnpm_[0-9a-zA-Z]{36}\b/g,
34
+ /\/\/[^/\s]+\/:_authToken=[A-Za-z0-9_-]{20,}/g,
35
+ /\bglpat-[0-9a-zA-Z_-]{20,99}\b/g,
36
+ /\bGOCSPX-[0-9a-zA-Z_-]{28,99}\b/g,
37
+
38
+ /\bAIzaSy[A-Za-z0-9_-]{33}\b/g,
39
+ /\bya29\.[A-Za-z0-9_-]{20,255}\b/g,
40
+ /\b(?:sk-ant-api03-|sk-proj-|sk-)[A-Za-z0-9_-]{20,255}\b/g,
41
+
42
+ /https:\/\/hooks\.slack\.com\/services\/T[A-Za-z0-9_]{8,}\/B[A-Za-z0-9_]{8,}\/[A-Za-z0-9_]{24,}/g,
43
+ /\bxox[baprs]-[0-9a-zA-Z-]{10,48}\b/g,
44
+
45
+ /\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b/g,
46
+ ];
47
+
48
+ export const LOW_CONFIDENCE_PATTERNS = [
49
+ /Bearer\s+[A-Za-z0-9._~+/-]{10,255}/gi,
50
+ /Authorization:\s*Bearer\s+[A-Za-z0-9._~+/-]{10,255}/gi,
51
+ /\bsk_test_[0-9a-zA-Z]{24,99}\b/g,
52
+ /(?:api[_-]?key|secret|password|passwd|token|auth[_-]?token)\s*[:=]\s*(?:['"`]([^'"`\n]{8,128})['"`]|([A-Za-z0-9._~+/-]{16,128}))/gi,
53
+ ];
54
+
55
+ export function shannonEntropy(str) {
56
+ if (!str || typeof str !== "string") return 0;
57
+ const len = str.length;
58
+ const frequencies = {};
59
+ for (let i = 0; i < len; i++) {
60
+ const char = str[i];
61
+ frequencies[char] = (frequencies[char] || 0) + 1;
62
+ }
63
+ let entropy = 0;
64
+ for (const char in frequencies) {
65
+ const p = frequencies[char] / len;
66
+ entropy -= p * Math.log2(p);
67
+ }
68
+ return entropy;
69
+ }
70
+
71
+ export function hasHighConfidenceSecret(text) {
72
+ if (!text) return false;
73
+ return HIGH_CONFIDENCE_PATTERNS.some((pat) => {
74
+ pat.lastIndex = 0;
75
+ const res = pat.test(text);
76
+ pat.lastIndex = 0;
77
+ return res;
78
+ });
79
+ }
80
+
81
+ export function hasLowConfidenceSecret(text) {
82
+ if (!text) return false;
83
+ return LOW_CONFIDENCE_PATTERNS.some((pat) => {
84
+ pat.lastIndex = 0;
85
+ const res = pat.test(text);
86
+ pat.lastIndex = 0;
87
+ return res;
88
+ });
89
+ }
90
+
91
+ export function redactSecrets(text) {
92
+ if (!text) return "";
93
+ let sanitized = text;
94
+
95
+ for (const [envKey, envVal] of Object.entries(process.env)) {
96
+ if (
97
+ envVal &&
98
+ (envVal.length >= 20 || shannonEntropy(envVal) > 3.6) &&
99
+ /KEY|SECRET|TOKEN|PASSWORD|CREDENTIAL|AUTH|PASSPHRASE|URL|URI|DSN|CONNECTION|ACCOUNT/i.test(envKey)
100
+ ) {
101
+ if (sanitized.includes(envVal)) {
102
+ sanitized = sanitized.split(envVal).join("[REDACTED_ENV_SECRET]");
103
+ }
104
+ }
105
+ }
106
+
107
+ const allPatterns = [...HIGH_CONFIDENCE_PATTERNS, ...LOW_CONFIDENCE_PATTERNS];
108
+ for (const pat of allPatterns) {
109
+ pat.lastIndex = 0;
110
+ sanitized = sanitized.replace(pat, "[REDACTED_BY_SECURITY_GATE]");
111
+ }
112
+
113
+ // A key the scanner can find inside a base64 blob must not survive redaction
114
+ // just because the literal bytes differ — otherwise scanDiff blocks the
115
+ // dispatch and the escalation payload leaks the very value it blocked on. The
116
+ // whole blob goes, not part of it: a partially-redacted encoding still
117
+ // decodes to the key.
118
+ const encoded = new Set();
119
+ decodeBase64Blobs(sanitized, (plain, blob) => {
120
+ if (hasHighConfidenceSecret(plain)) encoded.add(blob);
121
+ });
122
+ for (const blob of encoded) {
123
+ sanitized = sanitized.split(blob).join("[REDACTED_ENCODED_SECRET]");
124
+ }
125
+
126
+ return sanitized;
127
+ }
128
+
129
+ export function anonymizePii(text) {
130
+ if (!text) return "";
131
+ let sanitized = text;
132
+
133
+ sanitized = sanitized.replace(/\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}\b/g, "[REDACTED_EMAIL]");
134
+ sanitized = sanitized.replace(/\b(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\b/g, (ip) => {
135
+ if (ip === "127.0.0.1" || ip === "0.0.0.0") return ip;
136
+ return "[REDACTED_IP]";
137
+ });
138
+ sanitized = sanitized.replace(/(?:(?:\+\d{1,3}[\s-]?)|\b)\(?\d{2,4}\)?(?:[\s-]?\d{2,4}){2,4}\b/g, (phone) => {
139
+ const digitsOnly = phone.replace(/\D/g, "");
140
+ if (digitsOnly.length >= 7 && digitsOnly.length <= 15) {
141
+ return "[REDACTED_PHONE]";
142
+ }
143
+ return phone;
144
+ });
145
+
146
+ return sanitized;
147
+ }
148
+
149
+ // Zero-width, bidi-control characters, and Unicode tag plane (U+E0000..U+E007F).
150
+ // Inserting one mid-token defeats a regex without changing how the value renders, copies, or authenticates.
151
+ const INVISIBLE_CHARS = /[\u00AD\u200B-\u200F\u2028\u2029\u202A-\u202E\u2060-\u2064\u2066-\u2069\uFEFF]|[\u{E0000}-\u{E007F}]/gu;
152
+
153
+ // Unicode lookalikes that NFKD does NOT decompose. Full-width and other
154
+ // compatibility forms are handled by String#normalize("NFKD") below; these are
155
+ // the Cyrillic/Greek/Latin homoglyphs that survive NFKD because they are
156
+ // distinct code points with no compatibility decomposition. A credential
157
+ // scanner without this table can be defeated by a single substituted glyph,
158
+ // e.g. `ghp_` spelled with Cyrillic `р`.
159
+ const CONFUSABLE_TO_ASCII = new Map([
160
+ // Cyrillic
161
+ ["А", "A"], ["В", "B"], ["Е", "E"], ["К", "K"], ["М", "M"], ["Н", "H"],
162
+ ["О", "O"], ["Р", "P"], ["С", "C"], ["Т", "T"], ["У", "Y"], ["Х", "X"],
163
+ ["а", "a"], ["е", "e"], ["о", "o"], ["р", "p"], ["с", "c"], ["у", "y"],
164
+ ["х", "x"], ["і", "i"], ["ј", "j"], ["ѕ", "s"],
165
+ // Greek
166
+ ["Α", "A"], ["Β", "B"], ["Ε", "E"], ["Ζ", "Z"], ["Η", "H"], ["Ι", "I"],
167
+ ["Κ", "K"], ["Μ", "M"], ["Ν", "N"], ["Ο", "O"], ["Ρ", "P"], ["Τ", "T"],
168
+ ["Υ", "Y"], ["Χ", "X"], ["ο", "o"], ["ι", "i"], ["ν", "v"], ["υ", "u"],
169
+ ["ρ", "p"], ["τ", "t"], ["χ", "x"],
170
+ // Other Unicode lookalikes
171
+ ["ſ", "s"], // U+017F LATIN SMALL LETTER LONG S
172
+ ["K", "K"], // U+212A KELVIN SIGN
173
+ ]);
174
+
175
+ const CONFUSABLE_REGEX = new RegExp([...CONFUSABLE_TO_ASCII.keys()].join("|"), "g");
176
+
177
+ /**
178
+ * Reduces the confusable spellings a credential can hide behind to plain
179
+ * ASCII before the secret patterns run (`SEC-04`).
180
+ *
181
+ * 1. NFKD decomposes full-width and other compatibility forms
182
+ * (`ghp_…` → `ghp_…`).
183
+ * 2. Combining marks the decomposition may leave behind are stripped
184
+ * (`e\u0301` → `e`).
185
+ * 3. The curated lookalike table maps Cyrillic/Greek/Latin homoglyphs that
186
+ * NFKD cannot see through to their ASCII target.
187
+ *
188
+ * This is the zero-dependency subset of Unicode TR39 confusable handling; the
189
+ * full confusables data table is intentionally omitted so the kit keeps
190
+ * shipping with no runtime dependencies.
191
+ *
192
+ * @param {string} str
193
+ * @returns {string}
194
+ */
195
+ function normalizeSecretText(str) {
196
+ if (!str || typeof str !== "string") return str;
197
+ let out = str;
198
+ try {
199
+ out = out.normalize("NFKD");
200
+ } catch (_) {}
201
+ out = out.replace(/[\u0300-\u036f]/g, "");
202
+ out = out.replace(CONFUSABLE_REGEX, (m) => CONFUSABLE_TO_ASCII.get(m));
203
+ return out;
204
+ }
205
+
206
+ // A credential split across a source-level string concatenation is invisible to
207
+ // a line-oriented scanner. This is not only an evasion technique — formatters
208
+ // wrap long string literals exactly this way, so it also happens by accident.
209
+ const STRING_CONCAT_JOIN = /(["'`])\s*(?:\/\*[\s\S]*?\*\/)?\s*\+\s*(?:\/\*[\s\S]*?\*\/)?\s*(["'`])/g;
210
+
211
+ function tryHexDecodeTokens(str) {
212
+ return str.replace(/\b([0-9a-fA-F]{24,})\b/g, (match) => {
213
+ if (match.length % 2 !== 0) return match;
214
+ try {
215
+ const decoded = Buffer.from(match, "hex").toString("utf-8");
216
+ if (printableRatio(decoded) >= 0.9) return decoded;
217
+ } catch (_) {}
218
+ return match;
219
+ });
220
+ }
221
+
222
+ function tryPercentDecode(str) {
223
+ try {
224
+ return decodeURIComponent(str);
225
+ } catch (_) {
226
+ return str.replace(/%([0-9a-fA-F]{2})/g, (_, hex) => {
227
+ try {
228
+ return String.fromCharCode(parseInt(hex, 16));
229
+ } catch (_) {
230
+ return _;
231
+ }
232
+ });
233
+ }
234
+ }
235
+
236
+ /**
237
+ * Produces the variants of the added-line text that secret patterns are run
238
+ * against: as-written, with invisible characters stripped, and with
239
+ * source-level string concatenation collapsed.
240
+ *
241
+ * Exported so the diff scanner in `security.mjs` can run its classification
242
+ * over the same variants without re-deriving them; it is not part of the
243
+ * `security.mjs` public surface and is deliberately not re-exported there.
244
+ *
245
+ * @param {string} addedLines
246
+ * @returns {{ all: string[], normalized: string }}
247
+ */
248
+ export function secretScanVariants(addedLines) {
249
+ const stripped = addedLines.replace(INVISIBLE_CHARS, "");
250
+ // Collapse `"AAA" +\n "BBB"` into `"AAABBB"` before matching.
251
+ let dejoined = stripped.replace(/\s*\n\s*/g, " ").replace(STRING_CONCAT_JOIN, "");
252
+ // Collapse template literal empty expressions `${""}` and `${"VALUE"}`
253
+ dejoined = dejoined.replace(/\$\{\s*["'`]{2}\s*\}/g, "").replace(/\$\{\s*["'`]([^"'`]+)["'`]\s*\}/g, "$1");
254
+ // Collapse method concatenations like .concat("...") or .join("")
255
+ dejoined = dejoined.replace(/\.concat\(\s*["'`]/g, "").replace(/\.join\(\s*["'`]{2}\s*\)/g, "");
256
+
257
+ // Collapse whitespace/newlines between adjacent base64 characters (including line-wrapped PEM/base64, template literals, and quoted string chunks)
258
+ const base64Dejoined = stripped
259
+ .replace(/([A-Za-z0-9+/=_-])\s*[\r\n]+\s*(?=[A-Za-z0-9+/=_-])/g, "$1")
260
+ .replace(/([A-Za-z0-9+/=_-])["'`]\s*(?:\+\s*)?[\r\n]+\s*["'`]?([A-Za-z0-9+/=_-])/g, "$1$2");
261
+
262
+ const hexDecoded = tryHexDecodeTokens(dejoined);
263
+ const pctDecoded = tryPercentDecode(dejoined);
264
+ // Confusable / NFKD normalisation runs over both the raw text and the
265
+ // concatenation-collapsed text, so a credential that is both split across a
266
+ // source-level join AND spelled with homoglyphs still surfaces.
267
+ const confusable = normalizeSecretText(stripped);
268
+ const confusableDejoined = normalizeSecretText(dejoined);
269
+
270
+ return {
271
+ all: [...new Set([addedLines, stripped, dejoined, base64Dejoined, hexDecoded, pctDecoded, confusable, confusableDejoined])],
272
+ normalized: dejoined,
273
+ base64Normalized: base64Dejoined,
274
+ };
275
+ }
276
+
277
+ // Base64 is less an evasion technique than a file format. Every value in a
278
+ // Kubernetes Secret manifest is base64 by specification, and whole `.env` files
279
+ // get encoded into a single CI variable.
280
+ const BASE64_CANDIDATE = /[A-Za-z0-9+/\-_]{20,}={0,2}/g;
281
+
282
+ // Budgets the decoder spends before it gives up and reports `capped`.
283
+ //
284
+ // The count that matters is payloads *retained* — blobs that decoded to text
285
+ // and so could be carrying a credential. Counting every token that merely
286
+ // matches the base64 alphabet instead made a digest indistinguishable from a
287
+ // payload: a sha256 hex string is 64 characters of that alphabet, decodes to
288
+ // binary, gets discarded, and used to consume a slot anyway. Any diff holding
289
+ // 65 hashes — every lockfile bump — then tripped the cap and failed closed as
290
+ // a CRITICAL credential leak with no credential anywhere in it.
291
+ const BASE64_MAX_CANDIDATES = 64;
292
+ const BASE64_MAX_TOKENS_EXAMINED = 8192;
293
+ const BASE64_MAX_DECODED_BYTES = 64 * 1024;
294
+ // Per-blob ceiling, so one oversized payload cannot spend the whole budget and
295
+ // starve the blobs after it. The trade-off is deliberate: a credential buried
296
+ // past 8 KB inside a single blob is missed, where the old code caught it only
297
+ // by refusing to decode and then failing the entire diff closed. That refusal
298
+ // fired on every checked-in base64 asset, and a gate that cries wolf on
299
+ // ordinary input gets switched off. The cleartext scanners still run over the
300
+ // raw diff regardless.
301
+ const BASE64_MAX_BLOB_BYTES = 8 * 1024;
302
+
303
+ /**
304
+ * Share of characters that are printable ASCII (plus tab/newline/return).
305
+ */
306
+ function printableRatio(str) {
307
+ if (!str) return 0;
308
+ let printable = 0;
309
+ for (let i = 0; i < str.length; i++) {
310
+ const c = str.charCodeAt(i);
311
+ if (c === 9 || c === 10 || c === 13 || (c >= 32 && c <= 126)) printable++;
312
+ }
313
+ return printable / str.length;
314
+ }
315
+
316
+ /**
317
+ * Decode the base64-looking blobs in `text` that plausibly hold text.
318
+ *
319
+ * @param {string} text
320
+ * @param {(plain: string, blob: string) => void} [onDecoded] - Called per blob.
321
+ * @returns {{ decoded: string[], capped: boolean }}
322
+ */
323
+ function decodeBase64Blobs(text, onDecoded) {
324
+ if (!text) return { decoded: [], capped: false };
325
+ const decoded = [];
326
+ let examined = 0;
327
+ let retained = 0;
328
+ let bytes = 0;
329
+ let capped = false;
330
+
331
+ BASE64_CANDIDATE.lastIndex = 0;
332
+ let match;
333
+ while ((match = BASE64_CANDIDATE.exec(text)) !== null) {
334
+ if (examined++ >= BASE64_MAX_TOKENS_EXAMINED) {
335
+ capped = true;
336
+ break;
337
+ }
338
+ const rawBlob = match[0].replace(/[\s\r\n]+/g, "");
339
+ let stdBlob = rawBlob.replace(/-/g, "+").replace(/_/g, "/");
340
+ while (stdBlob.length % 4 !== 0) {
341
+ stdBlob += "=";
342
+ }
343
+
344
+ // An oversized blob is decoded up to a bounded prefix rather than skipped
345
+ // outright. Base64 decodes in independent 4-character groups, so a prefix
346
+ // is exact, and a credential near the head of a large payload still
347
+ // surfaces — where skipping used to hide it and then blame the whole diff.
348
+ const budget = Math.min(BASE64_MAX_BLOB_BYTES, BASE64_MAX_DECODED_BYTES - bytes);
349
+ if (budget <= 0) {
350
+ capped = true;
351
+ break;
352
+ }
353
+ const maxChars = Math.floor(budget / 3) * 4;
354
+ if (stdBlob.length > maxChars) stdBlob = stdBlob.slice(0, maxChars);
355
+
356
+ let plain;
357
+ try {
358
+ plain = Buffer.from(stdBlob, "base64").toString("utf-8");
359
+ } catch (_) {
360
+ continue;
361
+ }
362
+ bytes += plain.length;
363
+
364
+ // A blob that decodes to binary has been examined and cleared. It is not a
365
+ // blind spot, so it must not spend a payload slot.
366
+ if (printableRatio(plain) < 0.9) continue;
367
+
368
+ if (retained++ >= BASE64_MAX_CANDIDATES) {
369
+ capped = true;
370
+ break;
371
+ }
372
+
373
+ decoded.push(plain);
374
+ if (onDecoded) onDecoded(plain, rawBlob);
375
+
376
+ // Try 1 level of nested base64 decoding if printable
377
+ if (/[A-Za-z0-9+/\-_]{20,}={0,2}/.test(plain)) {
378
+ try {
379
+ let nestedStd = plain.trim().replace(/-/g, "+").replace(/_/g, "/");
380
+ while (nestedStd.length % 4 !== 0) nestedStd += "=";
381
+ const nestedPlain = Buffer.from(nestedStd, "base64").toString("utf-8");
382
+ if (printableRatio(nestedPlain) >= 0.9) {
383
+ decoded.push(nestedPlain);
384
+ }
385
+ } catch (_) {}
386
+ }
387
+ }
388
+ BASE64_CANDIDATE.lastIndex = 0;
389
+ return { decoded, capped };
390
+ }
391
+
392
+ /**
393
+ * True when a base64-encoded value on an added line decodes to a structured
394
+ * credential.
395
+ *
396
+ * @param {string} text
397
+ * @returns {boolean}
398
+ */
399
+ export function hasEncodedSecret(text) {
400
+ if (!text) return false;
401
+ const result = decodeBase64Blobs(text);
402
+ if (result.capped) return true; // Fail closed if cap exceeded
403
+ if (result.decoded.some((plain) => hasHighConfidenceSecret(plain))) return true;
404
+
405
+ if (text.includes("\n") || text.includes("\r")) {
406
+ const collapsed = text
407
+ .replace(/([A-Za-z0-9+/=_-])\s*[\r\n]+\s*(?=[A-Za-z0-9+/=_-])/g, "$1")
408
+ .replace(/([A-Za-z0-9+/=_-])["'`]\s*(?:\+\s*)?[\r\n]+\s*["'`]?([A-Za-z0-9+/=_-])/g, "$1$2");
409
+ if (collapsed !== text) {
410
+ const collapsedResult = decodeBase64Blobs(collapsed);
411
+ if (collapsedResult.capped) return true;
412
+ if (collapsedResult.decoded.some((plain) => hasHighConfidenceSecret(plain))) return true;
413
+ }
414
+ }
415
+ return false;
416
+ }
417
+
418
+ const CANDIDATE_TOKEN_REGEX = /[A-Za-z0-9_-]{24,}/g;
419
+
420
+ /** A whole data: URI, base64-encoded or not. Its payload is not a credential. */
421
+ const DATA_URI_TOKEN = /\bdata:[a-z0-9.+-]+\/[a-z0-9.+-]*(?:;[a-z0-9.+=-]+)*,[^\s"'`<>)\]}]*/gi;
422
+
423
+ /** A subresource-integrity digest. High entropy by construction, public by design. */
424
+ const INTEGRITY_TOKEN = /\bsha(?:256|384|512)-[A-Za-z0-9+/=]+/gi;
425
+
426
+ /** A URL, from its scheme to the first character that cannot be part of one. */
427
+ const URL_TOKEN = /\b[a-z][a-z0-9+.-]*:\/\/[^\s"'`<>)\]},;]+/gi;
428
+
429
+ /**
430
+ * Remove from a line the noise that made URLs worth ignoring, and keep the
431
+ * parts of a URL that can carry a credential.
432
+ *
433
+ * This used to be `if (rawLine.includes("://")) continue;` — one substring
434
+ * anywhere on the line switched off entropy analysis for the entire line. So
435
+ * the scanner caught a bare 32-character key and let the identical key through
436
+ * the moment a comment carrying any http link sat beside it. An agent does not
437
+ * need to know why that works to stumble into it; a fetch call and its endpoint
438
+ * on one line is ordinary code.
439
+ *
440
+ * What actually justified the skip is narrower: a CDN path segment or an npm
441
+ * integrity hash looks exactly like a secret and is neither. Those are dropped
442
+ * here. A URL's userinfo and its query values are the opposite — `?api_key=…`
443
+ * and `//user:password@host` are where credentials genuinely hide — so they
444
+ * are carried over and scanned on their own.
445
+ */
446
+ function stripEntropyNoise(rawLine) {
447
+ const carried = [];
448
+ let line = rawLine.replace(DATA_URI_TOKEN, " ").replace(INTEGRITY_TOKEN, " ");
449
+ line = line.replace(URL_TOKEN, (url) => {
450
+ const afterScheme = url.slice(url.indexOf("://") + 3);
451
+ const authority = afterScheme.split(/[/?#]/)[0];
452
+ const at = authority.lastIndexOf("@");
453
+ if (at > 0) carried.push(authority.slice(0, at));
454
+ const q = url.indexOf("?");
455
+ if (q !== -1) {
456
+ for (const pair of url.slice(q + 1).split(/[&;#]/)) {
457
+ const eq = pair.indexOf("=");
458
+ if (eq !== -1) carried.push(pair.slice(eq + 1));
459
+ }
460
+ }
461
+ return " ";
462
+ });
463
+ return carried.length ? `${line} ${carried.join(" ")}` : line;
464
+ }
465
+
466
+ /**
467
+ * Checks for high-entropy continuous tokens (>= 24 chars, entropy > 4.5) on added lines.
468
+ * Strips URLs, data: URIs, SRI hashes (sha512-, sha256-, sha384-) and skips lockfiles
469
+ * to eliminate false positives.
470
+ *
471
+ * @param {string} text - Text to scan
472
+ * @param {string|null} [file=null] - File path associated with the text
473
+ * @returns {boolean}
474
+ */
475
+ export function hasHighEntropyToken(text = "", file = null) {
476
+ if (!text || typeof text !== "string") return false;
477
+ if (
478
+ file &&
479
+ (file.endsWith(".lock") ||
480
+ file.endsWith(".lockb") ||
481
+ file.includes("package-lock.json") ||
482
+ file.includes("pnpm-lock.yaml") ||
483
+ file.includes("yarn.lock") ||
484
+ file.includes("Cargo.lock") ||
485
+ file.includes("composer.lock"))
486
+ ) {
487
+ return false;
488
+ }
489
+
490
+ const lines = text.split("\n");
491
+ for (const rawLine of lines) {
492
+ const line = stripEntropyNoise(rawLine);
493
+
494
+ CANDIDATE_TOKEN_REGEX.lastIndex = 0;
495
+ let match;
496
+ while ((match = CANDIDATE_TOKEN_REGEX.exec(line)) !== null) {
497
+ const token = match[0];
498
+ if (token.startsWith("sha512-") || token.startsWith("sha256-")) continue;
499
+ if (token.length >= 24) {
500
+ // If the token is a formatted base64 blob, distinguish binary assets and plain prose
501
+ if (token.length % 4 === 0 && /^[A-Za-z0-9+/]+={0,2}$/.test(token)) {
502
+ try {
503
+ const plain = Buffer.from(token, "base64").toString("utf-8");
504
+ const pr = printableRatio(plain);
505
+ if (pr < 0.9) {
506
+ // Ordinary binary asset (e.g. icon/font/wasm) - do not trip on binary entropy.
507
+ // Only treat as binary asset if sufficiently large (>= 256 chars);
508
+ // shorter tokens (24-255 chars) are keys/secrets/hashes, not embedded assets.
509
+ if (token.length >= 256) {
510
+ continue;
511
+ }
512
+ } else {
513
+ // If it decodes to text, check decoded plain text entropy
514
+ if (shannonEntropy(plain) > 4.5) {
515
+ return true;
516
+ }
517
+ continue;
518
+ }
519
+ } catch (_) {}
520
+ }
521
+
522
+ const ent = shannonEntropy(token);
523
+ if (ent > 4.5) {
524
+ return true;
525
+ }
526
+ }
527
+ }
528
+ }
529
+ return false;
530
+ }