failproofai 1.0.4-beta.2 → 1.0.4-beta.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/.next/standalone/.next/BUILD_ID +1 -1
  2. package/.next/standalone/.next/build-manifest.json +3 -3
  3. package/.next/standalone/.next/prerender-manifest.json +3 -3
  4. package/.next/standalone/.next/required-server-files.json +1 -1
  5. package/.next/standalone/.next/server/app/_global-error/page/server-reference-manifest.json +1 -1
  6. package/.next/standalone/.next/server/app/_global-error/page.js.nft.json +1 -1
  7. package/.next/standalone/.next/server/app/_global-error/page_client-reference-manifest.js +1 -1
  8. package/.next/standalone/.next/server/app/_global-error.html +1 -1
  9. package/.next/standalone/.next/server/app/_global-error.rsc +7 -7
  10. package/.next/standalone/.next/server/app/_global-error.segments/__PAGE__.segment.rsc +6 -6
  11. package/.next/standalone/.next/server/app/_global-error.segments/_full.segment.rsc +7 -7
  12. package/.next/standalone/.next/server/app/_global-error.segments/_tree.segment.rsc +1 -1
  13. package/.next/standalone/.next/server/app/_not-found/page/server-reference-manifest.json +1 -1
  14. package/.next/standalone/.next/server/app/_not-found/page.js.nft.json +1 -1
  15. package/.next/standalone/.next/server/app/_not-found/page_client-reference-manifest.js +1 -1
  16. package/.next/standalone/.next/server/app/_not-found.html +1 -1
  17. package/.next/standalone/.next/server/app/_not-found.rsc +14 -14
  18. package/.next/standalone/.next/server/app/_not-found.segments/_full.segment.rsc +14 -14
  19. package/.next/standalone/.next/server/app/_not-found.segments/_not-found/__PAGE__.segment.rsc +13 -13
  20. package/.next/standalone/.next/server/app/_not-found.segments/_tree.segment.rsc +1 -1
  21. package/.next/standalone/.next/server/app/api/audit/invite/route.js.nft.json +1 -1
  22. package/.next/standalone/.next/server/app/api/audit/run/route.js +1 -1
  23. package/.next/standalone/.next/server/app/api/audit/run/route.js.nft.json +1 -1
  24. package/.next/standalone/.next/server/app/api/audit/status/route.js.nft.json +1 -1
  25. package/.next/standalone/.next/server/app/api/auth/login-request/route.js.nft.json +1 -1
  26. package/.next/standalone/.next/server/app/api/auth/login-verify/route.js.nft.json +1 -1
  27. package/.next/standalone/.next/server/app/api/auth/logout/route.js.nft.json +1 -1
  28. package/.next/standalone/.next/server/app/api/auth/status/route.js.nft.json +1 -1
  29. package/.next/standalone/.next/server/app/api/download/[project]/[session]/route.js.nft.json +1 -1
  30. package/.next/standalone/.next/server/app/audit/page/server-reference-manifest.json +5 -5
  31. package/.next/standalone/.next/server/app/audit/page.js.nft.json +1 -1
  32. package/.next/standalone/.next/server/app/audit/page_client-reference-manifest.js +1 -1
  33. package/.next/standalone/.next/server/app/index.html +1 -1
  34. package/.next/standalone/.next/server/app/index.rsc +14 -14
  35. package/.next/standalone/.next/server/app/index.segments/__PAGE__.segment.rsc +13 -13
  36. package/.next/standalone/.next/server/app/index.segments/_full.segment.rsc +14 -14
  37. package/.next/standalone/.next/server/app/index.segments/_tree.segment.rsc +1 -1
  38. package/.next/standalone/.next/server/app/page/server-reference-manifest.json +1 -1
  39. package/.next/standalone/.next/server/app/page.js.nft.json +1 -1
  40. package/.next/standalone/.next/server/app/page_client-reference-manifest.js +1 -1
  41. package/.next/standalone/.next/server/app/policies/page/server-reference-manifest.json +14 -14
  42. package/.next/standalone/.next/server/app/policies/page.js.nft.json +1 -1
  43. package/.next/standalone/.next/server/app/policies/page_client-reference-manifest.js +1 -1
  44. package/.next/standalone/.next/server/app/project/[name]/page/server-reference-manifest.json +1 -1
  45. package/.next/standalone/.next/server/app/project/[name]/page.js.nft.json +1 -1
  46. package/.next/standalone/.next/server/app/project/[name]/page_client-reference-manifest.js +1 -1
  47. package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page/react-loadable-manifest.json +2 -2
  48. package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page/server-reference-manifest.json +2 -2
  49. package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page.js.nft.json +1 -1
  50. package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page_client-reference-manifest.js +1 -1
  51. package/.next/standalone/.next/server/app/projects/page/server-reference-manifest.json +1 -1
  52. package/.next/standalone/.next/server/app/projects/page.js.nft.json +1 -1
  53. package/.next/standalone/.next/server/app/projects/page_client-reference-manifest.js +1 -1
  54. package/.next/standalone/.next/server/app/settings/page/server-reference-manifest.json +4 -4
  55. package/.next/standalone/.next/server/app/settings/page.js.nft.json +1 -1
  56. package/.next/standalone/.next/server/app/settings/page_client-reference-manifest.js +1 -1
  57. package/.next/standalone/.next/server/chunks/{[externals]__1lh7m5d._.js → [externals]__0v1hz15._.js} +1 -1
  58. package/.next/standalone/.next/server/chunks/{[externals]__1rqkg_y._.js → [externals]__1nb206a._.js} +1 -1
  59. package/.next/standalone/.next/server/chunks/[root-of-the-server]__0--lkk6._.js +1 -1
  60. package/.next/standalone/.next/server/chunks/[root-of-the-server]__0o07qi9._.js +1 -1
  61. package/.next/standalone/.next/server/chunks/[root-of-the-server]__0q0qzx1._.js +1 -1
  62. package/.next/standalone/.next/server/chunks/[root-of-the-server]__1le9jqc._.js +1 -1
  63. package/.next/standalone/.next/server/chunks/[root-of-the-server]__1p-qi2t._.js +1 -1
  64. package/.next/standalone/.next/server/chunks/{_08w6xzm._.js → _0otft92._.js} +2 -2
  65. package/.next/standalone/.next/server/chunks/_0tovk6q._.js +1 -1
  66. package/.next/standalone/.next/server/chunks/_0trp3yc._.js +1 -1
  67. package/.next/standalone/.next/server/chunks/_1ek68ln._.js +3 -3
  68. package/.next/standalone/.next/server/chunks/_1vslkoe._.js +1 -1
  69. package/.next/standalone/.next/server/chunks/lib_factory-projects_ts_1eo_bk-._.js +1 -1
  70. package/.next/standalone/.next/server/chunks/package_json_[json]_cjs_1nxcc4v._.js +1 -1
  71. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__013jr2b._.js +2 -2
  72. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__01wy8d-._.js +2 -2
  73. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__02npjtd._.js +2 -2
  74. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0da85px._.js +2 -2
  75. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0ftmoxc._.js +2 -2
  76. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0oa1lav._.js +1 -1
  77. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0p-5p8u._.js +2 -2
  78. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0s740oi._.js +2 -2
  79. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0yl9wb5._.js +3 -0
  80. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1p2otjt._.js +2 -2
  81. package/.next/standalone/.next/server/chunks/ssr/_08x1r5t._.js +1 -1
  82. package/.next/standalone/.next/server/chunks/ssr/_0guxnh-._.js +23 -0
  83. package/.next/standalone/.next/server/chunks/ssr/_0oq1dh6._.js +1 -1
  84. package/.next/standalone/.next/server/chunks/ssr/_1_ozgif._.js +1 -1
  85. package/.next/standalone/.next/server/chunks/ssr/_1es2j7i._.js +13 -0
  86. package/.next/standalone/.next/server/chunks/ssr/_1u8-lu2._.js +1 -1
  87. package/.next/standalone/.next/server/chunks/ssr/_next-internal_server_app_policies_page_actions_1sp2-yo.js +2 -2
  88. package/.next/standalone/.next/server/chunks/ssr/app_audit__components_audit-dashboard_tsx_0p9ud47._.js +1 -1
  89. package/.next/standalone/.next/server/chunks/ssr/app_global-error_tsx_1kp6l3x._.js +1 -1
  90. package/.next/standalone/.next/server/chunks/ssr/app_policies_hooks-client_tsx_19dqvpc._.js +1 -1
  91. package/.next/standalone/.next/server/chunks/ssr/app_settings_settings-client_tsx_20lq-mq._.js +1 -1
  92. package/.next/standalone/.next/server/middleware-build-manifest.js +3 -3
  93. package/.next/standalone/.next/server/pages/404.html +1 -1
  94. package/.next/standalone/.next/server/pages/500.html +1 -1
  95. package/.next/standalone/.next/server/server-reference-manifest.js +1 -1
  96. package/.next/standalone/.next/server/server-reference-manifest.json +22 -22
  97. package/.next/standalone/.next/static/chunks/{04r6ch8uf_n8m.js → 04o5zfl7p5lhx.js} +1 -1
  98. package/.next/standalone/.next/static/chunks/{0o-hh5_turzlz.css → 0jpd8dv2vk930.css} +1 -1
  99. package/.next/standalone/.next/static/chunks/{2ej3b8gk5ittu.js → 0po8n4tyotjf2.js} +1 -1
  100. package/.next/standalone/.next/static/chunks/0qx7gav1c2_vu.js +1 -0
  101. package/.next/standalone/.next/static/chunks/{2aquitk72k2op.js → 17tfki9x5rd_7.js} +1 -1
  102. package/.next/standalone/.next/static/chunks/{010bv1w6j171t.js → 1i-i0pxg9q_8y.js} +1 -1
  103. package/.next/standalone/.next/static/chunks/{32spub4wqjem-.js → 2bxr2_2h4kaqq.js} +1 -1
  104. package/.next/standalone/.next/static/chunks/{0wz8yftk18ts2.js → 323cyb3a2s06y.js} +1 -1
  105. package/.next/standalone/.next/static/chunks/{1pb1oztsbwcss.js → 364cqbpeg1y5i.js} +1 -1
  106. package/.next/standalone/.next/static/chunks/{3m4upvybtrexd.js → 38dqur43qthrl.js} +1 -1
  107. package/.next/standalone/.next/static/chunks/{2bi_1y0a_smt7.js → 42gwu2bsz4ske.js} +2 -2
  108. package/.next/standalone/app/actions/get-leaks.ts +10 -1
  109. package/.next/standalone/app/audit/_components/audit-dashboard.tsx +29 -2
  110. package/.next/standalone/app/audit/_components/leak-section.tsx +15 -2
  111. package/.next/standalone/app/audit/audit-styles.css +18 -0
  112. package/.next/standalone/lib/sqlite-reader.ts +78 -1
  113. package/.next/standalone/package.json +10 -10
  114. package/.next/standalone/server.js +1 -1
  115. package/bin/failproofai.mjs +1 -1
  116. package/dist/cli.mjs +397 -225
  117. package/dist/worker.mjs +81 -35
  118. package/lib/sqlite-reader.ts +78 -1
  119. package/package.json +10 -10
  120. package/pi-extension/index.ts +19 -4
  121. package/src/audit/cache.ts +6 -0
  122. package/src/audit/cli-adapters/claude.ts +13 -0
  123. package/src/audit/cli-login.ts +4 -1
  124. package/src/audit/cli.ts +84 -59
  125. package/src/audit/index.ts +149 -8
  126. package/src/audit/leak-notice.ts +99 -1
  127. package/src/audit/leak-scan.ts +20 -44
  128. package/src/audit/leak-store.ts +49 -1
  129. package/src/audit/redact-example.ts +60 -5
  130. package/src/audit/types.ts +6 -0
  131. package/src/hooks/handler.ts +30 -10
  132. package/src/hooks/integrations.ts +14 -0
  133. package/src/hooks/notice.ts +45 -9
  134. package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0rgu2r3._.js +0 -3
  135. package/.next/standalone/.next/server/chunks/ssr/_0-oij9d._.js +0 -23
  136. package/.next/standalone/.next/static/chunks/0zebh1n9jkfbt.js +0 -1
  137. /package/.next/standalone/.next/static/{aKNv4Kmv98Xpwns0Xa-C- → P6icI2V2PvVuePJrqjHCr}/_buildManifest.js +0 -0
  138. /package/.next/standalone/.next/static/{aKNv4Kmv98Xpwns0Xa-C- → P6icI2V2PvVuePJrqjHCr}/_clientMiddlewareManifest.js +0 -0
  139. /package/.next/standalone/.next/static/{aKNv4Kmv98Xpwns0Xa-C- → P6icI2V2PvVuePJrqjHCr}/_ssgManifest.js +0 -0
package/src/audit/cli.ts CHANGED
@@ -37,7 +37,9 @@ import { openWhenReady } from "./open-browser";
37
37
  import { describeOutcome, reportHarm } from "./report-harm";
38
38
  import { notifyDesktop } from "./desktop-notify";
39
39
  import { macNotifierInstalled, pruneMacNotifyQueue, queueMacNotification } from "./macos-notifier";
40
- import { markLeakNoticeDelivered } from "./leak-notice";
40
+ import { markLeakNoticeDelivered, releaseLeakNoticeClaims } from "./leak-notice";
41
+ import { desktopLeakNotice } from "../hooks/notice";
42
+ import { activeFindings, readLeakRecord } from "./leak-store";
41
43
  import { readConfig } from "../hooks/fp-config";
42
44
  import { brandAnsi, ANSI_RESET, ANSI_BOLD, ANSI_DIM, helpScreen, helpOptsFor } from "../hooks/tui";
43
45
  import { version } from "../../package.json";
@@ -96,19 +98,19 @@ export function helpText(): string {
96
98
  {
97
99
  command: "audit",
98
100
  version,
99
- tagline: "review your agent CLIs for risky and wasteful patterns",
101
+ tagline: "find credentials your agents leaked into their own transcripts",
100
102
  sections: [
101
103
  {
102
104
  label: "usage",
103
105
  entries: [
104
- ["(bare)", `Scan your session history, then open http://localhost:${DASHBOARD_PORT}/audit`],
105
- ["--schedule [days]", "Scan on a timer and email the findings. Default 7 days, range 1-90. Signs you in the first time."],
106
+ ["(bare)", `Scan your session history for leaked keys, then open http://localhost:${DASHBOARD_PORT}/audit`],
107
+ ["--schedule [days]", "Scan on a timer, notify you on this machine, and email the findings. Default 7 days, range 1-90. Signs you in the first time."],
106
108
  // Its own row now, rather than a clause inside --schedule's. It
107
109
  // does not stand alone, which is why it used to be a clause — but
108
110
  // a clause wraps, and `--email <address>` landing with the flag at
109
111
  // the end of one line and its placeholder at the start of the next
110
112
  // is not a flag anybody can read or copy.
111
- ["--email <address>", "With --schedule, skips the sign-in prompt."],
113
+ ["--email <address>", "With --schedule, fills in your address. A sign-in code is still emailed to you to paste."],
112
114
  ["--no-schedule", "Stop the timer. Leaves you signed in."],
113
115
  // Its own pair of flags rather than a clause on --schedule: whether
114
116
  // this machine scans and whether it may interrupt you are separate
@@ -121,7 +123,11 @@ export function helpText(): string {
121
123
  ],
122
124
  },
123
125
  ],
124
- footer: ["Everything runs on this machine; only a scheduled digest ever leaves it."],
126
+ footer: [
127
+ "Everything runs on this machine. A scan tells you three ways: a desktop",
128
+ "notification, two lines in your agent session, and the emailed digest —",
129
+ "and only that digest ever leaves.",
130
+ ],
125
131
  },
126
132
  helpOptsFor(process.stdout),
127
133
  );
@@ -209,6 +215,18 @@ function startProgress(): Progress {
209
215
  const lines = Array.from({ length: n }, (_, i) => lineFor(i));
210
216
  // Move the cursor back up over the previously-drawn block, then clear and
211
217
  // rewrite each line in place.
218
+ //
219
+ // INVARIANT: nothing else may write to the terminal while this is running.
220
+ // The cursor-up is a fixed count, so any stray line pushes the cursor down
221
+ // and the next redraw repaints the block lower — stranding the top of the
222
+ // old frame above it, which reads as the audit having run twice. That is
223
+ // not hypothetical: Node's "SQLite is an experimental feature" warning did
224
+ // exactly this on every first run, two lines of it, and it is why
225
+ // `lib/sqlite-reader.ts` now filters that one warning at the source.
226
+ //
227
+ // There is no defensive fix from inside here — erasing to end of screen
228
+ // still leaves the stranded lines ABOVE the cursor. Keep the terminal
229
+ // quiet instead.
212
230
  if (printed) process.stdout.write(`\x1b[${n}A`);
213
231
  process.stdout.write(lines.map((l) => `\x1b[2K${l}`).join("\n") + "\n");
214
232
  printed = true;
@@ -390,66 +408,73 @@ async function announceLeaksOrThrow(result: AuditResult): Promise<void> {
390
408
  const claimed = markLeakNoticeDelivered(ids, undefined, "desktop");
391
409
  if (claimed.length === 0) return;
392
410
 
393
- const n = claimed.length === 1 ? "a credential" : `${claimed.length} credentials`;
394
- const title = "failproofai found a leaked credential";
395
- // Fixed text with a count interpolated the same rule as the in-CLI notice,
396
- // for the same reason. A finding's own text comes from a repository this
397
- // machine cloned, and notification bodies render markup on several Linux
398
- // desktops.
399
- //
400
- // The second sentence is the only place most users are ever offered the
401
- // digest. The scan is local and needs no account, so nothing else in the
402
- // product has a reason to ask for an address which is exactly why the
403
- // audit's findings have historically reached nobody. A banner someone is
404
- // already reading, about a key of their own, is the one moment the offer is
405
- // worth anything.
411
+ // Report what SURVIVED into the record, not how many ids the scan minted.
412
+ // `writeLeakRecord` prunes to MAX_FINDINGS, so on a large history the two
413
+ // numbers diverge wildly a real run announced "5703 credentials" while the
414
+ // dashboard it points at showed 500. A banner that disagrees with the page it
415
+ // sends you to is worse than no banner: it teaches you the number is noise.
416
+ const live = new Set(activeFindings(readLeakRecord()).map((f) => f.id));
417
+ const shown = claimed.filter((id) => live.has(id)).length || claimed.length;
418
+
419
+ const n = shown === 1 ? "a possible credential exposure" : `${shown} possible credential exposures`;
420
+ // Fixed text with no scanned content and no count. A finding's own text comes
421
+ // from a repository this machine cloned, and notification bodies render
422
+ // markup on several Linux desktops. The count is intentionally left to the
423
+ // report too: a banner must not turn detector candidates into a claim that
424
+ // the user leaked N real credentials.
406
425
  //
407
426
  // No action buttons, deliberately. `Notify` supports them, but a server
408
427
  // delivers the click back as an `ActionInvoked` signal to the sender — and
409
428
  // this process exits as soon as the scan finishes, so the button would be
410
- // dead. Two plain commands the user can copy beat one button that does
411
- // nothing.
412
- const body =
413
- `${n} in your agent transcripts. Run failproofai audit to see them — ` +
414
- "or failproofai audit --schedule to get them by email.";
415
-
416
- // The two platforms need opposite things, and the split is not cosmetic. On
417
- // Linux this process can reach the session bus itself. On macOS it cannot
418
- // reach Notification Center at all it is a child of a LaunchDaemon, outside
419
- // the GUI session so it hands the message to an agent that lives inside
420
- // one. See `macos-notifier.ts` for why that agent exists.
421
- if (process.platform === "darwin") {
422
- pruneMacNotifyQueue();
423
- if (!queueMacNotification(claimed[0], title, body)) {
424
- process.stderr.write(`failproofai: found ${n} but could not queue a notification\n`);
425
- } else if (!macNotifierInstalled()) {
426
- // Queued, but nothing will collect it: setup either never ran or could
427
- // not compile the applet. Worth a line, because the symptom otherwise is
428
- // simply silence.
429
- process.stderr.write(
430
- `failproofai: found ${n}; the notifier is not installed, so nothing will show it ` +
431
- `(run \`failproofai config\`)\n`,
432
- );
429
+ // dead. A plain command the user can copy beats a button that does nothing.
430
+ const { title, body } = desktopLeakNotice();
431
+ let delivered = false;
432
+
433
+ try {
434
+ // The two platforms need opposite things, and the split is not cosmetic. On
435
+ // Linux this process can reach the session bus itself. On macOS it cannot
436
+ // reach Notification Center at all it is a child of a LaunchDaemon, outside
437
+ // the GUI session so it hands the message to an agent that lives inside
438
+ // one. See `macos-notifier.ts` for why that agent exists.
439
+ if (process.platform === "darwin") {
440
+ pruneMacNotifyQueue();
441
+ if (!queueMacNotification(claimed[0], title, body)) {
442
+ process.stderr.write(`failproofai: found ${n} but could not queue a notification\n`);
443
+ } else {
444
+ delivered = true; // the durable queue owns delivery from here
445
+ }
446
+ if (delivered && !macNotifierInstalled()) {
447
+ // Queued, but nothing currently collects it. Keep the claim because the
448
+ // payload is durable and will be consumed when the LaunchAgent returns.
449
+ // Worth a line, because the immediate symptom is otherwise silence.
450
+ process.stderr.write(
451
+ `failproofai: found ${n}; the notifier is not installed, so nothing will show it ` +
452
+ `(run \`failproofai config\`)\n`,
453
+ );
454
+ }
455
+ return;
433
456
  }
434
- return;
435
- }
436
457
 
437
- const outcome = await notifyDesktop(
438
- title,
439
- body,
440
- // A stable id would let a repeat scan replace the previous bubble instead
441
- // of stacking. We do not have one yet — the server assigns it and we would
442
- // have to persist it — so a fresh notification is correct here, and it is
443
- // bounded: at most one per distinct credential, ever.
444
- 0,
445
- );
446
- if (!outcome.ok) {
447
- // stderr, so it lands in the journal beside the run it belongs to. Not a
448
- // failure of the audit — but "we found a key and could not tell you" is
449
- // exactly the sentence someone debugging this needs to find.
450
- process.stderr.write(
451
- `failproofai: found ${n} but could not raise a desktop notification (${outcome.reason})\n`,
458
+ const outcome = await notifyDesktop(
459
+ title,
460
+ body,
461
+ // A stable id would let a repeat scan replace the previous bubble instead
462
+ // of stacking. We do not have one yet — the server assigns it and we would
463
+ // have to persist it — so a fresh notification is correct here, and it is
464
+ // bounded: at most one per distinct credential, ever.
465
+ 0,
452
466
  );
467
+ delivered = outcome.ok;
468
+ if (!outcome.ok) {
469
+ // stderr, so it lands in the journal beside the run it belongs to. Not a
470
+ // failure of the audit — but "we found a key and could not tell you" is
471
+ // exactly the sentence someone debugging this needs to find.
472
+ process.stderr.write(
473
+ `failproofai: found ${n} but could not raise a desktop notification (${outcome.reason})\n`,
474
+ );
475
+ }
476
+ } finally {
477
+ if (!delivered) releaseLeakNoticeClaims(claimed, undefined, "desktop");
453
478
  }
454
479
  }
455
480
 
@@ -19,7 +19,7 @@ import { severityForBuiltin } from "./features";
19
19
  import { findSecrets, flattenToolInput } from "./leak-scan";
20
20
  import { fingerprintSecret, fingerprintId } from "./leak-fingerprint";
21
21
  import { readLeakIdentity, readLeakRecord, writeLeakRecord } from "./leak-store";
22
- import { upsertFinding, describeMechanism } from "./leak-record";
22
+ import { upsertFinding, describeMechanism, emptyRecord } from "./leak-record";
23
23
  import { shortenPaths } from "./redact-example";
24
24
  import { readCachedTranscript, writeCachedTranscriptResult } from "./cache";
25
25
  import { initReplay, replayEvent, restoreReplay } from "./replay";
@@ -37,6 +37,64 @@ import {
37
37
 
38
38
  const TRANSCRIPT_CONCURRENCY = 8;
39
39
 
40
+ /**
41
+ * Raw transcript bytes allowed in flight at once.
42
+ *
43
+ * The concurrency above counts FILES, which is the wrong unit. `streamEventsFrom`
44
+ * returns an array — despite the name it materialises every event of a
45
+ * transcript into JS objects — and a 57 MB JSONL becomes several hundred MB of
46
+ * them. On a real 1.15 GB corpus, 7 files hold 224 MB and the largest is
47
+ * 57.5 MB, so eight workers could put a quarter of the corpus in memory
48
+ * simultaneously. That machine printed `Aborted(OOM)` 22 times in one run.
49
+ *
50
+ * This is NOT what fixed those lines, and the distinction is worth keeping
51
+ * honest: they were measured at ~22 per run, appear within the first two
52
+ * seconds, are unchanged by this budget at 48 MB or at 16 MB, and do not fail
53
+ * the run — its counts come back correct. Their source is still unidentified.
54
+ * This bounds a real and separate failure mode (eight of the seven largest
55
+ * files in flight at once); it should not be read as having fixed the other.
56
+ *
57
+ * 48 MB of raw input is the budget because the expansion factor from JSONL to
58
+ * parsed objects is roughly 5-10x, which keeps the peak inside a default heap
59
+ * with room for the eight small-file workers this is meant not to slow down.
60
+ * Chosen to bound memory, not to be exactly right: the failure it prevents is a
61
+ * crash, and the cost of being conservative is that one very large file scans
62
+ * alone for a moment.
63
+ */
64
+ const TRANSCRIPT_BYTE_BUDGET = 48 * 1024 * 1024;
65
+
66
+ /**
67
+ * Admission control by weight, so a worker waits for MEMORY as well as a slot.
68
+ *
69
+ * A file bigger than the whole budget is admitted alone rather than refused —
70
+ * otherwise the largest transcript on the machine could never be scanned, which
71
+ * is precisely the one most likely to hold something.
72
+ */
73
+ class ByteGate {
74
+ private inFlight = 0;
75
+ private waiting: Array<() => void> = [];
76
+
77
+ constructor(private readonly budget: number) {}
78
+
79
+ async acquire(bytes: number): Promise<void> {
80
+ const want = Math.max(0, bytes);
81
+ while (this.inFlight > 0 && this.inFlight + want > this.budget) {
82
+ await new Promise<void>((resolve) => this.waiting.push(resolve));
83
+ }
84
+ this.inFlight += want;
85
+ }
86
+
87
+ release(bytes: number): void {
88
+ this.inFlight -= Math.max(0, bytes);
89
+ if (this.inFlight < 0) this.inFlight = 0;
90
+ // Wake everyone and let each re-test its own weight: a small task behind a
91
+ // large one should not be held by a queue position it does not need.
92
+ const waiters = this.waiting;
93
+ this.waiting = [];
94
+ for (const wake of waiters) wake();
95
+ }
96
+ }
97
+
40
98
  /** Canonicalize a policy name to its short, qualified form for display
41
99
  * (`failproofai/foo` → `foo`). */
42
100
  function shortPolicyName(name: string): string {
@@ -118,6 +176,50 @@ interface ScanOutcome {
118
176
  resumed: boolean;
119
177
  }
120
178
 
179
+ const CREDENTIAL_RESEARCH_MARKERS: ReadonlyArray<RegExp> = [
180
+ /\b(?:secret|credential)[-_ ](?:scanner|detection|detector|pattern|regex|corpus)\b/i,
181
+ /\b(?:trufflehog|trufflesecurity|gitleaks|detect-secrets|secret-detection-rules)\b/i,
182
+ /\b(?:SECRET_PATTERNS|findSecrets|isDocsLiteral|leak-scan)\b/,
183
+ /\b(?:fixture|placeholder|synthetic|sample) (?:api )?(?:key|token|secret|credential)s?\b/i,
184
+ /\b(?:grep|rg|ripgrep|scan|hunt|search)\b[^\n]{0,240}\b(?:secret|credential|token)s?\b/i,
185
+ ];
186
+
187
+ const CREDENTIAL_RESEARCH_PURPOSE =
188
+ /\b(?:hunt|scan|search|research|test)(?:ing)?\b[^\n]{0,80}\b(?:secret|credential|token)s?\b|\b(?:secret|credential|token)s?\b[^\n]{0,80}\b(?:scanner|detector|detection|research|fixtures?|corpus)\b/i;
189
+
190
+ /**
191
+ * Deliberate secret-detector research is full of realistic fake credentials.
192
+ * Reporting those as leaks is the feedback loop that made one measured session
193
+ * contribute 455 of the 500 rows in its own audit.
194
+ *
195
+ * Require independent signals across tool INPUTS. Results are excluded from
196
+ * classification: a normal command can print hostile text, while commands and
197
+ * URLs state what the agent intentionally set out to inspect. Two signals keep
198
+ * an ordinary discussion that happens to say "secret scanner" from silencing a
199
+ * session; dedicated detector/test work reliably carries several.
200
+ */
201
+ export function isCredentialResearchSession(
202
+ events: NormalizedToolEvent[],
203
+ sessionDescription?: string,
204
+ ): boolean {
205
+ if (sessionDescription && CREDENTIAL_RESEARCH_PURPOSE.test(sessionDescription)) return true;
206
+ const matched = new Set<number>();
207
+ let evidenceEvents = 0;
208
+ for (const event of events) {
209
+ const intent = flattenToolInput(event.toolInput);
210
+ let eventMatched = false;
211
+ for (let i = 0; i < CREDENTIAL_RESEARCH_MARKERS.length; i++) {
212
+ if (CREDENTIAL_RESEARCH_MARKERS[i].test(intent)) {
213
+ matched.add(i);
214
+ eventMatched = true;
215
+ }
216
+ }
217
+ if (eventMatched) evidenceEvents++;
218
+ if (matched.size >= 2 || evidenceEvents >= 3) return true;
219
+ }
220
+ return false;
221
+ }
222
+
121
223
  async function scanOneTranscript(
122
224
  meta: TranscriptMetadata,
123
225
  resume?: { fromByte: number; detectorState: DetectorSessionState },
@@ -186,6 +288,8 @@ async function scanOneTranscript(
186
288
  // Capture the session's cwd from the first event that carried one — every
187
289
  // event in a single transcript shares the same cwd by construction.
188
290
  result.cwd = result.cwd || events[0].cwd || "";
291
+ const suppressLeakScan = isCredentialResearchSession(events, meta.sessionDescription);
292
+ if (suppressLeakScan) result.leakScanSuppressed = "credential-research";
189
293
 
190
294
  for (const event of events) {
191
295
  // The 8 behavioural detectors are switched off with the rest of the old
@@ -226,7 +330,7 @@ async function scanOneTranscript(
226
330
  // decision per policy, never the text that matched. Scanned separately so
227
331
  // the leak report can name WHICH key, and so detection can be tuned for
228
332
  // precision without dragging the redactor's recall down with it.
229
- recordLeaks(result, event);
333
+ if (!suppressLeakScan) recordLeaks(result, event);
230
334
  }
231
335
 
232
336
  return { result, bytesScanned, detectorState: sessionState, resumed };
@@ -300,10 +404,19 @@ function leakSalt(): string | null {
300
404
  * human about. Never throws: a scan that finds leaks it cannot write down has
301
405
  * still found them, and the caller reports on the in-memory result either way.
302
406
  */
303
- function persistLeaks(perTranscript: TranscriptAuditResult[]): string[] {
407
+ export function persistLeaks(perTranscript: TranscriptAuditResult[], replaceExisting: boolean): string[] {
304
408
  const fresh: string[] = [];
305
409
  try {
306
- const record = readLeakRecord();
410
+ const previous = readLeakRecord();
411
+ const previousIds = new Set(previous.findings.map((f) => f.id));
412
+ // A successful all-history scan is authoritative. Rebuilding is what lets
413
+ // detector precision fixes remove findings they now reject; merging made
414
+ // every false positive live for 90 days even after the code was fixed.
415
+ // Scoped or incomplete scans still merge because absence outside their
416
+ // coverage says nothing.
417
+ const record = replaceExisting
418
+ ? emptyRecord(previous.salt, new Date().toISOString())
419
+ : previous;
307
420
  for (const t of perTranscript) {
308
421
  for (const leak of t.leaks ?? []) {
309
422
  const { isNew } = upsertFinding(record, {
@@ -322,7 +435,10 @@ function persistLeaks(perTranscript: TranscriptAuditResult[]): string[] {
322
435
  mechanism: describeMechanism(leak.toolName, leak.direction, leak.path),
323
436
  },
324
437
  });
325
- if (isNew) fresh.push(leak.id);
438
+ // During an authoritative rebuild every insertion is new to the empty
439
+ // record, but only ids absent from the previous record are new to the
440
+ // HUMAN and eligible for a notification.
441
+ if (isNew && !previousIds.has(leak.id)) fresh.push(leak.id);
326
442
  }
327
443
  }
328
444
  writeLeakRecord(record);
@@ -351,7 +467,7 @@ function formatPolicyExample(_policyName: string, event: NormalizedToolEvent): s
351
467
  * came from the FIRST event in the file and the tail's came from the first
352
468
  * event after the offset, so the older one wins.
353
469
  */
354
- function mergeIncremental(
470
+ export function mergeIncremental(
355
471
  cached: TranscriptAuditResult,
356
472
  tail: TranscriptAuditResult,
357
473
  ): TranscriptAuditResult {
@@ -365,6 +481,12 @@ function mergeIncremental(
365
481
  examplesByName: {},
366
482
  rangeByName: { ...cached.rangeByName },
367
483
  };
484
+ if (cached.leakScanSuppressed || tail.leakScanSuppressed) {
485
+ out.leakScanSuppressed = "credential-research";
486
+ out.leaks = [];
487
+ } else {
488
+ out.leaks = [...(cached.leaks ?? []), ...(tail.leaks ?? [])];
489
+ }
368
490
  for (const [name, list] of Object.entries(cached.examplesByName)) {
369
491
  out.examplesByName[name] = [...list];
370
492
  }
@@ -547,18 +669,22 @@ async function runAuditInner(opts: RunAuditOptions, startedAt: number): Promise<
547
669
 
548
670
  // 1. Discover transcripts across all selected CLIs.
549
671
  const allTranscripts: TranscriptMetadata[] = [];
672
+ let discoveryErrors = 0;
550
673
  for (const cli of clis) {
551
674
  const adapter = ADAPTERS[cli];
552
675
  let list: TranscriptMetadata[];
553
676
  try {
554
677
  list = await adapter.listTranscripts({ projects: opts.projects, sinceMs });
555
678
  } catch {
679
+ discoveryErrors++;
556
680
  continue; // adapter failures shouldn't kill the whole audit
557
681
  }
558
682
  allTranscripts.push(...list);
559
683
  }
560
684
 
561
- // 2. Scan each transcript (cache-aware), 8 in parallel.
685
+ // 2. Scan each transcript (cache-aware), 8 at a time and within a memory
686
+ // budget — see TRANSCRIPT_BYTE_BUDGET for why a file count is the wrong unit.
687
+ const gate = new ByteGate(TRANSCRIPT_BYTE_BUDGET);
562
688
  let skipped = 0;
563
689
  let errors = 0;
564
690
  const tasks = allTranscripts.map((meta) => async (): Promise<TranscriptAuditResult> => {
@@ -572,6 +698,9 @@ async function runAuditInner(opts: RunAuditOptions, startedAt: number): Promise<
572
698
  : undefined;
573
699
  cachedPrefix = found?.kind === "resume" ? found.result : null;
574
700
  }
701
+ // Memory admission. Taken AFTER the cache check above, so a cache hit — the
702
+ // common case on a warm machine — never waits for a slot it does not use.
703
+ await gate.acquire(meta.sizeBytes);
575
704
  try {
576
705
  const scan = await scanOneTranscript(meta, resume);
577
706
  // A resumed scan produced hits for the TAIL only; the cached result holds
@@ -615,6 +744,11 @@ async function runAuditInner(opts: RunAuditOptions, startedAt: number): Promise<
615
744
  examplesByName: {},
616
745
  rangeByName: {},
617
746
  };
747
+ } finally {
748
+ // Released on every path, including the error one — a task that threw
749
+ // still freed its memory, and holding its weight would shrink the budget
750
+ // permanently over a long scan until nothing could be admitted at all.
751
+ gate.release(meta.sizeBytes);
618
752
  }
619
753
  });
620
754
 
@@ -652,7 +786,14 @@ async function runAuditInner(opts: RunAuditOptions, startedAt: number): Promise<
652
786
  // not alert again however many fresh sightings it accumulates, and a
653
787
  // credential seen for the first time must alert even though its rule has
654
788
  // fired a thousand times before.
655
- const newFindingIds = persistLeaks(perTranscript);
789
+ const coversAllHistory =
790
+ opts.clis === undefined &&
791
+ opts.projects === undefined &&
792
+ opts.since === undefined &&
793
+ discoveryErrors === 0 &&
794
+ errors === 0 &&
795
+ skipped === 0;
796
+ const newFindingIds = persistLeaks(perTranscript, coversAllHistory);
656
797
 
657
798
  const auditResult: AuditResult = {
658
799
  version: 2,
@@ -36,7 +36,7 @@ import { existsSync, mkdirSync, readdirSync, statSync, writeFileSync, rmSync } f
36
36
  import { resolve } from "node:path";
37
37
 
38
38
  import { auditDir } from "../hooks/fp-home";
39
- import { readLeakRecord, activeFindings } from "./leak-store";
39
+ import { readLeakRecord, activeFindings, readLeakIdentity } from "./leak-store";
40
40
  import { isFindingId } from "./leak-fingerprint";
41
41
 
42
42
  /**
@@ -135,6 +135,25 @@ export function markLeakNoticeDelivered(
135
135
  return won;
136
136
  }
137
137
 
138
+ /** Release claims this process won when delivery failed, so the next scheduled
139
+ * run retries instead of turning a transient missing desktop into permanent
140
+ * silence. Only validated ids in the requested channel are touched. */
141
+ export function releaseLeakNoticeClaims(
142
+ ids: string[],
143
+ home?: string,
144
+ channel: NoticeChannel = "cli",
145
+ ): void {
146
+ try {
147
+ const dir = NOTICE_DIR(home, channel);
148
+ for (const id of ids) {
149
+ if (!isFindingId(id)) continue;
150
+ rmSync(resolve(dir, id), { force: true });
151
+ }
152
+ } catch {
153
+ // Retry bookkeeping must never turn a completed audit into a failure.
154
+ }
155
+ }
156
+
138
157
  /** Drop markers for findings that no longer exist, and very old ones. */
139
158
  export function pruneNoticeMarkers(home?: string, nowMs = Date.now()): void {
140
159
  try {
@@ -159,3 +178,82 @@ export function pruneNoticeMarkers(home?: string, nowMs = Date.now()): void {
159
178
  // Housekeeping only.
160
179
  }
161
180
  }
181
+
182
+ /**
183
+ * Should this session's agent tell the user about unviewed credentials?
184
+ *
185
+ * ## Why this is not the per-finding claim used for the desktop banner
186
+ *
187
+ * The per-finding marker answers "have we emitted this once?" — and emitting is
188
+ * not the same as arriving. Twice in one day it recorded every finding as
189
+ * delivered while the user saw nothing: once because the notice was attached to
190
+ * an event whose channel the host ignores (`SessionStart` takes
191
+ * `additionalContext`, not `systemMessage`), and once because hooks were
192
+ * disabled in the project under test. Neither is detectable from in here, and
193
+ * both were permanent — a claimed finding is never retried.
194
+ *
195
+ * So the in-CLI notice keys on the USER's action instead of on ours. It shows
196
+ * while findings exist that are newer than the last time the report was
197
+ * actually opened, and it stops when they open it. A dropped notice costs one
198
+ * session's silence rather than the alert; the only thing that silences it for
199
+ * good is the outcome we wanted anyway.
200
+ *
201
+ * Bounded by a per-SESSION marker so a long session gets one line, not one per
202
+ * turn. Sessions are cheap and few; findings are many and long-lived.
203
+ */
204
+ export function sessionNoticePending(sessionId: string, home?: string): PendingNotice {
205
+ try {
206
+ if (!sessionId || !/^[\w.:-]{1,200}$/.test(sessionId)) return { ids: [], count: 0 };
207
+ const dir = resolve(auditDir(home), "notified-sessions");
208
+ if (existsSync(resolve(dir, sessionId))) return { ids: [], count: 0 };
209
+
210
+ const findings = activeFindings(readLeakRecord(home), home);
211
+ if (findings.length === 0) return { ids: [], count: 0 };
212
+
213
+ // Only what the user has not already looked at. `firstSeen`, not
214
+ // `lastSeen`: re-encountering a credential the user has already reviewed is
215
+ // not news, however recently the scan tripped over it again.
216
+ const viewedAt = readLeakIdentity(home).reportViewedAt ?? 0;
217
+ const unviewed = findings.filter((f) => {
218
+ const seen = Date.parse(f.firstSeen);
219
+ return !Number.isFinite(seen) || seen > viewedAt;
220
+ });
221
+ return { ids: unviewed.map((f) => f.id), count: unviewed.length };
222
+ } catch {
223
+ return { ids: [], count: 0 };
224
+ }
225
+ }
226
+
227
+ /**
228
+ * Claim this session, so the notice appears once rather than every turn.
229
+ *
230
+ * Returns whether this caller won. O_EXCL again, for the same reason as the
231
+ * per-finding markers: several agents run in one project at once.
232
+ */
233
+ export function markSessionNoticed(sessionId: string, home?: string): boolean {
234
+ try {
235
+ if (!sessionId || !/^[\w.:-]{1,200}$/.test(sessionId)) return false;
236
+ const dir = resolve(auditDir(home), "notified-sessions");
237
+ mkdirSync(dir, { recursive: true });
238
+ writeFileSync(resolve(dir, sessionId), "", { flag: "wx", mode: 0o600 });
239
+ return true;
240
+ } catch {
241
+ return false;
242
+ }
243
+ }
244
+
245
+ /** Session markers outlive nothing useful — a session id never recurs. */
246
+ export function pruneSessionMarkers(home?: string, nowMs = Date.now()): void {
247
+ try {
248
+ const dir = resolve(auditDir(home), "notified-sessions");
249
+ if (!existsSync(dir)) return;
250
+ for (const name of readdirSync(dir)) {
251
+ const path = resolve(dir, name);
252
+ try {
253
+ if (nowMs - statSync(path).mtimeMs > 30 * 86_400_000) rmSync(path, { force: true });
254
+ } catch { /* raced */ }
255
+ }
256
+ } catch {
257
+ // Housekeeping only.
258
+ }
259
+ }