residoo 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +334 -46
  2. package/SECURITY.md +29 -22
  3. package/package.json +1 -1
  4. package/src/cli.js +249 -17
  5. package/src/integrity.js +689 -0
  6. package/src/patterns.js +78 -5
  7. package/src/report.js +188 -8
  8. package/src/rotation.js +834 -0
  9. package/src/sources/agent-configs.js +308 -0
  10. package/src/sources/aider.js +361 -0
  11. package/src/sources/amazon-q.js +199 -0
  12. package/src/sources/antigravity-cli.js +155 -0
  13. package/src/sources/cline.js +208 -0
  14. package/src/sources/codebuff.js +295 -0
  15. package/src/sources/codex-cli.js +258 -0
  16. package/src/sources/cody.js +325 -0
  17. package/src/sources/continue.js +408 -0
  18. package/src/sources/copilot-chat.js +272 -0
  19. package/src/sources/copilot-cli.js +300 -0
  20. package/src/sources/crush.js +364 -0
  21. package/src/sources/cursor.js +374 -0
  22. package/src/sources/devin-cli.js +241 -0
  23. package/src/sources/factory-droid.js +153 -0
  24. package/src/sources/fx.js +136 -0
  25. package/src/sources/gemini-cli.js +242 -0
  26. package/src/sources/goose.js +366 -0
  27. package/src/sources/grok-cli.js +267 -0
  28. package/src/sources/hermes.js +282 -0
  29. package/src/sources/index.js +172 -8
  30. package/src/sources/jetbrains-ai-assistant.js +343 -0
  31. package/src/sources/jetbrains-junie.js +292 -0
  32. package/src/sources/kilo-code.js +430 -0
  33. package/src/sources/kimi-code.js +147 -0
  34. package/src/sources/kiro-cli.js +393 -0
  35. package/src/sources/kiro-ide.js +230 -0
  36. package/src/sources/llm.js +328 -0
  37. package/src/sources/mentat.js +143 -0
  38. package/src/sources/open-interpreter.js +224 -0
  39. package/src/sources/openclaw.js +218 -0
  40. package/src/sources/opencode.js +379 -0
  41. package/src/sources/openhands.js +181 -0
  42. package/src/sources/pearai.js +151 -0
  43. package/src/sources/pi-agent.js +130 -0
  44. package/src/sources/project-artifacts.js +355 -0
  45. package/src/sources/qodo-gen.js +189 -0
  46. package/src/sources/qwen-code.js +244 -0
  47. package/src/sources/roo-code.js +239 -0
  48. package/src/sources/trae.js +294 -0
  49. package/src/sources/void.js +273 -0
  50. package/src/sources/warp.js +395 -0
  51. package/src/sources/windsurf.js +256 -0
  52. package/src/sources/zed.js +374 -0
@@ -0,0 +1,151 @@
1
+ "use strict";
2
+
3
+ const fs = require("fs");
4
+ const { createInterface } = require("readline/promises");
5
+ const path = require("path");
6
+ const os = require("os");
7
+
8
+ /**
9
+ * PearAI's local chat session history.
10
+ *
11
+ * VERIFICATION STATUS (read this before trusting anything below): PearAI is
12
+ * an open-source VS Code fork (github.com/trypear/pearai-app) whose AI chat
13
+ * is powered by "pearai-submodule" — itself an open-source fork of
14
+ * Continue (github.com/continuedev/continue), bundled into the app rather
15
+ * than installed as a marketplace extension. This matters because it means
16
+ * PearAI does NOT use Cursor/VS Code's per-profile `state.vscdb` SQLite
17
+ * approach for chat content — it inherited Continue's own file-based session
18
+ * store instead. This was confirmed directly from two independent sources:
19
+ *
20
+ * 1. PearAI's own shipped source, `pearai-submodule/core/util/paths.ts`
21
+ * (fetched from trypear/pearai-submodule@main), which defines:
22
+ * const CONTINUE_GLOBAL_DIR =
23
+ * process.env.CONTINUE_GLOBAL_DIR ?? path.join(os.homedir(), ".pearai");
24
+ * and derives the sessions folder as `<CONTINUE_GLOBAL_DIR>/sessions`,
25
+ * individual session files as `<sessionId>.json`, and an index file at
26
+ * `sessions.json`. Note this path has NO per-OS branching — it is
27
+ * `os.homedir()/.pearai` on every platform, unlike VS Code-derived
28
+ * products (Cursor, Trae, Void) whose Application Support-style path
29
+ * differs per OS. (PearAI still ships a VS Code-derived Application
30
+ * Support/state.vscdb tree too, for ordinary editor/window state, but
31
+ * that is generic VS Code chrome, not where chat content lives — kept
32
+ * out of scope here the same way claude-code.js and cursor.js each stay
33
+ * scoped to where the actual transcript content lives, not every file
34
+ * the host editor happens to write.)
35
+ * 2. `claude-code-history-viewer` (github.com/jhlee0409/claude-code-history-viewer),
36
+ * an actively maintained, independently authored desktop app that reads
37
+ * this exact same layout — its `src-tauri/src/providers/pearai.rs`
38
+ * module doc reads (fetched verbatim): "PearAI is a fork of Continue
39
+ * that rebrands the global directory from ~/.continue to ~/.pearai. The
40
+ * session store format is identical (<sessionId>.json + sessions.json
41
+ * index)."
42
+ *
43
+ * Both sources agree exactly on the directory, the per-file naming, and the
44
+ * index file. What this source has NOT been checked against is a real
45
+ * PearAI install — PearAI is not installed on the machine this was built on
46
+ * (checked: not in /Applications, not in ~/Library/Application Support, no
47
+ * mdfind hits). If you have PearAI installed and have used its chat at
48
+ * least once, the most useful thing you can do is run `residoo scan` and
49
+ * confirm `sourcesScanned`/`filesScanned` for "pearai" look right against
50
+ * what you can see under ~/.pearai/sessions, then report back either way.
51
+ *
52
+ * Session files are plain JSON text on disk (not SQLite), so — like
53
+ * claude-code.js's JSONL files — they can be streamed and pattern-matched
54
+ * line by line with no parsing required: a pretty-printed session file
55
+ * naturally splits into one scannable line per field, and even a minified
56
+ * one degrades gracefully into a single long line, still fully scanned.
57
+ * `sessions.json` (the index) is included too, on the same "never
58
+ * cherry-pick which files might matter" principle the other sources follow.
59
+ */
60
+ const PEARAI_DIR = path.join(os.homedir(), ".pearai");
61
+ const SESSIONS_DIR = path.join(PEARAI_DIR, "sessions");
62
+
63
+ // Same bounds and same rationale as claude-code.js — no PearAI-specific
64
+ // large-file data point exists (no real install to measure against), so
65
+ // these are carried over unchanged as a generous, conservative backstop.
66
+ const MAX_BYTES = 2 * 1024 * 1024 * 1024; // 2GB
67
+ const READ_TIMEOUT_MS = 60_000;
68
+
69
+ function id() { return "pearai"; }
70
+ function label() { return "PearAI"; }
71
+
72
+ function available() {
73
+ try { return fs.statSync(SESSIONS_DIR).isDirectory(); } catch { return false; }
74
+ }
75
+
76
+ /**
77
+ * Same defensive symlink-following helper as claude-code.js — see that
78
+ * file's docstring for the full reasoning. Duplicated rather than shared,
79
+ * matching this project's "each source is a small, self-contained file"
80
+ * convention (stated explicitly in cursor.js).
81
+ */
82
+ function isFileFollowingSymlink(fullPath, dirent) {
83
+ if (dirent.isFile()) return true;
84
+ if (!dirent.isSymbolicLink()) return false;
85
+ try { return fs.statSync(fullPath).isFile(); } catch { return false; }
86
+ }
87
+
88
+ /**
89
+ * Yield { file, mtimeMs, sizeBytes, broken } for every session file found.
90
+ *
91
+ * Unlike claude-code.js's two-level walk (project dir -> transcripts), the
92
+ * confirmed layout here is flat: every `*.json` file directly inside
93
+ * `~/.pearai/sessions` — individual `<sessionId>.json` files plus the
94
+ * `sessions.json` index — so this is a single readdir, not a nested one.
95
+ */
96
+ function* files() {
97
+ let entries;
98
+ try { entries = fs.readdirSync(SESSIONS_DIR, { withFileTypes: true }); }
99
+ catch { return; }
100
+
101
+ for (const e of entries) {
102
+ if (!e.name.endsWith(".json")) continue;
103
+ const file = path.join(SESSIONS_DIR, e.name);
104
+ if (!e.isFile()) {
105
+ const resolved = isFileFollowingSymlink(file, e);
106
+ if (!resolved) {
107
+ if (e.isSymbolicLink()) yield { file, broken: true };
108
+ continue;
109
+ }
110
+ }
111
+ let stat;
112
+ try { stat = fs.statSync(file); } catch { yield { file, broken: true }; continue; }
113
+ yield { file, mtimeMs: stat.mtimeMs, sizeBytes: stat.size, broken: false };
114
+ }
115
+ }
116
+
117
+ /**
118
+ * Read one session file as an array of raw text lines. Same streamed,
119
+ * bounded, honest-partial-status approach as claude-code.js's readLines() —
120
+ * see that file's docstring for the full reasoning, which applies unchanged
121
+ * here since this is likewise a plain text file on disk.
122
+ */
123
+ async function readLines(file) {
124
+ let stat;
125
+ try { stat = fs.statSync(file); }
126
+ catch { return { lines: [], status: "failed", bytesRead: 0 }; }
127
+ if (stat.size > MAX_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
128
+
129
+ const lines = [];
130
+ let bytesRead = 0;
131
+ const stream = fs.createReadStream(file, { encoding: "utf-8" });
132
+ const rl = createInterface({ input: stream, crlfDelay: Infinity });
133
+
134
+ const timer = setTimeout(() => stream.destroy(new Error("read timed out")), READ_TIMEOUT_MS);
135
+
136
+ try {
137
+ for await (const line of rl) {
138
+ lines.push(line);
139
+ bytesRead += Buffer.byteLength(line, "utf-8") + 1;
140
+ }
141
+ return { lines, status: "complete", bytesRead };
142
+ } catch {
143
+ return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
144
+ } finally {
145
+ clearTimeout(timer);
146
+ rl.close();
147
+ stream.destroy();
148
+ }
149
+ }
150
+
151
+ module.exports = { id, label, available, files, readLines };
@@ -0,0 +1,130 @@
1
+ "use strict";
2
+
3
+ const fs = require("fs");
4
+ const { createInterface } = require("readline/promises");
5
+ const path = require("path");
6
+ const os = require("os");
7
+
8
+ /**
9
+ * "Pi" (earendil-works/pi on GitHub; installed as `@mariozechner/pi-coding-agent`
10
+ * from npm, run as the `pi` CLI) local session transcripts.
11
+ *
12
+ * VERIFICATION STATUS: corroborated directly from the project's own shipped
13
+ * documentation (fetched from the live repo during this source's research),
14
+ * but NOT checked against a real install on the machine this source was
15
+ * built on (no `~/.pi` directory exists there; see CONTRIBUTING.md).
16
+ *
17
+ * `packages/coding-agent/docs/sessions.md` in the `earendil-works/pi` repo —
18
+ * the project's own docs, not a third party's description of it — states
19
+ * plainly: "Sessions auto-save to `~/.pi/agent/sessions/`, organized by
20
+ * working directory. Each session is a JSONL file with a tree structure,"
21
+ * further describing entries with `id`/`parentId` fields (branching), model
22
+ * changes, thinking-level changes, labels, compactions, and branch summaries
23
+ * all living in the same JSONL stream. "Organized by working directory"
24
+ * means sessions live under per-project subdirectories rather than flat in
25
+ * `sessions/` itself — the exact subdirectory naming isn't spelled out in
26
+ * that doc, so this source walks recursively for `*.jsonl` rather than
27
+ * assuming a fixed depth, the same tolerance claude-code.js applies to
28
+ * project-slug directory names it doesn't try to decode either.
29
+ *
30
+ * Independently, jazzyalex/agent-sessions (github.com/jazzyalex/agent-sessions,
31
+ * 800+ stars, a real macOS app built specifically to parse local
32
+ * AI-coding-agent session history) lists Pi among the CLI agents whose local
33
+ * history it reads, corroborating that this is real, currently-scanned-by-
34
+ * someone-else session data rather than a doc describing an unshipped plan.
35
+ */
36
+ const HOME = os.homedir();
37
+ const ROOT = path.join(HOME, ".pi", "agent", "sessions");
38
+
39
+ const MAX_BYTES = 2 * 1024 * 1024 * 1024; // 2GB — same backstop as claude-code.js.
40
+ const READ_TIMEOUT_MS = 60_000;
41
+ const MAX_WALK_DEPTH = 8;
42
+
43
+ function id() { return "pi-agent"; }
44
+ function label() { return "Pi"; }
45
+
46
+ function available() {
47
+ try { return fs.statSync(ROOT).isDirectory(); } catch { return false; }
48
+ }
49
+
50
+ /**
51
+ * Same defensive symlink-following helpers as claude-code.js — see that
52
+ * file's docstring. Duplicated rather than imported, per this project's
53
+ * self-contained-source-file convention (see cursor.js's docstring).
54
+ */
55
+ function isKindFollowingSymlink(fullPath, dirent, checkFn) {
56
+ if (checkFn(dirent)) return true;
57
+ if (!dirent.isSymbolicLink()) return false;
58
+ try { return checkFn(fs.statSync(fullPath)); } catch { return false; }
59
+ }
60
+ const isDirFollowingSymlink = (p, d) => isKindFollowingSymlink(p, d, (x) => x.isDirectory());
61
+ const isFileFollowingSymlink = (p, d) => isKindFollowingSymlink(p, d, (x) => x.isFile());
62
+
63
+ /**
64
+ * Recursively yield { file, mtimeMs, sizeBytes, broken } for every plain file
65
+ * under `dir` whose name passes `matchFn`, following symlinks and reporting
66
+ * one that resolves to neither a file nor a directory as `broken: true` — see
67
+ * factory-droid.js's walk() for the identical reasoning (this project
68
+ * duplicates this small helper per source file rather than sharing it; see
69
+ * cursor.js's docstring on why).
70
+ */
71
+ function* walk(dir, depth, matchFn) {
72
+ if (depth > MAX_WALK_DEPTH) return;
73
+ let entries;
74
+ try { entries = fs.readdirSync(dir, { withFileTypes: true }); }
75
+ catch { return; }
76
+
77
+ for (const e of entries) {
78
+ const full = path.join(dir, e.name);
79
+ if (isDirFollowingSymlink(full, e)) {
80
+ yield* walk(full, depth + 1, matchFn);
81
+ continue;
82
+ }
83
+ const isFile = isFileFollowingSymlink(full, e);
84
+ if (!isFile) {
85
+ if (e.isSymbolicLink()) yield { file: full, broken: true };
86
+ continue;
87
+ }
88
+ if (!matchFn(e.name)) continue;
89
+ let stat;
90
+ try { stat = fs.statSync(full); } catch { yield { file: full, broken: true }; continue; }
91
+ yield { file: full, mtimeMs: stat.mtimeMs, sizeBytes: stat.size, broken: false };
92
+ }
93
+ }
94
+
95
+ function* files() {
96
+ yield* walk(ROOT, 0, (name) => name.endsWith(".jsonl"));
97
+ }
98
+
99
+ /**
100
+ * Read one JSONL session as raw text lines. Identical streaming/timeout/
101
+ * partial-read discipline to claude-code.js's readLines().
102
+ */
103
+ async function readLines(file) {
104
+ let stat;
105
+ try { stat = fs.statSync(file); }
106
+ catch { return { lines: [], status: "failed", bytesRead: 0 }; }
107
+ if (stat.size > MAX_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
108
+
109
+ const lines = [];
110
+ let bytesRead = 0;
111
+ const stream = fs.createReadStream(file, { encoding: "utf-8" });
112
+ const rl = createInterface({ input: stream, crlfDelay: Infinity });
113
+ const timer = setTimeout(() => stream.destroy(new Error("read timed out")), READ_TIMEOUT_MS);
114
+
115
+ try {
116
+ for await (const line of rl) {
117
+ lines.push(line);
118
+ bytesRead += Buffer.byteLength(line, "utf-8") + 1;
119
+ }
120
+ return { lines, status: "complete", bytesRead };
121
+ } catch {
122
+ return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
123
+ } finally {
124
+ clearTimeout(timer);
125
+ rl.close();
126
+ stream.destroy();
127
+ }
128
+ }
129
+
130
+ module.exports = { id, label, available, files, readLines };
@@ -0,0 +1,355 @@
1
+ "use strict";
2
+
3
+ const fs = require("fs");
4
+ const path = require("path");
5
+ const { createInterface } = require("readline/promises");
6
+
7
+ /**
8
+ * Committed agent artifacts inside a PROJECT directory (a repo checkout).
9
+ *
10
+ * Every other source scans the machine's home-level stores: what the agent
11
+ * wrote for itself. This one scans what a repo is about to ship: transcripts,
12
+ * agent configs, and .env files sitting inside a checkout, where `git add .`
13
+ * and `npm publish` will carry them to everyone. The evidence that this is
14
+ * where the bodies are buried is the strongest in the whole research base:
15
+ * GitGuardian measured Claude Code-assisted commits leaking at 3.2% vs the
16
+ * 1.5% GitHub baseline; Lakera found live credentials inside
17
+ * `.claude/settings.local.json` files shipped in ~30 published npm packages
18
+ * precisely because no packaging tool ignores `.claude/` by default; and the
19
+ * Miasma campaign infected on repo OPEN via planted `.claude/settings.json`,
20
+ * `.gemini/settings.json`, `.cursor/rules/setup.mdc`, and `.vscode/tasks.json`
21
+ * (see the research digest, 2026-09-02, and integrity.js's campaign headers).
22
+ *
23
+ * OPT-IN BY CONSTRUCTION, never part of the default scan. The registry in
24
+ * index.js holds one singleton per source and filters with available(); a
25
+ * project scan needs a PARAMETER (which directory), and a singleton cannot
26
+ * carry one honestly. Two designs were considered:
27
+ *
28
+ * - setRoot() mutating this module's singleton: rejected. The registry
29
+ * object is shared process-wide, so one scan's --project argument would
30
+ * leak into any later scan in the same process, and "available() is
31
+ * false unless configured" would silently stop being true in a way no
32
+ * local reading of this file could reveal.
33
+ * - withRoot(root) factory: chosen. The module's default export still
34
+ * satisfies the full { id, label, available, files, readLines } contract
35
+ * (so registering it in index.js is harmless: available() is always
36
+ * false and files() yields nothing), while the CLI's --project handling
37
+ * constructs a configured instance and passes it straight into
38
+ * scan({ sources: [...] }). No shared mutable state, no registry change.
39
+ *
40
+ * WHAT IS SCANNED, with the verification trail per CONTRIBUTING.md's
41
+ * no-guessed-paths rule ("real install" means this project's own build
42
+ * machine, checked read-only):
43
+ *
44
+ * (a) Committed agent transcripts:
45
+ * - `*.jsonl` under any `.claude/` path component. Claude Code's own
46
+ * transcript layout is `<root>/projects/<slug>/<session>.jsonl` (real
47
+ * install, and claude-code.js's territory at home level); a copy of any
48
+ * part of that tree committed into a repo keeps the `.claude` component.
49
+ * - `*.jsonl` inside a directory whose name starts with "-": the
50
+ * project-slug shape Claude Code uses (the absolute project path with
51
+ * separators replaced by "-", e.g. `-Users-.../<uuid>.jsonl`; verified
52
+ * against the real install's ~/.claude/projects). A slug directory
53
+ * copied into a repo WITHOUT its `.claude` parent still matches this.
54
+ * - `rollout-*.jsonl` at any depth: Codex CLI's per-session file naming,
55
+ * corroborated in codex-cli.js's header (openai/codex issues #21660 and
56
+ * the archived-sessions issue both name `rollout-*.jsonl` verbatim).
57
+ * - any file under a `.specstory/` path component: SpecStory saves
58
+ * Cursor/Copilot chat history as Markdown into `.specstory/history/`
59
+ * inside the project, and its own docs describe committing that
60
+ * directory to share reasoning in PRs (docs.specstory.com/integrations/
61
+ * cursor; github.com/specstoryai/getspecstory). This is the one
62
+ * Cursor-export shape with a stable, citable on-disk location.
63
+ *
64
+ * Deliberately NOT matched, and why:
65
+ * - Cursor's built-in "export chat" output: the exported Markdown carries
66
+ * no stable name (community exporters observed during research use
67
+ * "{chat title}_{session id}.md", bare timestamps, and other schemes
68
+ * that disagree with each other). Any filename matcher here would be a
69
+ * guessed path; matching all `*.md` would scan every doc in the repo.
70
+ * A clean run therefore says nothing about hand-exported chat files.
71
+ * - generic `*.jsonl` anywhere: repos legitimately hold JSONL datasets
72
+ * and fixtures far larger than any transcript; scanning them all would
73
+ * drown the honest signal. The three transcript shapes above are the
74
+ * ones with citable naming.
75
+ *
76
+ * (b) Agent config/rules files at ANY depth (monorepos nest them):
77
+ * - `.claude/settings*.json` (settings.json, settings.local.json: the
78
+ * Lakera leak vector and the Mini Shai-Hulud/Miasma plant site)
79
+ * - `.mcp.json` (project-scope MCP config, Claude Code's own docs; the
80
+ * GitGuardian 24,008-secrets-in-MCP-configs category)
81
+ * - `.cursor/rules/*` (Miasma's setup.mdc plant site)
82
+ * - `.cursorrules` (TrapDoor's zero-width carrier)
83
+ * - `CLAUDE.md` and `CLAUDE.local.md` (Claude Code memory files, per its
84
+ * own memory docs; the other TrapDoor carrier)
85
+ * - `AGENTS.md` (the cross-vendor agent-instructions convention Codex and
86
+ * others load; codex-cli.js's research trail covers it)
87
+ * - `.gemini/settings.json` (Miasma plant site)
88
+ * - `.vscode/tasks.json` (the "runOn": "folderOpen" persistence surface)
89
+ *
90
+ * (c) `.env` files at the ROOT only (`.env`, `.env.local`, `.env.production`,
91
+ * `.env.example`, any `.env.*`). The root .env is the classic accidental
92
+ * commit. Deeper .env files are very often fixtures, scaffold templates,
93
+ * and per-package samples; a monorepo's `packages/x/.env` is therefore a
94
+ * NAMED exclusion with a real false-negative risk, not an oversight.
95
+ * Revisit with evidence if deeper .envs prove to leak in practice.
96
+ * `.env.example` at the root IS included on purpose: a real key pasted
97
+ * into an example file gets committed by design, and scan.js's
98
+ * placeholder suppression already keeps template content quiet.
99
+ *
100
+ * NOT walked at all: the CONTENTS of `node_modules/` and `.git/`.
101
+ * node_modules is other people's published code (a scan of it is an audit of
102
+ * the npm registry, not of this repo, and it blows any node budget on every
103
+ * real project); .git holds zlib-compressed objects the line engine cannot
104
+ * read meaningfully. Both skips are unconditional and silent because the
105
+ * directories are expected on virtually every repo; a skipped EXPECTED
106
+ * directory is not a truncation. Every UNEXPECTED cut (depth cap, node cap,
107
+ * unreadable directory) is surfaced as a broken entry instead, because a
108
+ * bounded walk that ends quietly is a false all-clear (CONTRIBUTING.md
109
+ * rule 5).
110
+ *
111
+ * Symlinks are followed like claude-code.js (see its isKindFollowingSymlink
112
+ * docstring for the lstat-vs-stat reasoning) but CONTAINED to the project
113
+ * root by realpath: a directory or candidate file that resolves outside the
114
+ * root is never walked or read, and is surfaced as a broken (not fully
115
+ * scanned) entry instead of silently skipped. Without containment a
116
+ * committed symlink ("vendored -> ../../somewhere") would pull the invoking
117
+ * machine's own files into the repo verdict, which is precisely the
118
+ * wrong-thing claim project mode exists to prevent: this scan's verdict is
119
+ * about the checkout, never about the machine around it. Directory symlink
120
+ * loops are cut with a realpath visited-set rather than left to the node
121
+ * cap, so a loop cannot eat the whole node budget before legitimate files
122
+ * are reached.
123
+ */
124
+
125
+ const MAX_DEPTH = 12; // deep enough for any real monorepo layout; a
126
+ // deeper tree gets a broken entry, not silence
127
+ const MAX_NODES = 20_000; // directory entries examined, not files yielded
128
+ const SKIP_DIRS = new Set(["node_modules", ".git"]);
129
+
130
+ // Committed transcripts are the same artifact class claude-code.js reads at
131
+ // home level, so the same bound applies: generous headroom over the largest
132
+ // real transcript this project has been tested against (818MB).
133
+ const MAX_BYTES = 2 * 1024 * 1024 * 1024;
134
+ const READ_TIMEOUT_MS = 60_000;
135
+
136
+ const ID = "project-artifacts";
137
+ const LABEL = "Project artifacts";
138
+
139
+ // Same shape as claude-code.js; duplicated per the one-file-per-source
140
+ // convention (each source stays auditable on its own).
141
+ function isKindFollowingSymlink(fullPath, dirent, checkFn) {
142
+ if (checkFn(dirent)) return true;
143
+ if (!dirent.isSymbolicLink()) return false;
144
+ try { return checkFn(fs.statSync(fullPath)); } catch { return false; }
145
+ }
146
+ const isDirFollowingSymlink = (p, d) => isKindFollowingSymlink(p, d, (x) => x.isDirectory());
147
+ const isFileFollowingSymlink = (p, d) => isKindFollowingSymlink(p, d, (x) => x.isFile());
148
+
149
+ /**
150
+ * Decide whether one regular file is a candidate, given its path segments
151
+ * relative to the root (segs includes the basename; depth 0 means the file
152
+ * sits directly in the root). Pure function, no filesystem access, so the
153
+ * whole inclusion policy is testable in one place.
154
+ */
155
+ function isCandidate(segs) {
156
+ const name = segs[segs.length - 1];
157
+ const parent = segs.length >= 2 ? segs[segs.length - 2] : null;
158
+ const inClaudeDir = segs.slice(0, -1).includes(".claude");
159
+ const inSpecstoryDir = segs.slice(0, -1).includes(".specstory");
160
+ const inCursorRules = segs.slice(0, -1).some(
161
+ (s, i) => s === ".cursor" && segs[i + 1] === "rules" && i + 1 < segs.length - 1
162
+ );
163
+
164
+ // (a) transcripts
165
+ if (name.endsWith(".jsonl")) {
166
+ if (inClaudeDir) return true;
167
+ if (parent && parent.startsWith("-")) return true; // claude-projects slug shape
168
+ if (/^rollout-.*\.jsonl$/.test(name)) return true; // Codex session naming
169
+ }
170
+ if (inSpecstoryDir) return true;
171
+
172
+ // (b) configs, any depth
173
+ if (parent === ".claude" && /^settings.*\.json$/.test(name)) return true;
174
+ if (name === ".mcp.json") return true;
175
+ if (inCursorRules) return true;
176
+ if (name === ".cursorrules") return true;
177
+ if (name === "CLAUDE.md" || name === "CLAUDE.local.md") return true;
178
+ if (name === "AGENTS.md") return true;
179
+ if (parent === ".gemini" && name === "settings.json") return true;
180
+ if (parent === ".vscode" && name === "tasks.json") return true;
181
+
182
+ // (c) root-level .env family only; see the header for why depth matters
183
+ if (segs.length === 1 && /^\.env(\..+)?$/.test(name)) return true;
184
+
185
+ return false;
186
+ }
187
+
188
+ /**
189
+ * Same streaming reader as claude-code.js and agent-configs.js (see the
190
+ * former for the timeout rationale: an open() on a retargeted symlink can
191
+ * block forever, and destroying the stream is the only way out). Standalone
192
+ * so both the disabled default export and every withRoot() instance share
193
+ * one implementation.
194
+ */
195
+ async function readLines(file) {
196
+ let stat;
197
+ try { stat = fs.statSync(file); }
198
+ catch { return { lines: [], status: "failed", bytesRead: 0 }; }
199
+ if (stat.size > MAX_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
200
+
201
+ const lines = [];
202
+ let bytesRead = 0;
203
+ const stream = fs.createReadStream(file, { encoding: "utf-8" });
204
+ const rl = createInterface({ input: stream, crlfDelay: Infinity });
205
+ const timer = setTimeout(() => stream.destroy(new Error("read timed out")), READ_TIMEOUT_MS);
206
+
207
+ try {
208
+ for await (const line of rl) {
209
+ lines.push(line);
210
+ bytesRead += Buffer.byteLength(line, "utf-8") + 1; // +1 for the stripped newline
211
+ }
212
+ return { lines, status: "complete", bytesRead };
213
+ } catch {
214
+ // Lines read before the failure are real content and may hold a real
215
+ // secret; an honest "partial" beats a silent false negative.
216
+ return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
217
+ } finally {
218
+ clearTimeout(timer);
219
+ rl.close();
220
+ stream.destroy();
221
+ }
222
+ }
223
+
224
+ /**
225
+ * Build a configured source instance for one project root. Returns a fresh
226
+ * object satisfying the full { id, label, available, files, readLines }
227
+ * contract, ready to be passed to scan({ sources: [...] }).
228
+ *
229
+ * label() deliberately does NOT embed the root path: source labels reach the
230
+ * report, and an absolute path can carry a username or project name the rest
231
+ * of the report is careful never to print (the same reasoning scan.js gives
232
+ * for basenames in unreadableFiles).
233
+ */
234
+ function withRoot(root = process.cwd()) {
235
+ const ROOT = path.resolve(root);
236
+
237
+ function available() {
238
+ try { return fs.statSync(ROOT).isDirectory(); } catch { return false; }
239
+ }
240
+
241
+ /**
242
+ * Iterative depth-first walk yielding { file, mtimeMs, sizeBytes, broken }
243
+ * for every candidate. Truncation policy, restated from the header because
244
+ * it is the load-bearing part: SKIP_DIRS vanish silently (expected on
245
+ * every repo, not a truncation); a directory cut by MAX_DEPTH, an
246
+ * unreadable directory, and a walk stopped by MAX_NODES each yield a
247
+ * broken entry, so scan.js surfaces them in unreadableFiles instead of
248
+ * folding the cut into a clean report.
249
+ */
250
+ function* files() {
251
+ let nodesSeen = 0;
252
+ const visitedDirs = new Set(); // realpaths, symlink-loop cut
253
+ // The containment anchor: everything walked or read must resolve to
254
+ // rootReal or below. If the root itself cannot be realpath'd the walk
255
+ // still runs bounded, but containment cannot be enforced; that is the
256
+ // caller's own unreadable-root situation, not an attacker-created one.
257
+ let rootReal = null;
258
+ try { rootReal = fs.realpathSync(ROOT); visitedDirs.add(rootReal); } catch { /* walk still bounded without it */ }
259
+ const inRoot = (real) =>
260
+ rootReal === null || real === rootReal || real.startsWith(rootReal + path.sep);
261
+
262
+ const stack = [{ dir: ROOT, segs: [] }];
263
+ while (stack.length > 0) {
264
+ const { dir, segs } = stack.pop();
265
+
266
+ let entries;
267
+ try { entries = fs.readdirSync(dir, { withFileTypes: true }); }
268
+ catch { yield { file: dir, broken: true }; continue; }
269
+
270
+ for (const e of entries) {
271
+ if (++nodesSeen > MAX_NODES) {
272
+ // The walk is stopping with work left. Reported against the
273
+ // directory being read because that is the most precise location
274
+ // the files() contract can carry.
275
+ yield { file: dir, broken: true };
276
+ return;
277
+ }
278
+ const full = path.join(dir, e.name);
279
+ const childSegs = segs.concat(e.name);
280
+
281
+ if (isDirFollowingSymlink(full, e)) {
282
+ if (SKIP_DIRS.has(e.name)) continue;
283
+ if (childSegs.length >= MAX_DEPTH) { yield { file: full, broken: true }; continue; }
284
+ // Every directory is deduped by realpath, not only symlinks: a
285
+ // symlinked route and the real directory reached later would
286
+ // otherwise both be walked, and one secret would be reported
287
+ // twice under two paths.
288
+ let real = null;
289
+ try { real = fs.realpathSync(full); }
290
+ catch {
291
+ // A symlink whose target cannot be resolved is a reportable
292
+ // failure; a plain directory failing realpath is unusual, and
293
+ // the readdir above will surface it loudly if it is unreadable.
294
+ if (e.isSymbolicLink()) { yield { file: full, broken: true }; continue; }
295
+ }
296
+ if (real !== null) {
297
+ if (!inRoot(real)) {
298
+ // A directory that resolves OUTSIDE the project root (a
299
+ // committed symlink to the invoking machine's own tree) is
300
+ // never walked: whatever lives there is not part of this
301
+ // checkout, and pulling it in would make a repo verdict about
302
+ // someone's home directory. Surfaced as broken, not skipped
303
+ // silently, so the report says this subtree went unexamined.
304
+ yield { file: full, broken: true };
305
+ continue;
306
+ }
307
+ if (visitedDirs.has(real)) continue; // loop or duplicate route, already covered
308
+ visitedDirs.add(real);
309
+ }
310
+ stack.push({ dir: full, segs: childSegs });
311
+ continue;
312
+ }
313
+
314
+ if (!isFileFollowingSymlink(full, e)) {
315
+ // A dangling symlink with a candidate name is exactly the entry
316
+ // the broken convention exists for; any other non-file oddity is
317
+ // out of scope, same as claude-code.js.
318
+ if (e.isSymbolicLink() && isCandidate(childSegs)) yield { file: full, broken: true };
319
+ continue;
320
+ }
321
+
322
+ if (!isCandidate(childSegs)) continue;
323
+ if (e.isSymbolicLink()) {
324
+ // Same containment as directories: a candidate-named symlink whose
325
+ // target resolves outside the root is disclosed, never read. Only
326
+ // targets inside the checkout are the checkout's content.
327
+ let realf = null;
328
+ try { realf = fs.realpathSync(full); }
329
+ catch { yield { file: full, broken: true }; continue; }
330
+ if (!inRoot(realf)) { yield { file: full, broken: true }; continue; }
331
+ }
332
+ let stat;
333
+ try { stat = fs.statSync(full); }
334
+ catch { yield { file: full, broken: true }; continue; }
335
+ yield { file: full, mtimeMs: stat.mtimeMs, sizeBytes: stat.size, broken: false };
336
+ }
337
+ }
338
+ }
339
+
340
+ return { id: () => ID, label: () => LABEL, available, files, readLines };
341
+ }
342
+
343
+ // Default export: the registry-safe DISABLED form. index.js may register it
344
+ // like any other singleton; available() is unconditionally false, so the
345
+ // default home scan never includes it, and files() yielding nothing is a
346
+ // harmless backstop should anything iterate it anyway. A project scan only
347
+ // ever happens through withRoot().
348
+ module.exports = {
349
+ id: () => ID,
350
+ label: () => LABEL,
351
+ available: () => false,
352
+ files: function* () {},
353
+ readLines,
354
+ withRoot,
355
+ };