residoo 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +225 -46
- package/SECURITY.md +29 -22
- package/package.json +1 -1
- package/src/cli.js +82 -16
- package/src/integrity.js +669 -0
- package/src/patterns.js +78 -5
- package/src/report.js +74 -7
- package/src/sources/agent-configs.js +308 -0
- package/src/sources/aider.js +361 -0
- package/src/sources/amazon-q.js +199 -0
- package/src/sources/antigravity-cli.js +155 -0
- package/src/sources/cline.js +208 -0
- package/src/sources/codebuff.js +295 -0
- package/src/sources/codex-cli.js +258 -0
- package/src/sources/cody.js +325 -0
- package/src/sources/continue.js +408 -0
- package/src/sources/copilot-chat.js +272 -0
- package/src/sources/copilot-cli.js +300 -0
- package/src/sources/crush.js +364 -0
- package/src/sources/cursor.js +374 -0
- package/src/sources/devin-cli.js +241 -0
- package/src/sources/factory-droid.js +153 -0
- package/src/sources/fx.js +136 -0
- package/src/sources/gemini-cli.js +242 -0
- package/src/sources/goose.js +366 -0
- package/src/sources/grok-cli.js +267 -0
- package/src/sources/hermes.js +282 -0
- package/src/sources/index.js +172 -8
- package/src/sources/jetbrains-ai-assistant.js +343 -0
- package/src/sources/jetbrains-junie.js +292 -0
- package/src/sources/kilo-code.js +430 -0
- package/src/sources/kimi-code.js +147 -0
- package/src/sources/kiro-cli.js +393 -0
- package/src/sources/kiro-ide.js +230 -0
- package/src/sources/llm.js +328 -0
- package/src/sources/mentat.js +143 -0
- package/src/sources/open-interpreter.js +224 -0
- package/src/sources/openclaw.js +218 -0
- package/src/sources/opencode.js +379 -0
- package/src/sources/openhands.js +181 -0
- package/src/sources/pearai.js +151 -0
- package/src/sources/pi-agent.js +130 -0
- package/src/sources/qodo-gen.js +189 -0
- package/src/sources/qwen-code.js +244 -0
- package/src/sources/roo-code.js +239 -0
- package/src/sources/trae.js +294 -0
- package/src/sources/void.js +273 -0
- package/src/sources/warp.js +395 -0
- package/src/sources/windsurf.js +256 -0
- package/src/sources/zed.js +374 -0
|
@@ -0,0 +1,366 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
|
|
3
|
+
const fs = require("fs");
|
|
4
|
+
const { createInterface } = require("readline/promises");
|
|
5
|
+
const path = require("path");
|
|
6
|
+
const os = require("os");
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Goose (github.com/aaif-goose/goose — Block/Square's open-source AI agent,
|
|
10
|
+
* transferred to the Linux Foundation's Agentic AI Foundation in April 2026;
|
|
11
|
+
* same project, same code, `github.com/block/goose` now redirects there).
|
|
12
|
+
*
|
|
13
|
+
* VERIFICATION STATUS: NOT installed on the machine this adapter was built on
|
|
14
|
+
* (checked: no `goose` on PATH, no `~/.local/share/goose`, no
|
|
15
|
+
* `~/Library/Application Support/Block/goose`, no `~/.config/goose`, no
|
|
16
|
+
* Homebrew cask/formula installed — only the unrelated `mongoose` library and
|
|
17
|
+
* the Go database-migration tool of the same name turned up). Ships anyway
|
|
18
|
+
* per CONTRIBUTING.md rule 3, on unusually strong grounds for an "unverified"
|
|
19
|
+
* source: every claim below was confirmed against goose's OWN real source
|
|
20
|
+
* code on GitHub — read directly, not inferred from a description of it —
|
|
21
|
+
* cross-checked against goose's own published docs and, for one genuinely
|
|
22
|
+
* surprising point, the actual source of the third-party crate it delegates
|
|
23
|
+
* the decision to.
|
|
24
|
+
*
|
|
25
|
+
* Storage location — confirmed directly from source
|
|
26
|
+
* (`crates/goose/src/config/paths.rs`, `crates/goose/src/session/session_manager.rs`,
|
|
27
|
+
* both fetched from the `main` branch at commit-current state, schema_version 16):
|
|
28
|
+
*
|
|
29
|
+
* - `Paths::data_dir()`: if the `GOOSE_PATH_ROOT` env var is set to an
|
|
30
|
+
* ABSOLUTE path (validated — a relative value is ignored), the base is
|
|
31
|
+
* `$GOOSE_PATH_ROOT/data`. Otherwise it calls
|
|
32
|
+
* `etcetera::choose_app_strategy(AppStrategyArgs { top_level_domain:
|
|
33
|
+
* "Block", author: "Block", app_name: "goose" })` and returns that
|
|
34
|
+
* strategy's `data_dir()`.
|
|
35
|
+
* - The surprising point, resolved by reading the `etcetera` crate's own
|
|
36
|
+
* source (pinned to exactly 0.11.0 in goose's real `Cargo.lock`, fetched
|
|
37
|
+
* and read directly — `src/app_strategy.rs`'s `cfg_if!` block and
|
|
38
|
+
* `xdg.rs`): `choose_app_strategy` (unlike `choose_native_strategy`,
|
|
39
|
+
* which goose does NOT call) resolves to the **Xdg** strategy on macOS
|
|
40
|
+
* — not the Apple/`~/Library/Application Support` convention most macOS
|
|
41
|
+
* apps use, matching a documented, deliberate choice ("This is the
|
|
42
|
+
* convention used by most CLI applications") — and the Xdg strategy's
|
|
43
|
+
* `data_dir()` ignores `author`/`top_level_domain` entirely, using only
|
|
44
|
+
* `app_name`. So on BOTH macOS and Linux this resolves to
|
|
45
|
+
* `$XDG_DATA_HOME/goose` or `~/.local/share/goose` if that's unset — the
|
|
46
|
+
* `top_level_domain`/`author: "Block"` fields passed above are, per this
|
|
47
|
+
* reading, dead for the current default path (a maintainer comment right
|
|
48
|
+
* next to that call, kept for backwards compatibility with older
|
|
49
|
+
* installs, cites `~/Library/Application Support/Block/goose/` — that is
|
|
50
|
+
* the OLDER, pre-etcetera-migration path, not what today's code
|
|
51
|
+
* resolves to; the comment is about not orphaning those old installs,
|
|
52
|
+
* not about where new ones land). This macOS-follows-Linux choice is
|
|
53
|
+
* independently confirmed by goose's own published docs (goose-docs.ai's
|
|
54
|
+
* logging guide states the session DB path as `~/.local/share/goose/sessions/sessions.db`
|
|
55
|
+
* for "macOS and Linux" as one line, and separately
|
|
56
|
+
* `%APPDATA%\Block\goose\data\sessions\sessions.db` for Windows) — which
|
|
57
|
+
* also matches the `etcetera` Windows strategy's own documented formula
|
|
58
|
+
* (`AppData/Roaming/<author>/<app_name>/data`, author defaulting to
|
|
59
|
+
* nothing special here since it's passed explicitly as `"Block"`) read
|
|
60
|
+
* directly from `windows.rs`.
|
|
61
|
+
* - `SessionStorage::new()` joins `data_dir` with the literal constants
|
|
62
|
+
* `SESSIONS_FOLDER = "sessions"` then, for the DB, `DB_NAME = "sessions.db"`.
|
|
63
|
+
*
|
|
64
|
+
* Two formats live in that SAME `sessions/` directory, confirmed directly
|
|
65
|
+
* from source, and this adapter reads both:
|
|
66
|
+
*
|
|
67
|
+
* 1. `sessions.db` — the current format (schema_version 16 as of this
|
|
68
|
+
* research), a plain SQLite file opened via `sqlx`/`SqliteConnectOptions`
|
|
69
|
+
* (WAL mode, no encryption). Tables confirmed directly from the actual
|
|
70
|
+
* `CREATE TABLE` statements in `session_manager.rs`: `sessions`
|
|
71
|
+
* (metadata — name, working_dir, recipe_json, model_config_json, ...),
|
|
72
|
+
* `messages` (session_id, role, `content_json` — the actual message
|
|
73
|
+
* content, one row per message), `usage_ledger` (token/cost accounting,
|
|
74
|
+
* no message content), and, added by a later migration, `threads` /
|
|
75
|
+
* `thread_messages` (a second, parallel message-content table pair with
|
|
76
|
+
* its own `content_json` column — confirmed from the migration-9 SQL
|
|
77
|
+
* block). Given the schema has ALREADY gone through 16 versions of
|
|
78
|
+
* `ALTER TABLE`/new-table migrations in this one file, and per-table
|
|
79
|
+
* column sets keep changing, this source does not hardcode a
|
|
80
|
+
* table/column allowlist — same reasoning opencode.js's own docstring
|
|
81
|
+
* gives for its generic scan, and the exact same tradeoff: reads every
|
|
82
|
+
* user table generically (see readDbFile() below), so a future
|
|
83
|
+
* migration adding yet another `..._json` content column can't silently
|
|
84
|
+
* stop being scanned the way a hardcoded list would.
|
|
85
|
+
* 2. `*.jsonl` — the legacy pre-1.10.0 format, confirmed directly from
|
|
86
|
+
* `crates/goose/src/session/legacy.rs`: `list_sessions()` does a flat,
|
|
87
|
+
* non-recursive `fs::read_dir` over the SAME `sessions/` directory for
|
|
88
|
+
* `.jsonl` entries, and `load_session()` confirms the shape — the FIRST
|
|
89
|
+
* line is one JSON object (session metadata), every subsequent line is
|
|
90
|
+
* one JSON message object — genuine JSONL, no re-serialization needed
|
|
91
|
+
* to turn it into scannable lines. Per goose's own docs (see above,
|
|
92
|
+
* also independently corroborated by ccusage.com's own goose-integration
|
|
93
|
+
* guide), upgrading to v1.10.0+ auto-imports these into `sessions.db`
|
|
94
|
+
* but leaves the original `.jsonl` files sitting on disk, unmanaged —
|
|
95
|
+
* real, orphaned, still-scannable content on any machine that has ever
|
|
96
|
+
* upgraded across that boundary, the same "two formats really do
|
|
97
|
+
* coexist on one real machine" situation opencode.js's own docstring
|
|
98
|
+
* describes for its own legacy JSON files.
|
|
99
|
+
*
|
|
100
|
+
* The Goose Desktop app (`ui/desktop/`, Electron/TS) was not traced all the
|
|
101
|
+
* way through its IPC/HTTP layer, but `ui/desktop/src/sessions.ts`'s own
|
|
102
|
+
* `Session` type uses the exact same field names as the Rust `Session` struct
|
|
103
|
+
* above (`user_set_name`, `recipe`, `name`, ...) — strong circumstantial
|
|
104
|
+
* evidence, not a full trace, that it talks to the same local `goosed`
|
|
105
|
+
* server backed by the same `SessionManager`/`sessions.db`, not a separate
|
|
106
|
+
* storage format of its own.
|
|
107
|
+
*
|
|
108
|
+
* Sources consulted: aaif-goose/goose source on GitHub (`crates/goose/src/
|
|
109
|
+
* config/paths.rs`, `crates/goose/src/session/session_manager.rs`,
|
|
110
|
+
* `crates/goose/src/session/legacy.rs`, `Cargo.lock`, all fetched from
|
|
111
|
+
* `main`); the `etcetera` crate's own source at the exact pinned version
|
|
112
|
+
* (lunacookies/etcetera tag v0.11.0 — `src/app_strategy.rs`, `src/app_strategy/
|
|
113
|
+
* xdg.rs`, `src/app_strategy/windows.rs`); goose's own published docs
|
|
114
|
+
* (goose-docs.ai/docs/guides/logs/); ccusage.com's goose integration guide
|
|
115
|
+
* (an independent tool that reads this same sessions.db in production,
|
|
116
|
+
* corroborating table/column names from the consumer side); DeepWiki's
|
|
117
|
+
* goose session-management summary (schema_version terminology cross-check).
|
|
118
|
+
*/
|
|
119
|
+
function goosePathRoot() {
|
|
120
|
+
const v = process.env.GOOSE_PATH_ROOT;
|
|
121
|
+
return v && path.isAbsolute(v) ? v : null;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
function xdgDataDir(appName) {
|
|
125
|
+
const home = os.homedir();
|
|
126
|
+
const dataHome = process.env.XDG_DATA_HOME || path.join(home, ".local", "share");
|
|
127
|
+
return path.join(dataHome, appName);
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
function windowsDataDir(author, appName) {
|
|
131
|
+
const home = os.homedir();
|
|
132
|
+
const base = process.env.APPDATA || path.join(home, "AppData", "Roaming");
|
|
133
|
+
return path.join(base, author, appName, "data");
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
function gooseDataDir() {
|
|
137
|
+
const root = goosePathRoot();
|
|
138
|
+
if (root) return path.join(root, "data");
|
|
139
|
+
if (process.platform === "win32") return windowsDataDir("Block", "goose");
|
|
140
|
+
// macOS AND Linux both resolve through etcetera's Xdg strategy here — see
|
|
141
|
+
// the module docstring for why macOS does NOT get the `~/Library/
|
|
142
|
+
// Application Support` treatment most native macOS apps get.
|
|
143
|
+
return xdgDataDir("goose");
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
const SESSIONS_DIR = path.join(gooseDataDir(), "sessions");
|
|
147
|
+
const DB_FILE = path.join(SESSIONS_DIR, "sessions.db");
|
|
148
|
+
|
|
149
|
+
const MAX_BYTES = 2 * 1024 * 1024 * 1024; // 2GB — same reasoning as claude-code.js
|
|
150
|
+
const READ_TIMEOUT_MS = 60_000;
|
|
151
|
+
|
|
152
|
+
// Same backstop-not-evidence caveat as cursor.js's MAX_DB_BYTES: no real
|
|
153
|
+
// sessions.db has been observed during this research to size this against.
|
|
154
|
+
const MAX_DB_BYTES = 512 * 1024 * 1024;
|
|
155
|
+
const BUSY_TIMEOUT_MS = 5_000;
|
|
156
|
+
const YIELD_EVERY_N_ROWS = 500;
|
|
157
|
+
|
|
158
|
+
function id() { return "goose"; }
|
|
159
|
+
function label() { return "Goose"; }
|
|
160
|
+
|
|
161
|
+
function sessionsDirExists() {
|
|
162
|
+
try { return fs.statSync(SESSIONS_DIR).isDirectory(); } catch { return false; }
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* Unlike cursor.js, availability does NOT depend on node:sqlite being
|
|
167
|
+
* present: the legacy `*.jsonl` format (see module docstring) is plain text,
|
|
168
|
+
* readable with zero special modules. Gating the whole source on sqlite
|
|
169
|
+
* would silently drop real, scannable content for exactly the
|
|
170
|
+
* upgraded-past-1.10.0-with-orphaned-jsonl case goose's own docs describe, or
|
|
171
|
+
* for a Node runtime too old for node:sqlite. sqlite is required only for
|
|
172
|
+
* sessions.db specifically — see readDbFile()'s own "failed" return when
|
|
173
|
+
* it's unavailable, surfaced per-file rather than by hiding the whole
|
|
174
|
+
* source. Mirrors opencode.js's identical reasoning for its own two-format
|
|
175
|
+
* (sqlite + legacy plain-text) source.
|
|
176
|
+
*/
|
|
177
|
+
function available() {
|
|
178
|
+
return sessionsDirExists();
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Lazily loaded exactly like cursor.js's/opencode.js's getDatabaseSync() —
|
|
183
|
+
* see cursor.js's docstring for the full reasoning (avoid the one-time
|
|
184
|
+
* ExperimentalWarning node:sqlite prints from being paid by every user on
|
|
185
|
+
* every invocation, since index.js requires every source unconditionally).
|
|
186
|
+
* Duplicated rather than shared, per this project's one-file-per-source
|
|
187
|
+
* convention.
|
|
188
|
+
*/
|
|
189
|
+
let sqliteRequireAttempted = false;
|
|
190
|
+
let DatabaseSync = null;
|
|
191
|
+
function getDatabaseSync() {
|
|
192
|
+
if (!sqliteRequireAttempted) {
|
|
193
|
+
sqliteRequireAttempted = true;
|
|
194
|
+
try { ({ DatabaseSync } = require("node:sqlite")); }
|
|
195
|
+
catch { DatabaseSync = null; }
|
|
196
|
+
}
|
|
197
|
+
return DatabaseSync;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** Same lstat-vs-stat symlink-following pattern duplicated across every source in this project. */
|
|
201
|
+
function isFileFollowingSymlink(fullPath, dirent) {
|
|
202
|
+
if (dirent.isFile()) return true;
|
|
203
|
+
if (!dirent.isSymbolicLink()) return false;
|
|
204
|
+
try { return fs.statSync(fullPath).isFile(); } catch { return false; }
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/**
|
|
208
|
+
* Yield { file, mtimeMs, sizeBytes, broken } for sessions.db (if present) and
|
|
209
|
+
* every legacy `*.jsonl` session file — both live directly inside
|
|
210
|
+
* SESSIONS_DIR, flat and non-recursive, matching `legacy.rs`'s own
|
|
211
|
+
* `fs::read_dir` (no subfolder nesting in this format). Purely a filesystem
|
|
212
|
+
* walk + stat, same division of labour as cursor.js's/opencode.js's files():
|
|
213
|
+
* never opens the database, so this works even without node:sqlite; only
|
|
214
|
+
* readLines() needs it.
|
|
215
|
+
*/
|
|
216
|
+
function* files() {
|
|
217
|
+
let entries;
|
|
218
|
+
try { entries = fs.readdirSync(SESSIONS_DIR, { withFileTypes: true }); }
|
|
219
|
+
catch { return; } // no sessions dir at all — available() already said so, but stay defensive
|
|
220
|
+
|
|
221
|
+
for (const e of entries) {
|
|
222
|
+
const isDb = e.name === "sessions.db";
|
|
223
|
+
const isLegacy = e.name.endsWith(".jsonl");
|
|
224
|
+
if (!isDb && !isLegacy) continue;
|
|
225
|
+
|
|
226
|
+
const file = path.join(SESSIONS_DIR, e.name);
|
|
227
|
+
if (!isFileFollowingSymlink(file, e)) {
|
|
228
|
+
if (e.isSymbolicLink()) yield { file, broken: true };
|
|
229
|
+
continue;
|
|
230
|
+
}
|
|
231
|
+
let stat;
|
|
232
|
+
try { stat = fs.statSync(file); } catch { yield { file, broken: true }; continue; }
|
|
233
|
+
yield { file, mtimeMs: stat.mtimeMs, sizeBytes: stat.size, broken: false };
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
/** Same string/Uint8Array decode as cursor.js's/opencode.js's valueToText — see cursor.js's docstring for why. */
|
|
238
|
+
function valueToText(value) {
|
|
239
|
+
if (typeof value === "string") return value;
|
|
240
|
+
if (value instanceof Uint8Array) return Buffer.from(value).toString("utf-8");
|
|
241
|
+
return null;
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* Read sessions.db generically: every user table in sqlite_master, every
|
|
246
|
+
* column of every row, each non-null string/blob value becoming one scanned
|
|
247
|
+
* line (see module docstring for why table/column names beyond knowing
|
|
248
|
+
* `sessions`/`messages`/`usage_ledger`/`threads`/`thread_messages` exist
|
|
249
|
+
* aren't hardcoded further). Row-by-row with a periodic yield-and-deadline
|
|
250
|
+
* check, identical strategy to cursor.js's/opencode.js's readLines/readDbFile
|
|
251
|
+
* for the same reason: node:sqlite's DatabaseSync is fully synchronous, so
|
|
252
|
+
* this is the only preemption point available without adding a dependency.
|
|
253
|
+
*/
|
|
254
|
+
async function readDbFile(file) {
|
|
255
|
+
const DB = getDatabaseSync();
|
|
256
|
+
if (!DB) return { lines: [], status: "failed", bytesRead: 0 };
|
|
257
|
+
|
|
258
|
+
let stat;
|
|
259
|
+
try { stat = fs.statSync(file); }
|
|
260
|
+
catch { return { lines: [], status: "failed", bytesRead: 0 }; }
|
|
261
|
+
if (stat.size > MAX_DB_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
|
|
262
|
+
|
|
263
|
+
let db;
|
|
264
|
+
try {
|
|
265
|
+
db = new DB(file, { readOnly: true });
|
|
266
|
+
db.exec(`PRAGMA busy_timeout = ${BUSY_TIMEOUT_MS}`);
|
|
267
|
+
} catch {
|
|
268
|
+
// Covers: deleted between files() and this call, a corrupt/non-SQLite
|
|
269
|
+
// file, or goose holding a lock this readonly open can't get past within
|
|
270
|
+
// BUSY_TIMEOUT_MS. All three are "could not read this," not "read it,
|
|
271
|
+
// found nothing" — status "failed" keeps the report honest either way.
|
|
272
|
+
return { lines: [], status: "failed", bytesRead: 0 };
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
let tables;
|
|
276
|
+
try {
|
|
277
|
+
tables = db.prepare(
|
|
278
|
+
"SELECT name FROM sqlite_master WHERE type = 'table' AND name NOT LIKE 'sqlite_%'"
|
|
279
|
+
).all().map((r) => r.name);
|
|
280
|
+
} catch {
|
|
281
|
+
try { db.close(); } catch { /* best-effort */ }
|
|
282
|
+
return { lines: [], status: "failed", bytesRead: 0 };
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
const lines = [];
|
|
286
|
+
let bytesRead = 0;
|
|
287
|
+
const deadline = Date.now() + READ_TIMEOUT_MS;
|
|
288
|
+
let timedOut = false;
|
|
289
|
+
let sawError = false;
|
|
290
|
+
|
|
291
|
+
for (const table of tables) {
|
|
292
|
+
let rows;
|
|
293
|
+
try {
|
|
294
|
+
// Table names come from sqlite_master itself, not external input, but
|
|
295
|
+
// still quoted defensively rather than trusted to be bare-word-safe.
|
|
296
|
+
rows = db.prepare(`SELECT * FROM "${table.replace(/"/g, '""')}"`).iterate();
|
|
297
|
+
} catch {
|
|
298
|
+
continue; // this table vanished or is a view/virtual table that doesn't support this — move on
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
let n = 0;
|
|
302
|
+
try {
|
|
303
|
+
for (const row of rows) {
|
|
304
|
+
for (const key in row) {
|
|
305
|
+
const text = valueToText(row[key]);
|
|
306
|
+
if (text) { lines.push(text); bytesRead += Buffer.byteLength(text, "utf-8"); }
|
|
307
|
+
}
|
|
308
|
+
n++;
|
|
309
|
+
if (n % YIELD_EVERY_N_ROWS === 0) {
|
|
310
|
+
await new Promise((resolve) => setImmediate(resolve));
|
|
311
|
+
if (Date.now() > deadline) { timedOut = true; break; }
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
} catch {
|
|
315
|
+
// A row iterator can itself throw partway (e.g. a corrupted page hit
|
|
316
|
+
// mid-scan) — whatever WAS read before that is real content, kept the
|
|
317
|
+
// same way claude-code.js keeps a partial read rather than discarding it.
|
|
318
|
+
sawError = true;
|
|
319
|
+
}
|
|
320
|
+
if (timedOut) break;
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
try { db.close(); } catch { /* best-effort close; nothing left to do if this fails */ }
|
|
324
|
+
|
|
325
|
+
if (sawError && lines.length === 0) return { lines: [], status: "failed", bytesRead };
|
|
326
|
+
if (timedOut || sawError) return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
|
|
327
|
+
return { lines, status: "complete", bytesRead };
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
/**
|
|
331
|
+
* Read one legacy `*.jsonl` session file as raw text lines — genuinely
|
|
332
|
+
* line-delimited already (see module docstring), so this is the same plain
|
|
333
|
+
* streamed read as claude-code.js's readLines(), no reformatting needed.
|
|
334
|
+
*/
|
|
335
|
+
async function readJsonlFile(file) {
|
|
336
|
+
let stat;
|
|
337
|
+
try { stat = fs.statSync(file); }
|
|
338
|
+
catch { return { lines: [], status: "failed", bytesRead: 0 }; }
|
|
339
|
+
if (stat.size > MAX_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
|
|
340
|
+
|
|
341
|
+
const lines = [];
|
|
342
|
+
let bytesRead = 0;
|
|
343
|
+
const stream = fs.createReadStream(file, { encoding: "utf-8" });
|
|
344
|
+
const rl = createInterface({ input: stream, crlfDelay: Infinity });
|
|
345
|
+
const timer = setTimeout(() => stream.destroy(new Error("read timed out")), READ_TIMEOUT_MS);
|
|
346
|
+
|
|
347
|
+
try {
|
|
348
|
+
for await (const line of rl) {
|
|
349
|
+
lines.push(line);
|
|
350
|
+
bytesRead += Buffer.byteLength(line, "utf-8") + 1;
|
|
351
|
+
}
|
|
352
|
+
return { lines, status: "complete", bytesRead };
|
|
353
|
+
} catch {
|
|
354
|
+
return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
|
|
355
|
+
} finally {
|
|
356
|
+
clearTimeout(timer);
|
|
357
|
+
rl.close();
|
|
358
|
+
stream.destroy();
|
|
359
|
+
}
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
async function readLines(file) {
|
|
363
|
+
return file === DB_FILE ? readDbFile(file) : readJsonlFile(file);
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
module.exports = { id, label, available, files, readLines };
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
|
|
3
|
+
const fs = require("fs");
|
|
4
|
+
const { createInterface } = require("readline/promises");
|
|
5
|
+
const path = require("path");
|
|
6
|
+
const os = require("os");
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Grok Build (xAI's official coding-agent CLI/TUI, binary name `grok` — the
|
|
10
|
+
* product commonly referred to as "Grok CLI") session transcripts.
|
|
11
|
+
*
|
|
12
|
+
* VERIFICATION STATUS (read this before trusting anything below): this
|
|
13
|
+
* source is corroborated by the actual current Rust source code AND
|
|
14
|
+
* first-party documentation of the official xai-org/grok-build repository on
|
|
15
|
+
* GitHub — fetched and read directly from the `main` branch (path-resolution
|
|
16
|
+
* code with its own unit tests, the JSONL storage adapter, and the shipped
|
|
17
|
+
* user-guide doc page), not inferred from a description of it. It has NOT
|
|
18
|
+
* been checked against a real Grok Build install — it is not installed on
|
|
19
|
+
* the machine this adapter was built on (checked: no `grok` on PATH, no
|
|
20
|
+
* `~/.grok` directory). If you have Grok Build installed, the most useful
|
|
21
|
+
* thing you can do is run `residoo scan` and confirm
|
|
22
|
+
* `sourcesScanned`/`filesScanned` look right for what you know is actually
|
|
23
|
+
* under `~/.grok/sessions`, then report back either way.
|
|
24
|
+
*
|
|
25
|
+
* Note on naming: this project's own repo (public, confirmed via `gh api
|
|
26
|
+
* repos/xai-org/grok-build`) and README call the product "Grok Build"; the
|
|
27
|
+
* command is `grok`. Various unofficial third-party npm packages are also
|
|
28
|
+
* named "grok-cli" (e.g. `superagent-ai/grok-cli`) — those are NOT xAI's own
|
|
29
|
+
* tool and are out of scope here; this source is specifically xAI's
|
|
30
|
+
* first-party agent, whatever it's branded as this month.
|
|
31
|
+
*
|
|
32
|
+
* HOME DIRECTORY. `crates/codegen/xai-dirs/src/lib.rs` (`grok_home()`,
|
|
33
|
+
* `home_dir()`, with its own unit tests) resolves, in this exact order:
|
|
34
|
+
* 1. `$GROK_HOME`, used verbatim (not canonicalized), when set and
|
|
35
|
+
* non-empty.
|
|
36
|
+
* 2. Otherwise `<home>/.grok`, where `<home>` comes from
|
|
37
|
+
* `std::env::home_dir()` — `$HOME` on Unix (with a passwd-database
|
|
38
|
+
* fallback), `%USERPROFILE%` on Windows — canonicalized via `dunce`
|
|
39
|
+
* (resolves symlinks without producing Windows `\\?\` verbatim paths).
|
|
40
|
+
* No OS-specific base directory (no XDG, no `Library/Application Support`):
|
|
41
|
+
* always `<home>/.grok`, on every platform, confirmed directly from source —
|
|
42
|
+
* unlike several other sources in this project (Cursor, Gemini CLI), there
|
|
43
|
+
* is no per-OS branch to replicate here. This source does not replicate the
|
|
44
|
+
* `dunce` canonicalization step (Node has no direct equivalent, and the one
|
|
45
|
+
* documented case it matters for — macOS's `/var` vs `/private/var` — does
|
|
46
|
+
* not apply to a user's own home directory in practice); a noted, minor,
|
|
47
|
+
* intentional gap.
|
|
48
|
+
*
|
|
49
|
+
* STORAGE LAYOUT. `crates/codegen/xai-grok-config/src/paths.rs`
|
|
50
|
+
* (`encode_cwd_dirname`, `sessions_cwd_dir_in`, both with extensive unit
|
|
51
|
+
* tests covering short/long/non-ASCII working directories) and
|
|
52
|
+
* `crates/codegen/xai-grok-shell/src/session/storage/jsonl/mod.rs` (doc
|
|
53
|
+
* comment: "JSONL storage under `{root}/sessions/{url_encoded_cwd}/
|
|
54
|
+
* {session_id}/`") confirm the on-disk shape:
|
|
55
|
+
*
|
|
56
|
+
* ~/.grok/sessions/<encoded-cwd>/<session-id>/
|
|
57
|
+
*
|
|
58
|
+
* `<encoded-cwd>` is the session's working directory, percent-encoded
|
|
59
|
+
* (`urlencoding::encode`) when that fits in 255 bytes, else a
|
|
60
|
+
* `{slug}-{blake3_hex16}` fallback with the real path recorded in a `.cwd`
|
|
61
|
+
* file inside — see `encode_cwd_dirname()`'s doc comment for the exact
|
|
62
|
+
* rule. This source does not attempt to reproduce that encoding or recover
|
|
63
|
+
* the original working directory: it walks whatever directories exist under
|
|
64
|
+
* `sessions/` generically (same judgment call gemini-cli.js makes for
|
|
65
|
+
* `~/.gemini/tmp/<projectIdentifier>/` rather than recomputing that tool's
|
|
66
|
+
* own hash/slug scheme), because doing so is unnecessary — files() only
|
|
67
|
+
* needs to find candidate files, not decode what project they belong to.
|
|
68
|
+
*
|
|
69
|
+
* Per session, xAI's own shipped user-guide doc
|
|
70
|
+
* (`crates/codegen/xai-grok-pager/docs/user-guide/17-sessions.md`, mirrored
|
|
71
|
+
* near-verbatim in `crates/codegen/xai-grok-shell/README.md`) documents the
|
|
72
|
+
* files inside each session directory:
|
|
73
|
+
*
|
|
74
|
+
* summary.json # metadata: summary/title, timestamps, model ID, message counts
|
|
75
|
+
* updates.jsonl # ACP session update stream (conversation + tool calls) — SOURCE OF TRUTH
|
|
76
|
+
* chat_history.jsonl # raw chat messages sent to the model
|
|
77
|
+
* plan.json # TODO/task list state
|
|
78
|
+
* rewind_points.jsonl # rewind points for /rewind undo
|
|
79
|
+
* signals.json # session signals (token usage, tool/turn counters)
|
|
80
|
+
* feedback.jsonl # user feedback and ratings
|
|
81
|
+
* compaction_checkpoints/ # saved state from compaction (manual or auto)
|
|
82
|
+
* subagents/ # per-subagent metadata (meta.json) only — the
|
|
83
|
+
* # child sessions themselves are ordinary
|
|
84
|
+
* # top-level session dirs elsewhere in the tree
|
|
85
|
+
* # (confirmed in jsonl/mod.rs:
|
|
86
|
+
* # `with_explicit_session_dir` doc comment,
|
|
87
|
+
* # "Subagent child sessions use this (top-level
|
|
88
|
+
* # dirs; only their metadata nests under the
|
|
89
|
+
* # parent's session dir)").
|
|
90
|
+
*
|
|
91
|
+
* The doc page states outright which files matter for content: "JSONL is
|
|
92
|
+
* the source of truth for session content" (`updates.jsonl` and
|
|
93
|
+
* `chat_history.jsonl` specifically — both are grep-confirmed by name
|
|
94
|
+
* dozens of times across the actual production code, its own tests, and its
|
|
95
|
+
* prompt templates, not just the doc prose). A separate local SQLite FTS5
|
|
96
|
+
* index (`grok sessions search`) exists purely as a derived keyword index
|
|
97
|
+
* over titles/prompts, explicitly secondary to the JSONL per the same doc —
|
|
98
|
+
* not scanned here, on the same reasoning cursor.js gives for preferring a
|
|
99
|
+
* primary store over a derived cache: scanning the source of truth is both
|
|
100
|
+
* sufficient and simpler than also opening a SQLite index that can only
|
|
101
|
+
* echo a subset of what the JSONL already holds.
|
|
102
|
+
*
|
|
103
|
+
* This source deliberately does NOT hardcode that filename list. It walks
|
|
104
|
+
* every file under each session directory (bounded depth, so
|
|
105
|
+
* `compaction_checkpoints/` and `subagents/` are covered too) and scans any
|
|
106
|
+
* `.json`/`.jsonl` file found there, for the same reason cursor.js declines
|
|
107
|
+
* to hardcode a key-name allowlist and warp.js declines to hardcode a table
|
|
108
|
+
* list: this project is under fast, active development (its own docs were
|
|
109
|
+
* last updated within the research window for this source) and a stale
|
|
110
|
+
* hardcoded filename list is exactly the kind of silent, permanent gap
|
|
111
|
+
* CONTRIBUTING.md rule 5 exists to prevent. A file this source doesn't
|
|
112
|
+
* specifically recognize costs nothing extra to scan.
|
|
113
|
+
*/
|
|
114
|
+
function grokHome() {
|
|
115
|
+
const grokHomeEnv = process.env.GROK_HOME;
|
|
116
|
+
if (grokHomeEnv && grokHomeEnv !== "") return grokHomeEnv;
|
|
117
|
+
return path.join(os.homedir(), ".grok");
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
const ROOT = grokHome();
|
|
121
|
+
const SESSIONS_DIR = path.join(ROOT, "sessions");
|
|
122
|
+
const SESSION_FILE_EXT = /\.(jsonl|json)$/i;
|
|
123
|
+
|
|
124
|
+
// How many levels deep files() will descend from a session-id directory
|
|
125
|
+
// looking for content files. The only nesting actually documented is one
|
|
126
|
+
// extra level (compaction_checkpoints/*, subagents/*), but this is kept
|
|
127
|
+
// bounded rather than hard-coded at exactly 1 so a future extra level of
|
|
128
|
+
// nesting gets scanned rather than silently missed — and bounded at all so
|
|
129
|
+
// a symlink cycle can't turn this into an infinite walk. Same convention
|
|
130
|
+
// and same reasoning as gemini-cli.js's MAX_CHATS_DEPTH.
|
|
131
|
+
const MAX_SESSION_DEPTH = 4;
|
|
132
|
+
|
|
133
|
+
// Bounds for readLines() — same shape as claude-code.js's, but the actual
|
|
134
|
+
// number is NOT backed by a real large Grok Build transcript this tool was
|
|
135
|
+
// tested against (unlike claude-code.js's, which cites a real 818MB file);
|
|
136
|
+
// no install was available to produce one. `grok du`'s own sample output
|
|
137
|
+
// (quoted in the sessions doc) shows a `sessions` directory reaching low
|
|
138
|
+
// gigabytes in ordinary heavy use, well under this cap.
|
|
139
|
+
const MAX_BYTES = 2 * 1024 * 1024 * 1024; // 2GB
|
|
140
|
+
const READ_TIMEOUT_MS = 60_000;
|
|
141
|
+
|
|
142
|
+
function id() { return "grok-cli"; }
|
|
143
|
+
function label() { return "Grok Build"; }
|
|
144
|
+
|
|
145
|
+
function available() {
|
|
146
|
+
try { return fs.statSync(SESSIONS_DIR).isDirectory(); } catch { return false; }
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Same defensive symlink-following pattern as claude-code.js — see that
|
|
151
|
+
* file's docstring for the full reasoning. Duplicated rather than imported:
|
|
152
|
+
* each source here is meant to be a small, self-contained file a reviewer
|
|
153
|
+
* can audit on its own (CONTRIBUTING.md).
|
|
154
|
+
*/
|
|
155
|
+
function isKindFollowingSymlink(fullPath, dirent, checkFn) {
|
|
156
|
+
if (checkFn(dirent)) return true;
|
|
157
|
+
if (!dirent.isSymbolicLink()) return false;
|
|
158
|
+
try { return checkFn(fs.statSync(fullPath)); } catch { return false; }
|
|
159
|
+
}
|
|
160
|
+
const isDirFollowingSymlink = (p, d) => isKindFollowingSymlink(p, d, (x) => x.isDirectory());
|
|
161
|
+
const isFileFollowingSymlink = (p, d) => isKindFollowingSymlink(p, d, (x) => x.isFile());
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* Walk one session-id directory (or one of its subdirectories) for
|
|
165
|
+
* candidate content files, up to MAX_SESSION_DEPTH levels deep. See the
|
|
166
|
+
* module docstring for why every `.json`/`.jsonl` file is read rather than
|
|
167
|
+
* an exact filename allowlist.
|
|
168
|
+
*/
|
|
169
|
+
function* walkSessionFiles(dir, depth) {
|
|
170
|
+
let entries;
|
|
171
|
+
try { entries = fs.readdirSync(dir, { withFileTypes: true }); }
|
|
172
|
+
catch { return; }
|
|
173
|
+
|
|
174
|
+
for (const e of entries) {
|
|
175
|
+
const full = path.join(dir, e.name);
|
|
176
|
+
|
|
177
|
+
if (isDirFollowingSymlink(full, e)) {
|
|
178
|
+
if (depth < MAX_SESSION_DEPTH) yield* walkSessionFiles(full, depth + 1);
|
|
179
|
+
continue;
|
|
180
|
+
}
|
|
181
|
+
if (isFileFollowingSymlink(full, e)) {
|
|
182
|
+
if (!SESSION_FILE_EXT.test(e.name)) continue;
|
|
183
|
+
let stat;
|
|
184
|
+
try { stat = fs.statSync(full); } catch { yield { file: full, broken: true }; continue; }
|
|
185
|
+
yield { file: full, mtimeMs: stat.mtimeMs, sizeBytes: stat.size, broken: false };
|
|
186
|
+
continue;
|
|
187
|
+
}
|
|
188
|
+
// Neither resolves as a directory nor a file: a dangling symlink is the
|
|
189
|
+
// one case worth reporting.
|
|
190
|
+
if (e.isSymbolicLink()) yield { file: full, broken: true };
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Yield { file, mtimeMs, sizeBytes, broken } for every Grok Build session
|
|
196
|
+
* content file found under `~/.grok/sessions/<encoded-cwd>/<session-id>/`.
|
|
197
|
+
*
|
|
198
|
+
* Every entry directly under `sessions/` is treated as a candidate
|
|
199
|
+
* encoded-cwd directory rather than trying to decode or validate the
|
|
200
|
+
* encoding scheme (percent-encoding vs. the long-path slug+hash fallback —
|
|
201
|
+
* see the module docstring) — this mirrors gemini-cli.js's and cursor.js's
|
|
202
|
+
* own choice to walk a similarly-shaped directory generically rather than
|
|
203
|
+
* recompute the source tool's own hashing/encoding.
|
|
204
|
+
*/
|
|
205
|
+
function* files() {
|
|
206
|
+
let cwdDirs;
|
|
207
|
+
try { cwdDirs = fs.readdirSync(SESSIONS_DIR, { withFileTypes: true }); }
|
|
208
|
+
catch { return; }
|
|
209
|
+
|
|
210
|
+
for (const cwdEnt of cwdDirs) {
|
|
211
|
+
const cwdDir = path.join(SESSIONS_DIR, cwdEnt.name);
|
|
212
|
+
if (!isDirFollowingSymlink(cwdDir, cwdEnt)) {
|
|
213
|
+
if (cwdEnt.isSymbolicLink()) yield { file: cwdDir, broken: true };
|
|
214
|
+
continue;
|
|
215
|
+
}
|
|
216
|
+
yield* walkSessionFiles(cwdDir, 0);
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
/**
|
|
221
|
+
* Read one transcript file as an array of raw text lines. Identical approach
|
|
222
|
+
* to claude-code.js's readLines() — streamed via readline/promises (not
|
|
223
|
+
* readFileSync+split, for the same V8 string-length-ceiling reason), a
|
|
224
|
+
* generous size cap, and a hard read timeout since Node's stream/readline
|
|
225
|
+
* stack has no built-in one. See claude-code.js's own docstring for the
|
|
226
|
+
* full reasoning; not re-derived here since nothing about it is
|
|
227
|
+
* Grok-Build-specific.
|
|
228
|
+
*
|
|
229
|
+
* Works the same whether `file` is JSONL (updates.jsonl, chat_history.jsonl,
|
|
230
|
+
* rewind_points.jsonl, feedback.jsonl — one record per line) or a plain
|
|
231
|
+
* multi-line JSON document (summary.json, plan.json, signals.json,
|
|
232
|
+
* subagents/*.json) — per the adapter contract, lines don't need to be
|
|
233
|
+
* valid JSON individually, they just need pattern-matching against.
|
|
234
|
+
*/
|
|
235
|
+
async function readLines(file) {
|
|
236
|
+
let stat;
|
|
237
|
+
try { stat = fs.statSync(file); }
|
|
238
|
+
catch { return { lines: [], status: "failed", bytesRead: 0 }; }
|
|
239
|
+
if (stat.size > MAX_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
|
|
240
|
+
|
|
241
|
+
const lines = [];
|
|
242
|
+
let bytesRead = 0;
|
|
243
|
+
const stream = fs.createReadStream(file, { encoding: "utf-8" });
|
|
244
|
+
const rl = createInterface({ input: stream, crlfDelay: Infinity });
|
|
245
|
+
|
|
246
|
+
const timer = setTimeout(() => stream.destroy(new Error("read timed out")), READ_TIMEOUT_MS);
|
|
247
|
+
|
|
248
|
+
try {
|
|
249
|
+
for await (const line of rl) {
|
|
250
|
+
lines.push(line);
|
|
251
|
+
bytesRead += Buffer.byteLength(line, "utf-8") + 1; // +1 for the stripped newline
|
|
252
|
+
}
|
|
253
|
+
return { lines, status: "complete", bytesRead };
|
|
254
|
+
} catch {
|
|
255
|
+
// Whatever WAS read before the failure is real content and may contain
|
|
256
|
+
// a real secret — discarding it because the file didn't finish cleanly
|
|
257
|
+
// would be a silent false negative, which is worse than an honest
|
|
258
|
+
// "partial" label.
|
|
259
|
+
return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
|
|
260
|
+
} finally {
|
|
261
|
+
clearTimeout(timer);
|
|
262
|
+
rl.close();
|
|
263
|
+
stream.destroy();
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
module.exports = { id, label, available, files, readLines };
|