residoo 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +225 -46
- package/SECURITY.md +29 -22
- package/package.json +1 -1
- package/src/cli.js +82 -16
- package/src/integrity.js +669 -0
- package/src/patterns.js +78 -5
- package/src/report.js +74 -7
- package/src/sources/agent-configs.js +308 -0
- package/src/sources/aider.js +361 -0
- package/src/sources/amazon-q.js +199 -0
- package/src/sources/antigravity-cli.js +155 -0
- package/src/sources/cline.js +208 -0
- package/src/sources/codebuff.js +295 -0
- package/src/sources/codex-cli.js +258 -0
- package/src/sources/cody.js +325 -0
- package/src/sources/continue.js +408 -0
- package/src/sources/copilot-chat.js +272 -0
- package/src/sources/copilot-cli.js +300 -0
- package/src/sources/crush.js +364 -0
- package/src/sources/cursor.js +374 -0
- package/src/sources/devin-cli.js +241 -0
- package/src/sources/factory-droid.js +153 -0
- package/src/sources/fx.js +136 -0
- package/src/sources/gemini-cli.js +242 -0
- package/src/sources/goose.js +366 -0
- package/src/sources/grok-cli.js +267 -0
- package/src/sources/hermes.js +282 -0
- package/src/sources/index.js +172 -8
- package/src/sources/jetbrains-ai-assistant.js +343 -0
- package/src/sources/jetbrains-junie.js +292 -0
- package/src/sources/kilo-code.js +430 -0
- package/src/sources/kimi-code.js +147 -0
- package/src/sources/kiro-cli.js +393 -0
- package/src/sources/kiro-ide.js +230 -0
- package/src/sources/llm.js +328 -0
- package/src/sources/mentat.js +143 -0
- package/src/sources/open-interpreter.js +224 -0
- package/src/sources/openclaw.js +218 -0
- package/src/sources/opencode.js +379 -0
- package/src/sources/openhands.js +181 -0
- package/src/sources/pearai.js +151 -0
- package/src/sources/pi-agent.js +130 -0
- package/src/sources/qodo-gen.js +189 -0
- package/src/sources/qwen-code.js +244 -0
- package/src/sources/roo-code.js +239 -0
- package/src/sources/trae.js +294 -0
- package/src/sources/void.js +273 -0
- package/src/sources/warp.js +395 -0
- package/src/sources/windsurf.js +256 -0
- package/src/sources/zed.js +374 -0
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
|
|
3
|
+
const fs = require("fs");
|
|
4
|
+
const path = require("path");
|
|
5
|
+
const os = require("os");
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Simon Willison's `llm` CLI (llm.datasette.io) — logs every prompt/response
|
|
9
|
+
* to a local SQLite database.
|
|
10
|
+
*
|
|
11
|
+
* VERIFICATION STATUS (read this before trusting anything below): this
|
|
12
|
+
* source is corroborated by multiple independent, official/primary sources,
|
|
13
|
+
* but was NOT checked against a real `llm` install — this machine doesn't
|
|
14
|
+
* have one (checked: no `llm` on PATH, no `io.datasette.llm` directory under
|
|
15
|
+
* `~/Library/Application Support`, `pip show llm` empty). Treat it the same
|
|
16
|
+
* way cursor.js's header asks you to treat that source: solid on paper,
|
|
17
|
+
* unconfirmed against real data. What was actually checked, on 2026-09-02:
|
|
18
|
+
*
|
|
19
|
+
* 1. Official docs — https://llm.datasette.io/en/stable/logging.html —
|
|
20
|
+
* states the default macOS path in a worked example
|
|
21
|
+
* (`/Users/simon/Library/Application Support/io.datasette.llm/logs.db`),
|
|
22
|
+
* documents `llm logs path` / `llm logs status`, and describes both the
|
|
23
|
+
* legacy schema (`conversations`, `responses`, ...) and the newer
|
|
24
|
+
* content-addressed schema (`threads`, `turns`, `messages`, `parts`,
|
|
25
|
+
* ...), explicitly noting the legacy tables stay read-only and `llm
|
|
26
|
+
* logs` merges both generations.
|
|
27
|
+
* 2. Primary source — the `llm` package's own code on GitHub, fetched
|
|
28
|
+
* directly (raw.githubusercontent.com/simonw/llm/main/...), not a
|
|
29
|
+
* summary of it:
|
|
30
|
+
* - `llm/__init__.py`, `user_dir()`:
|
|
31
|
+
* llm_user_path = os.environ.get("LLM_USER_PATH")
|
|
32
|
+
* if llm_user_path: path = pathlib.Path(llm_user_path)
|
|
33
|
+
* else: path = pathlib.Path(click.get_app_dir("io.datasette.llm"))
|
|
34
|
+
* - `llm/cli.py`, `logs_db_path()`: `return user_dir() / "logs.db"`
|
|
35
|
+
* i.e. the path is genuinely `<user_dir>/logs.db`, `LLM_USER_PATH`
|
|
36
|
+
* genuinely overrides it, and the app-id string genuinely is
|
|
37
|
+
* `io.datasette.llm` — not inferred, read verbatim from source.
|
|
38
|
+
* 3. Click's own source (pallets/click, `src/click/utils.py`,
|
|
39
|
+
* `get_app_dir()`) for what `click.get_app_dir("io.datasette.llm")`
|
|
40
|
+
* resolves to per OS — llm's own docs only ever show the macOS case, so
|
|
41
|
+
* the Linux/Windows branches below are derived from click's documented
|
|
42
|
+
* and implemented behavior, not from an llm-specific source. See
|
|
43
|
+
* llmDefaultUserDir() below for the exact per-OS logic mirrored from it.
|
|
44
|
+
* 4. A real user's own report — github.com/simonw/llm/issues/193 —
|
|
45
|
+
* independently corroborating the `.../Application Support/
|
|
46
|
+
* io.datasette.llm` directory from the install side (a bug about that
|
|
47
|
+
* directory not existing yet at first run), not just the docs page.
|
|
48
|
+
* 5. The project's own changelog, release 0.32rc1 (2026-07-30): the
|
|
49
|
+
* content-addressed schema is the NEW generation, added recently, with
|
|
50
|
+
* old `responses`/`conversations` data explicitly left in place and
|
|
51
|
+
* still readable — i.e. this schema has already changed shape once,
|
|
52
|
+
* which is exactly why readLines() below does not hardcode either
|
|
53
|
+
* generation's table names (see its docstring).
|
|
54
|
+
*
|
|
55
|
+
* That's genuine primary-source verification of the PATH (docs + the actual
|
|
56
|
+
* source lines that compute it + an independent bug report), which is a
|
|
57
|
+
* stronger basis than cursor.js had for its paths. What's still unverified
|
|
58
|
+
* is what a REAL logs.db, written by a real running `llm`, actually looks
|
|
59
|
+
* like on disk — table-by-table, row-by-row. readLines()'s dynamic
|
|
60
|
+
* table-introspection strategy (below) is the direct mitigation for that gap.
|
|
61
|
+
*/
|
|
62
|
+
function llmDefaultUserDir() {
|
|
63
|
+
const home = os.homedir();
|
|
64
|
+
if (process.platform === "win32") {
|
|
65
|
+
// click.get_app_dir(): WIN branch — os.environ.get("APPDATA"), falling
|
|
66
|
+
// back to the home directory if APPDATA is unset (roaming=True is
|
|
67
|
+
// click's default, which is what llm calls it with).
|
|
68
|
+
const appData = process.env.APPDATA || home;
|
|
69
|
+
return path.join(appData, "io.datasette.llm");
|
|
70
|
+
}
|
|
71
|
+
if (process.platform === "darwin") {
|
|
72
|
+
// click.get_app_dir(): darwin branch. Matches llm's own doc example.
|
|
73
|
+
return path.join(home, "Library", "Application Support", "io.datasette.llm");
|
|
74
|
+
}
|
|
75
|
+
// click.get_app_dir(): remaining POSIX branch (Linux and friends).
|
|
76
|
+
// _posixify("io.datasette.llm") is a no-op here — it only lowercases and
|
|
77
|
+
// joins on whitespace, and the app id has neither.
|
|
78
|
+
const xdgConfigHome = process.env.XDG_CONFIG_HOME || path.join(home, ".config");
|
|
79
|
+
return path.join(xdgConfigHome, "io.datasette.llm");
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// user_dir(), verbatim per llm/__init__.py: LLM_USER_PATH wins outright when
|
|
83
|
+
// set, otherwise the per-OS default above. Read once at module load — same
|
|
84
|
+
// convention cursor.js uses for its own env-derived paths.
|
|
85
|
+
const USER_DIR = process.env.LLM_USER_PATH || llmDefaultUserDir();
|
|
86
|
+
const LOGS_DB = path.join(USER_DIR, "logs.db");
|
|
87
|
+
|
|
88
|
+
const NODE_SQLITE_REQUIREMENT = "needs Node.js 22.5+ (node:sqlite not present in this runtime)";
|
|
89
|
+
let sqliteRequireAttempted = false;
|
|
90
|
+
let DatabaseSync = null;
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Same lazy-require discipline as cursor.js, and for the identical reason:
|
|
94
|
+
* index.js requires every source unconditionally and cli.js calls
|
|
95
|
+
* available() on all of them every run, so requiring node:sqlite eagerly
|
|
96
|
+
* would print Node's ExperimentalWarning on every invocation for every user,
|
|
97
|
+
* including the (large) majority who have never touched `llm`. Deferred
|
|
98
|
+
* until USER_DIR is confirmed to actually exist — see available() below.
|
|
99
|
+
*/
|
|
100
|
+
function getDatabaseSync() {
|
|
101
|
+
if (!sqliteRequireAttempted) {
|
|
102
|
+
sqliteRequireAttempted = true;
|
|
103
|
+
try { ({ DatabaseSync } = require("node:sqlite")); }
|
|
104
|
+
catch { DatabaseSync = null; }
|
|
105
|
+
}
|
|
106
|
+
return DatabaseSync;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function id() { return "llm"; }
|
|
110
|
+
function label() { return "LLM (Datasette)"; }
|
|
111
|
+
|
|
112
|
+
function userDirExists() {
|
|
113
|
+
try { return fs.statSync(USER_DIR).isDirectory(); } catch { return false; }
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
function available() {
|
|
117
|
+
// Cheap fs check first, short-circuiting getDatabaseSync() (and its
|
|
118
|
+
// possible warning) for the common case where `llm` was never installed —
|
|
119
|
+
// identical shape to cursor.js's available().
|
|
120
|
+
return userDirExists() && Boolean(getDatabaseSync());
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Optional, additive export — see cursor.js's own unavailableReason() for
|
|
125
|
+
* the full rationale. Only distinguishes the one case worth calling out:
|
|
126
|
+
* `llm`'s directory is really there but this Node runtime can't open SQLite.
|
|
127
|
+
*/
|
|
128
|
+
function unavailableReason() {
|
|
129
|
+
if (!userDirExists()) return null;
|
|
130
|
+
if (getDatabaseSync()) return null;
|
|
131
|
+
return `LLM (Datasette) detected but not scanned — ${NODE_SQLITE_REQUIREMENT}`;
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* Resolve the single logs.db candidate path into zero or one files() entries.
|
|
136
|
+
* Same convention as cursor.js's statIfPresent: lstat first so a symlink is
|
|
137
|
+
* detected and followed rather than silently treated as a plain file, a
|
|
138
|
+
* dangling symlink is reported broken rather than skipped, and a path that
|
|
139
|
+
* simply doesn't exist yet (llm installed but never run) yields nothing —
|
|
140
|
+
* that's normal, not broken.
|
|
141
|
+
*/
|
|
142
|
+
function* statIfPresent(dbPath) {
|
|
143
|
+
let lst;
|
|
144
|
+
try { lst = fs.lstatSync(dbPath); }
|
|
145
|
+
catch { return; }
|
|
146
|
+
|
|
147
|
+
if (lst.isSymbolicLink()) {
|
|
148
|
+
try {
|
|
149
|
+
const st = fs.statSync(dbPath); // follow the link
|
|
150
|
+
if (!st.isFile()) { yield { file: dbPath, broken: true }; return; }
|
|
151
|
+
yield { file: dbPath, mtimeMs: st.mtimeMs, sizeBytes: st.size, broken: false };
|
|
152
|
+
} catch {
|
|
153
|
+
yield { file: dbPath, broken: true }; // dangling symlink
|
|
154
|
+
}
|
|
155
|
+
return;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
if (!lst.isFile()) return; // something unexpected sits at this path — out of scope, not broken
|
|
159
|
+
yield { file: dbPath, mtimeMs: lst.mtimeMs, sizeBytes: lst.size, broken: false };
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Yield { file, mtimeMs, sizeBytes, broken } for logs.db, if present.
|
|
164
|
+
*
|
|
165
|
+
* Unlike cursor.js there is exactly one candidate path — `llm` keeps one
|
|
166
|
+
* user-wide database, not one per workspace — so this is a single
|
|
167
|
+
* statIfPresent() call. Pure filesystem work, no SQLite touched, so it works
|
|
168
|
+
* even on a Node runtime without node:sqlite (only readLines() needs that).
|
|
169
|
+
*/
|
|
170
|
+
function* files() {
|
|
171
|
+
yield* statIfPresent(LOGS_DB);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// Not backed by an observed real logs.db (no install to measure — see the
|
|
175
|
+
// header). Set generously above cursor.js's 512MB precedent because llm logs
|
|
176
|
+
// can embed attachments (e.g. base64 image/PDF bytes passed to multimodal
|
|
177
|
+
// prompts) directly in BLOB columns, which plausibly grows a heavy user's
|
|
178
|
+
// database well past pure-text chat history the way Cursor's UI-state DB
|
|
179
|
+
// never would. An honest guess, not a measurement.
|
|
180
|
+
const MAX_DB_BYTES = 1 * 1024 * 1024 * 1024; // 1GB
|
|
181
|
+
const READ_TIMEOUT_MS = 60_000;
|
|
182
|
+
const BUSY_TIMEOUT_MS = 5_000; // bound how long a read waits on a lock `llm` itself may be holding
|
|
183
|
+
const YIELD_EVERY_N_ROWS = 500;
|
|
184
|
+
|
|
185
|
+
// FTS5 virtual tables (llm's docs mention at least one, for full-text search
|
|
186
|
+
// over turns) create shadow tables alongside the virtual table itself, named
|
|
187
|
+
// `<table>_data`, `<table>_idx`, `<table>_docsize`, `<table>_config`, and
|
|
188
|
+
// sometimes `<table>_content`. These hold compressed/internal index segments,
|
|
189
|
+
// not distinct content — the same text is reachable through the virtual
|
|
190
|
+
// table itself (a plain `SELECT * FROM <fts-table>` works and returns the
|
|
191
|
+
// indexed text). Skipping the shadow tables avoids scanning opaque segment
|
|
192
|
+
// blobs for no benefit; if this pattern-match misses a shadow table under
|
|
193
|
+
// some future naming, it just gets queried like any other table — caught by
|
|
194
|
+
// the per-table try/catch below if that errors, or scanned as harmless extra
|
|
195
|
+
// noise if it doesn't. Nothing is ever silently dropped because of this list.
|
|
196
|
+
const FTS_SHADOW_SUFFIX_RE = /_(data|idx|docsize|config|content|content_rowid)$/;
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* A column value comes back from node:sqlite as a JS string (TEXT), number
|
|
200
|
+
* or bigint (INTEGER), null, or Uint8Array (BLOB) — same storage-class
|
|
201
|
+
* behavior cursor.js's valueToText() documents and relies on. Uint8Array is
|
|
202
|
+
* decoded as UTF-8 text (attachments are commonly base64-encoded JSON/text
|
|
203
|
+
* already, and even a genuinely binary blob just decodes to a harmless,
|
|
204
|
+
* unmatchable string rather than breaking JSON.stringify, which cannot
|
|
205
|
+
* serialize a Uint8Array or a bigint on its own).
|
|
206
|
+
*/
|
|
207
|
+
function rowToText(row) {
|
|
208
|
+
try {
|
|
209
|
+
return JSON.stringify(row, (_key, value) => {
|
|
210
|
+
if (value instanceof Uint8Array) return Buffer.from(value).toString("utf-8");
|
|
211
|
+
if (typeof value === "bigint") return value.toString();
|
|
212
|
+
return value;
|
|
213
|
+
});
|
|
214
|
+
} catch {
|
|
215
|
+
return null;
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Read logs.db as an array of raw text "lines" — one per database row,
|
|
221
|
+
* across every user table the file actually contains. Returns { lines,
|
|
222
|
+
* status, bytesRead } with the same status vocabulary claude-code.js and
|
|
223
|
+
* cursor.js use: "complete", "partial", "too-large", "failed".
|
|
224
|
+
*
|
|
225
|
+
* Table names are discovered at read time via `sqlite_master` rather than
|
|
226
|
+
* hardcoded, deliberately — llm's own changelog documents that its schema
|
|
227
|
+
* has already changed shape once (the 0.32rc1 rewrite from
|
|
228
|
+
* `conversations`/`responses` to `threads`/`turns`/`messages`/`parts`, old
|
|
229
|
+
* tables left in place but no longer written to). Hardcoding either
|
|
230
|
+
* generation's table list risks the exact failure this project refuses to
|
|
231
|
+
* ship: a schema that quietly drifts out from under a hardcoded assumption,
|
|
232
|
+
* with the mismatch swallowed instead of surfaced. Introspecting
|
|
233
|
+
* `sqlite_master` and reading whatever tables are actually there is correct
|
|
234
|
+
* against the legacy schema, the current schema, and whatever comes next,
|
|
235
|
+
* without needing to know which one a given logs.db is on.
|
|
236
|
+
*
|
|
237
|
+
* Same synchronous-native-call constraint as cursor.js's readLines() (no
|
|
238
|
+
* 'error'/'close' event, no AbortSignal to hang a preemptive timeout off of)
|
|
239
|
+
* — the same mitigation applies: iterate row-by-row, yield to the event loop
|
|
240
|
+
* and check a wall-clock deadline every YIELD_EVERY_N_ROWS rows, now across
|
|
241
|
+
* however many tables sqlite_master reports rather than a fixed two.
|
|
242
|
+
*/
|
|
243
|
+
async function readLines(file) {
|
|
244
|
+
const DB = getDatabaseSync();
|
|
245
|
+
if (!DB) return { lines: [], status: "failed", bytesRead: 0 };
|
|
246
|
+
|
|
247
|
+
let stat;
|
|
248
|
+
try { stat = fs.statSync(file); }
|
|
249
|
+
catch { return { lines: [], status: "failed", bytesRead: 0 }; }
|
|
250
|
+
if (stat.size > MAX_DB_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
|
|
251
|
+
|
|
252
|
+
let db;
|
|
253
|
+
try {
|
|
254
|
+
db = new DB(file, { readOnly: true });
|
|
255
|
+
db.exec(`PRAGMA busy_timeout = ${BUSY_TIMEOUT_MS}`);
|
|
256
|
+
} catch {
|
|
257
|
+
// File deleted between files() and this call, a corrupt/non-SQLite file
|
|
258
|
+
// at this path, or `llm` holding a lock this readonly open can't get
|
|
259
|
+
// past within BUSY_TIMEOUT_MS — all genuinely "could not read this."
|
|
260
|
+
return { lines: [], status: "failed", bytesRead: 0 };
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
let tableNames = [];
|
|
264
|
+
try {
|
|
265
|
+
const rows = db
|
|
266
|
+
.prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name NOT LIKE 'sqlite_%'")
|
|
267
|
+
.all();
|
|
268
|
+
tableNames = rows
|
|
269
|
+
.map((r) => r.name)
|
|
270
|
+
.filter((n) => typeof n === "string" && !FTS_SHADOW_SUFFIX_RE.test(n));
|
|
271
|
+
} catch {
|
|
272
|
+
try { db.close(); } catch { /* best-effort */ }
|
|
273
|
+
// Opened as SQLite but couldn't even list its own tables — treat like
|
|
274
|
+
// cursor.js treats "neither known table existed": a real failure to
|
|
275
|
+
// extract anything, not "extracted zero real rows."
|
|
276
|
+
return { lines: [], status: "failed", bytesRead: 0 };
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
const lines = [];
|
|
280
|
+
let bytesRead = 0;
|
|
281
|
+
const deadline = Date.now() + READ_TIMEOUT_MS;
|
|
282
|
+
let timedOut = false;
|
|
283
|
+
let sawError = false;
|
|
284
|
+
|
|
285
|
+
for (const table of tableNames) {
|
|
286
|
+
let rows;
|
|
287
|
+
try {
|
|
288
|
+
// Quoted identifier: table names come from sqlite_master itself, not
|
|
289
|
+
// user input, but quoting costs nothing and avoids any accidental
|
|
290
|
+
// reserved-word collision.
|
|
291
|
+
rows = db.prepare(`SELECT * FROM "${table}"`).iterate();
|
|
292
|
+
} catch {
|
|
293
|
+
// A shadow table this source's suffix filter didn't catch, or any
|
|
294
|
+
// other table this build of node:sqlite can't directly SELECT from —
|
|
295
|
+
// move on to the next table rather than aborting the whole file.
|
|
296
|
+
continue;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
let n = 0;
|
|
300
|
+
try {
|
|
301
|
+
for (const row of rows) {
|
|
302
|
+
const text = rowToText(row);
|
|
303
|
+
if (text) { lines.push(text); bytesRead += Buffer.byteLength(text, "utf-8"); }
|
|
304
|
+
n++;
|
|
305
|
+
if (n % YIELD_EVERY_N_ROWS === 0) {
|
|
306
|
+
await new Promise((resolve) => setImmediate(resolve));
|
|
307
|
+
if (Date.now() > deadline) { timedOut = true; break; }
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
} catch {
|
|
311
|
+
// A row iterator can itself throw partway (e.g. a corrupted page hit
|
|
312
|
+
// mid-scan) — whatever WAS read before that is real content, kept the
|
|
313
|
+
// same way claude-code.js and cursor.js keep a partial read rather
|
|
314
|
+
// than discarding it.
|
|
315
|
+
sawError = true;
|
|
316
|
+
}
|
|
317
|
+
if (timedOut) break;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
try { db.close(); } catch { /* best-effort close; nothing left to do if this fails */ }
|
|
321
|
+
|
|
322
|
+
if (tableNames.length === 0) return { lines: [], status: "failed", bytesRead: 0 };
|
|
323
|
+
if (sawError && lines.length === 0) return { lines: [], status: "failed", bytesRead };
|
|
324
|
+
if (timedOut || sawError) return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
|
|
325
|
+
return { lines, status: "complete", bytesRead };
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
module.exports = { id, label, available, unavailableReason, files, readLines };
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
|
|
3
|
+
const fs = require("fs");
|
|
4
|
+
const { createInterface } = require("readline/promises");
|
|
5
|
+
const path = require("path");
|
|
6
|
+
const os = require("os");
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Mentat (github.com/AbanteAI/mentat, originally "the AI coding assistant"
|
|
10
|
+
* from AbanteAI, predates most of this project's other sources) session
|
|
11
|
+
* transcripts.
|
|
12
|
+
*
|
|
13
|
+
* VERIFICATION STATUS: NOT checked against a real Mentat install — `mentat`
|
|
14
|
+
* is not on PATH and no `~/.mentat` directory exists on the machine this
|
|
15
|
+
* adapter was built on (checked PATH, pip3 show, pipx list, mdfind). Mentat
|
|
16
|
+
* itself is no longer maintained: its original repo now redirects to
|
|
17
|
+
* github.com/AbanteAI/archive-old-cli-mentat, archived 2025-01-07. It ships
|
|
18
|
+
* here anyway per CONTRIBUTING.md rule 3 — a tool being unmaintained doesn't
|
|
19
|
+
* make transcripts it already wrote to a real user's disk any less worth
|
|
20
|
+
* scanning — on the strength of the single strongest kind of source this
|
|
21
|
+
* project cites anywhere: the project's OWN final, real source code, read
|
|
22
|
+
* directly (not summarized, not a blog post), which settles the path and
|
|
23
|
+
* format questions unambiguously:
|
|
24
|
+
*
|
|
25
|
+
* - `mentat/utils.py`: `mentat_dir_path = Path.home() / ".mentat"`
|
|
26
|
+
* - `mentat/logging_config.py`: `logs_path = mentat_dir_path / "logs"`,
|
|
27
|
+
* and inside `setup_logging()` (run on every real session, skipped only
|
|
28
|
+
* when `is_test_environment()`):
|
|
29
|
+
* `transcripts_handler = logging.FileHandler(logs_path /
|
|
30
|
+
* f"transcript_{timestamp}.log")`
|
|
31
|
+
* with `timestamp = datetime.datetime.now().strftime("%Y%m%d_%H%M%S")`
|
|
32
|
+
* and the handler's formatter set to bare `"%(message)s")` — i.e. each
|
|
33
|
+
* logged line is written to the file verbatim, nothing prepended.
|
|
34
|
+
* - `mentat/transcripts.py`'s own `get_transcript_logs()` confirms the
|
|
35
|
+
* shape from the reader side: it globs `transcript_*` under `logs_path`,
|
|
36
|
+
* calls `f.readlines()`, and reconstructs a JSON array as
|
|
37
|
+
* `json.loads("[" + ", ".join(transcript) + "]")` — which only works if
|
|
38
|
+
* every physical line is already one complete, self-contained JSON
|
|
39
|
+
* object (a user turn, a model turn, or an agent-only side message; see
|
|
40
|
+
* the `TranscriptMessage`/`UserMessage`/`ModelMessage` TypedDicts in
|
|
41
|
+
* that same file). That is exactly the "one JSON object per line" shape
|
|
42
|
+
* this source reads — no different in spirit from Claude Code's own
|
|
43
|
+
* JSONL transcripts, just named `transcript_<timestamp>.log` instead of
|
|
44
|
+
* `<session-id>.jsonl`.
|
|
45
|
+
*
|
|
46
|
+
* Deliberately NOT scanned: `mentat_<timestamp>.log` (Mentat's general debug
|
|
47
|
+
* log, same directory, same timestamp) and `costs.log` — neither is the
|
|
48
|
+
* transcript log Mentat's own code defines as the conversation record; only
|
|
49
|
+
* `transcript_*.log` is what `get_transcript_logs()` itself reads back.
|
|
50
|
+
*/
|
|
51
|
+
const LOGS_DIR = path.join(os.homedir(), ".mentat", "logs");
|
|
52
|
+
const FILENAME_RE = /^transcript_.+\.log$/;
|
|
53
|
+
|
|
54
|
+
const MAX_BYTES = 2 * 1024 * 1024 * 1024; // 2GB — same generous headroom claude-code.js uses; not
|
|
55
|
+
// empirically tested against a real huge Mentat transcript.
|
|
56
|
+
const READ_TIMEOUT_MS = 60_000;
|
|
57
|
+
|
|
58
|
+
function id() { return "mentat"; }
|
|
59
|
+
function label() { return "Mentat"; }
|
|
60
|
+
|
|
61
|
+
function available() {
|
|
62
|
+
try { return fs.statSync(LOGS_DIR).isDirectory(); } catch { return false; }
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Same defensive symlink-following pattern as claude-code.js's
|
|
67
|
+
* isDirFollowingSymlink/isFileFollowingSymlink — duplicated rather than
|
|
68
|
+
* imported per this project's one-small-self-contained-file convention.
|
|
69
|
+
*/
|
|
70
|
+
function isKindFollowingSymlink(fullPath, dirent, checkFn) {
|
|
71
|
+
if (checkFn(dirent)) return true;
|
|
72
|
+
if (!dirent.isSymbolicLink()) return false;
|
|
73
|
+
try { return checkFn(fs.statSync(fullPath)); } catch { return false; }
|
|
74
|
+
}
|
|
75
|
+
const isFileFollowingSymlink = (p, d) => isKindFollowingSymlink(p, d, (x) => x.isFile());
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Yield { file, mtimeMs, sizeBytes, broken } for every transcript log found
|
|
79
|
+
* directly under ~/.mentat/logs (a flat directory — Mentat has no
|
|
80
|
+
* per-project nesting the way Claude Code does). `broken: true` marks a
|
|
81
|
+
* `transcript_*.log`-named entry that looked scannable but wasn't — chiefly
|
|
82
|
+
* a dangling symlink — never silently skipped, per CONTRIBUTING.md rule 5.
|
|
83
|
+
*/
|
|
84
|
+
function* files() {
|
|
85
|
+
let entries;
|
|
86
|
+
try { entries = fs.readdirSync(LOGS_DIR, { withFileTypes: true }); }
|
|
87
|
+
catch { return; }
|
|
88
|
+
|
|
89
|
+
for (const e of entries) {
|
|
90
|
+
if (!FILENAME_RE.test(e.name)) continue;
|
|
91
|
+
const file = path.join(LOGS_DIR, e.name);
|
|
92
|
+
if (!e.isFile()) {
|
|
93
|
+
const resolved = isFileFollowingSymlink(file, e);
|
|
94
|
+
if (!resolved) {
|
|
95
|
+
if (e.isSymbolicLink()) yield { file, broken: true };
|
|
96
|
+
continue;
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
let stat;
|
|
100
|
+
try { stat = fs.statSync(file); } catch { yield { file, broken: true }; continue; }
|
|
101
|
+
yield { file, mtimeMs: stat.mtimeMs, sizeBytes: stat.size, broken: false };
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Read one transcript log as an array of raw text lines — same
|
|
107
|
+
* streamed/bounded/timed-out shape as claude-code.js's readLines(), for the
|
|
108
|
+
* same reasons (see that file's docstring): line-by-line via
|
|
109
|
+
* readline/promises rather than readFileSync+split to avoid V8's
|
|
110
|
+
* single-string ceiling on a large file, a wall-clock destroy() timeout
|
|
111
|
+
* because nothing in Node's stream stack provides one natively, and a
|
|
112
|
+
* partial read's lines are kept and reported as "partial" rather than
|
|
113
|
+
* discarded, because a secret in content that WAS read is still a finding.
|
|
114
|
+
*/
|
|
115
|
+
async function readLines(file) {
|
|
116
|
+
let stat;
|
|
117
|
+
try { stat = fs.statSync(file); }
|
|
118
|
+
catch { return { lines: [], status: "failed", bytesRead: 0 }; }
|
|
119
|
+
if (stat.size > MAX_BYTES) return { lines: [], status: "too-large", bytesRead: 0 };
|
|
120
|
+
|
|
121
|
+
const lines = [];
|
|
122
|
+
let bytesRead = 0;
|
|
123
|
+
const stream = fs.createReadStream(file, { encoding: "utf-8" });
|
|
124
|
+
const rl = createInterface({ input: stream, crlfDelay: Infinity });
|
|
125
|
+
|
|
126
|
+
const timer = setTimeout(() => stream.destroy(new Error("read timed out")), READ_TIMEOUT_MS);
|
|
127
|
+
|
|
128
|
+
try {
|
|
129
|
+
for await (const line of rl) {
|
|
130
|
+
lines.push(line);
|
|
131
|
+
bytesRead += Buffer.byteLength(line, "utf-8") + 1;
|
|
132
|
+
}
|
|
133
|
+
return { lines, status: "complete", bytesRead };
|
|
134
|
+
} catch {
|
|
135
|
+
return { lines, status: lines.length > 0 ? "partial" : "failed", bytesRead };
|
|
136
|
+
} finally {
|
|
137
|
+
clearTimeout(timer);
|
|
138
|
+
rl.close();
|
|
139
|
+
stream.destroy();
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
module.exports = { id, label, available, files, readLines };
|