@awebai/oats 0.22.17 → 0.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +7 -2
  2. package/bin/oats.mjs +365 -31
  3. package/docs/configuration.md +65 -0
  4. package/docs/design/2026-09-07-mobile-agent-management-proposal.md +228 -0
  5. package/docs/design/2026-09-08-expert-assisted-deployment-proposal.md +558 -0
  6. package/docs/design/2026-09-13-knowledge-and-memory-direction.md +744 -0
  7. package/docs/design/2026-09-13-knowledge-implementation.md +127 -0
  8. package/docs/design/2026-09-13-knowledge-location-contract.md +340 -0
  9. package/docs/design/launch-configurations.md +164 -0
  10. package/docs/design/package-runtime-api.md +177 -3
  11. package/docs/desktop-cli-api.md +68 -2
  12. package/docs/desktop-instance-start.md +39 -3
  13. package/docs/execution-targets.md +16 -0
  14. package/docs/knowledge-capability-authoring.md +98 -0
  15. package/docs/knowledge-reference/acceptance.md +108 -0
  16. package/docs/knowledge-reference/adoption.md +61 -0
  17. package/docs/knowledge-reference/harvester.md +107 -0
  18. package/docs/knowledge-reference/model.md +84 -0
  19. package/docs/knowledge-reference/package-craft.md +126 -0
  20. package/docs/knowledge-reference/provider-mapping.md +77 -0
  21. package/docs/knowledge-reference/reader-capture.md +87 -0
  22. package/docs/knowledge-theory.md +20 -6
  23. package/docs/layers.md +8 -7
  24. package/docs/oats-config.schema.json +33 -2
  25. package/docs/release-notes/v0.22.18.md +101 -0
  26. package/docs/release-notes/v0.22.19.md +115 -0
  27. package/docs/release-notes/v0.23.0.md +93 -0
  28. package/docs/souls-and-instances.md +18 -1
  29. package/injects/work-directory.md +18 -0
  30. package/lib/core.mjs +1109 -187
  31. package/lib/schedule.mjs +12 -2
  32. package/lib/servers.mjs +89 -4
  33. package/package.json +2 -2
  34. package/packages/record/README.md +19 -0
  35. package/packages/record/bin/capture.mjs +144 -53
  36. package/packages/record/bin/recall.mjs +17 -11
  37. package/packages/record/bin/record-native-start.mjs +11 -0
  38. package/packages/record/lib/capture-cc.mjs +82 -27
  39. package/packages/record/lib/capture-lock.mjs +81 -5
  40. package/packages/record/lib/formats.mjs +108 -21
  41. package/packages/record/lib/native-history.mjs +87 -0
  42. package/packages/record/lib/session-roots.mjs +90 -0
  43. package/packages/record/lib/session-snapshot.mjs +61 -0
  44. package/packages/record/lib/sessions-for-home.mjs +88 -56
  45. package/skills/oats/SKILL.md +3 -1
@@ -20,12 +20,14 @@
20
20
  // ignore.mjs) are skipped before being opened: no turn, no offset entry —
21
21
  // un-ignoring a file later makes the next pass capture it normally.
22
22
 
23
- import { closeSync, mkdirSync, openSync, readFileSync, readSync, statSync, writeFileSync } from "node:fs";
23
+ import { isUtf8 } from "node:buffer";
24
+ import { closeSync, fstatSync, mkdirSync, openSync, readFileSync, readSync, statSync, writeFileSync } from "node:fs";
24
25
  import { basename, dirname, join } from "node:path";
25
26
 
26
27
  import { finishTurn } from "./canonical.mjs";
27
28
  import { jsonlLines, SESSION_FORMATS } from "./formats.mjs";
28
29
  import { loadIgnore } from "./ignore.mjs";
30
+ import { assertIdentity, digest, identity, readRange, verifySnapshot } from "./session-snapshot.mjs";
29
31
 
30
32
  export const SESSION_STREAM_SOURCE = "cc";
31
33
 
@@ -49,7 +51,7 @@ export function scanTranscript(bytes) {
49
51
  if (text === null) continue;
50
52
  try {
51
53
  const d = JSON.parse(text);
52
- if (typeof d.timestamp === "string") ts = d.timestamp;
54
+ if (typeof d?.timestamp === "string") ts = d.timestamp;
53
55
  } catch {
54
56
  /* verbatim content; nothing to extract */
55
57
  }
@@ -83,8 +85,11 @@ function offsetsPath(store) {
83
85
  function loadOffsets(store) {
84
86
  try {
85
87
  return JSON.parse(readFileSync(offsetsPath(store), "utf8"));
86
- } catch {
87
- return {};
88
+ } catch (err) {
89
+ // This cache is derived, so malformed JSON can be rebuilt. An I/O
90
+ // failure is different: never hide an unreadable cache as missing.
91
+ if (err.code === "ENOENT" || err instanceof SyntaxError) return {};
92
+ throw err;
88
93
  }
89
94
  }
90
95
 
@@ -102,7 +107,7 @@ function readFrom(path, start, size) {
102
107
  let done = 0;
103
108
  while (done < buf.length) {
104
109
  const n = readSync(fd, buf, done, buf.length - done, start + done);
105
- if (n === 0) break;
110
+ if (n === 0) throw new Error(`short read of session/journal source: ${path}`);
106
111
  done += n;
107
112
  }
108
113
  return buf.subarray(0, done);
@@ -119,8 +124,9 @@ function lastJournalLine(store, streamId) {
119
124
  let size;
120
125
  try {
121
126
  size = statSync(path).size;
122
- } catch {
123
- return 0;
127
+ } catch (err) {
128
+ if (err.code === "ENOENT") return 0;
129
+ throw err;
124
130
  }
125
131
  let window = 64 * 1024;
126
132
  while (true) {
@@ -152,12 +158,12 @@ function lastJournalLine(store, streamId) {
152
158
  // Honest limit: an in-place REWRITE of already-captured lines is not
153
159
  // detected (only growth is; a shrink triggers a rescan via the size
154
160
  // check in the caller). Transcript writers are append-only in practice.
155
- function offsetFromJournal(store, streamId, sourcePath) {
161
+ function offsetFromJournal(store, streamId, sourcePath, final, sourceBytes) {
156
162
  const turns = store.readStream(streamId);
157
163
  if (turns.length === 0) return { bytes: 0, line: 0, lastTs: "" };
158
164
  const last = turns[turns.length - 1];
159
165
  const lastLine = last.provenance?.origin?.line ?? 0;
160
- const bytes = readFileSync(sourcePath);
166
+ const bytes = sourceBytes ?? readFileSync(sourcePath);
161
167
  let line = 0;
162
168
  let offset = 0;
163
169
  while (line < lastLine && offset < bytes.length) {
@@ -166,6 +172,7 @@ function offsetFromJournal(store, streamId, sourcePath) {
166
172
  line++;
167
173
  offset = nl + 1;
168
174
  }
175
+ if (final && line < lastLine) throw new Error(`session source is shorter than its captured journal: ${sourcePath}`);
169
176
  return { bytes: offset, line, lastTs: last.ts ?? "" };
170
177
  }
171
178
 
@@ -173,7 +180,14 @@ function offsetFromJournal(store, streamId, sourcePath) {
173
180
  // of every session file under `roots`. Unstamped leading lines are held
174
181
  // until the file shows its first timestamp (then they carry it forward),
175
182
  // so every turn is stamped and ts stays a pure function of the source.
176
- export function captureSessions(store, { owner, roots, format = "cc", ignore = null }) {
183
+ // `files` accepts discovery entries (including their attribution snapshot) or
184
+ // explicit paths for callers not making home-attribution claims. It pins a set,
185
+ // rather than rescanning directories
186
+ // and silently losing disappeared sources (or sweeping in other homes).
187
+ // `final` also verifies unchanged offsets against journals and checks source
188
+ // stability through the pass. The caller must quiesce writers for retirement;
189
+ // a performed pass is a snapshot, not a promise about future writes.
190
+ export function captureSessions(store, { owner, roots, files, format = "cc", ignore = null, final = false }) {
177
191
  const fmt = SESSION_FORMATS[format];
178
192
  if (!fmt) throw new Error(`unknown session format ${format}`);
179
193
  const ign = ignore ?? loadIgnore(store.root);
@@ -183,21 +197,35 @@ export function captureSessions(store, { owner, roots, format = "cc", ignore = n
183
197
  let unchanged = 0;
184
198
  let held = 0;
185
199
  let ignored = 0;
200
+ let incomplete = 0;
201
+ const issues = []; // source metadata only, never native record contents
186
202
  const streams = new Set();
187
203
 
188
- for (const path of fmt.listFiles(roots)) {
204
+ for (const file of files ?? fmt.listFiles(roots)) {
205
+ const path = typeof file === "string" ? file : file.path;
206
+ const expected = typeof file === "string" ? null : file.snapshot;
189
207
  sessions++;
190
208
  const sessionId = fmt.sessionId(path);
191
- if (ign.ignores(path, [basename(path), sessionId])) {
209
+ if (ign.ignores(path, [basename(path), sessionId, ...(fmt.ignoreKeys?.(path) ?? [])])) {
192
210
  ignored++;
193
211
  continue; // never opened: nothing stored, nothing remembered
194
212
  }
195
- let stat;
213
+ const fd = openSync(path, "r");
196
214
  try {
197
- stat = statSync(path);
198
- } catch {
199
- continue; // vanished between listing and stat; next pass catches it
215
+ const stat = fstatSync(fd);
216
+ const snapshot = identity(stat);
217
+ if (expected) assertIdentity(stat, expected, path); // BEFORE reading bytes
218
+ // Final home capture stages a descriptor-pinned snapshot. All attribution
219
+ // and stability checks precede the first append, never a post-write alarm.
220
+ const sourceBytes = final || expected ? readRange(fd, 0, stat.size, path) : undefined;
221
+ if (expected && digest(sourceBytes.subarray(0, expected.size)) !== expected.hash) {
222
+ throw new Error(`session source content changed since attribution: ${path}`);
200
223
  }
224
+ if (sourceBytes) snapshot.hash = digest(sourceBytes);
225
+ const verifySource = () => {
226
+ if (sourceBytes) verifySnapshot(fd, path, snapshot);
227
+ };
228
+ verifySource();
201
229
  const streamId = `${owner}~${fmt.source}.${sessionId}`;
202
230
  // Keyed by stream, not by source path: the same source captured under
203
231
  // two owners must not share offset state (owner is part of the stream).
@@ -210,31 +238,44 @@ export function captureSessions(store, { owner, roots, format = "cc", ignore = n
210
238
  // saved offsets for appends that landed in an unlinked inode (skipping
211
239
  // would lose lines — this happened during the live migration). The
212
240
  // journal is the truth; before appending anything, any disagreement
213
- // rebuilds the offset from it (checked only when the source grew:
214
- // an unchanged file appends nothing, so its cache cannot mislead).
215
- if (state && stat.size > state.bytes && state.line !== lastJournalLine(store, streamId)) {
241
+ // rebuilds the offset from it. Background passes check on growth;
242
+ // final passes also verify unchanged files before confirming capture.
243
+ if (state && (final || stat.size > state.bytes) && state.line !== lastJournalLine(store, streamId)) {
216
244
  state = null;
217
245
  }
218
- if (!state) state = offsetFromJournal(store, streamId, path);
246
+ if (!state) state = offsetFromJournal(store, streamId, path, final, sourceBytes);
219
247
  if (stat.size <= state.bytes) {
220
248
  unchanged++;
221
249
  offsets[offKey] = state;
250
+ verifySource();
222
251
  continue;
223
252
  }
224
253
 
225
- const chunk = readFrom(path, state.bytes, stat.size);
254
+ const chunk = sourceBytes ? sourceBytes.subarray(state.bytes) : readRange(fd, state.bytes, stat.size, path);
226
255
  // Phase 1: collect the COMPLETE lines of the chunk with their stamps.
227
256
  const lines = [];
228
257
  let scanned = 0;
229
- for (const { text } of jsonlLines(chunk)) {
230
- if (text === null) break; // over-string-limit line: retry later
231
- const lineBytes = Buffer.byteLength(text, "utf8") + 1;
232
- if (scanned + lineBytes > chunk.length) break; // no trailing newline yet
258
+ let reason;
259
+ while (scanned < chunk.length) {
260
+ const nl = chunk.indexOf(10, scanned);
261
+ if (nl === -1) { reason = "torn-tail"; break; }
262
+ const bytes = chunk.subarray(scanned, nl);
263
+ // Decoding replacement characters would change both the verbatim line
264
+ // and its byte offset, possibly treating a later fragment as a record.
265
+ if (!isUtf8(bytes)) { reason = "invalid-utf8"; break; }
266
+ let text;
267
+ try { text = bytes.toString("utf8"); }
268
+ catch (err) {
269
+ if (err.code !== "ERR_STRING_TOO_LONG") throw err;
270
+ reason = "oversized-line";
271
+ break;
272
+ }
273
+ const lineBytes = nl - scanned + 1;
233
274
  let ts = "";
234
275
  if (text.trim() !== "") {
235
276
  try {
236
277
  const d = JSON.parse(text);
237
- if (typeof d.timestamp === "string") ts = d.timestamp;
278
+ if (typeof d?.timestamp === "string") ts = d.timestamp;
238
279
  } catch {
239
280
  /* unparseable native line: captured verbatim below */
240
281
  }
@@ -242,6 +283,10 @@ export function captureSessions(store, { owner, roots, format = "cc", ignore = n
242
283
  lines.push({ text, ts, bytes: lineBytes });
243
284
  scanned += lineBytes;
244
285
  }
286
+ if (reason) {
287
+ incomplete++;
288
+ issues.push({ source: fmt.source, path, reason, offset: state.bytes + scanned });
289
+ }
245
290
  // Phase 2: every turn needs a stamp. Leading lines before the file's
246
291
  // first stamp carry it backward (deterministic: the file's first
247
292
  // stamp is invariant however capture is scheduled); if the file has
@@ -250,11 +295,16 @@ export function captureSessions(store, { owner, roots, format = "cc", ignore = n
250
295
  if (!lastTs) {
251
296
  const first = lines.find((l) => l.ts);
252
297
  if (!first) {
253
- if (lines.length > 0) held++;
298
+ if (lines.length > 0) {
299
+ held++;
300
+ issues.push({ source: fmt.source, path, reason: "unstamped", offset: state.bytes });
301
+ }
302
+ verifySource();
254
303
  continue; // do not advance; retry when a stamp exists
255
304
  }
256
305
  lastTs = first.ts;
257
306
  }
307
+ verifySource(); // parsing/recovery may take time; still no journal write yet
258
308
  // Turns flush to the journal in bounded batches, so memory stays flat
259
309
  // however large the backlog (a first capture of a huge transcript is
260
310
  // one file's worth of NEW lines). A crash between flushes cannot
@@ -295,6 +345,8 @@ export function captureSessions(store, { owner, roots, format = "cc", ignore = n
295
345
  // re-appends duplicate turn lines — logically deduped by id, but
296
346
  // wasted append-only bytes).
297
347
  if (grew) saveOffsets(store, offsets);
348
+ verifySource();
349
+ } finally { closeSync(fd); }
298
350
  }
299
351
  saveOffsets(store, offsets);
300
352
  return {
@@ -303,6 +355,9 @@ export function captureSessions(store, { owner, roots, format = "cc", ignore = n
303
355
  unchanged,
304
356
  held,
305
357
  ignored,
358
+ incomplete,
359
+ complete: held === 0 && incomplete === 0,
360
+ issues,
306
361
  streams: streams.size,
307
362
  stream: `${owner}~${fmt.source}.*`,
308
363
  };
@@ -12,7 +12,8 @@
12
12
  // message says exactly that. (A reclaim protocol was reviewed and rejected:
13
13
  // rename is not compare-and-swap, and stealing from a stalled live
14
14
  // initializer under memory pressure is the failure we are preventing.)
15
- import { mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs";
15
+ import { closeSync, existsSync, fstatSync, lstatSync, mkdirSync, openSync, readFileSync, rmSync, writeFileSync } from "node:fs";
16
+ import { randomBytes } from "node:crypto";
16
17
  import { join } from "node:path";
17
18
 
18
19
  export function captureLockPath(root) { return join(root, ".capture.lock"); }
@@ -41,8 +42,28 @@ export function recoveryInstruction(dir, owner, liveness) {
41
42
 
42
43
  /** Try to take the root's capture lock. Returns { path, release } when
43
44
  * taken, or { path, held: { pid, startedAt, liveness, recovery } } when any
44
- * lock exists. Never removes a lock it did not create. */
45
- export function acquireCaptureLock(root, { now = Date.now, pid = process.pid, liveness = holderLiveness } = {}) {
45
+ * lock exists. Never removes a lock it did not create.
46
+ *
47
+ * Two failure points are reported rather than left behind. If the owner
48
+ * record cannot be written after THIS call created the directory (a full
49
+ * disk, say), the directory is removed only while it is still the very
50
+ * directory this call created (same inode) and carries no other owner's
51
+ * record; a replacement that appeared meanwhile (operator recovery, then a
52
+ * newer pass) is left alone. The original error is rethrown with
53
+ * `lockCleanup: { path, removed, reason?, owner?, error?, recovery? }`
54
+ * saying what happened. And `release()` never throws: it answers
55
+ * `{ released: true }` only when the lock is verifiably gone, otherwise
56
+ * `{ released: false, reason: "gone" | "unknown-owner" | "not-owner" |
57
+ * "remove-failed", ... }` with the actual observation, never a guess.
58
+ *
59
+ * The owner record carries a per-acquisition nonce, so a release kept from
60
+ * an earlier acquisition cannot erase a later one by the same pid (an
61
+ * operator recovery followed by a new pass in the same long-lived process).
62
+ * That is ownership checking; no lock is ever reclaimed.
63
+ *
64
+ * `io` exists for fault injection in tests only. */
65
+ export function acquireCaptureLock(root, { now = Date.now, pid = process.pid, liveness = holderLiveness, io = {} } = {}) {
66
+ const fs = { writeFileSync, rmSync, openSync, closeSync, lstatSync, ...io };
46
67
  const dir = captureLockPath(root);
47
68
  mkdirSync(root, { recursive: true }); // the store creates the root lazily; the lock may come first
48
69
  try {
@@ -53,12 +74,67 @@ export function acquireCaptureLock(root, { now = Date.now, pid = process.pid, li
53
74
  const live = owner ? (owner.pid === pid ? "alive" : liveness(owner.pid)) : "unknown";
54
75
  return { path: dir, held: { pid: owner?.pid, startedAt: owner?.startedAt, liveness: live, recovery: recoveryInstruction(dir, owner, live) } };
55
76
  }
56
- writeFileSync(join(dir, "owner.json"), JSON.stringify({ pid, startedAt: new Date(now()).toISOString() }));
77
+ const nonce = randomBytes(8).toString("hex");
78
+ let directoryFd, identity;
79
+ try {
80
+ // Keep the directory alive until initialization or its cleanup finishes.
81
+ // Otherwise Linux can reuse its inode immediately after an unlink, making
82
+ // a record-less replacement look like the directory we created.
83
+ directoryFd = fs.openSync(dir, "r");
84
+ identity = fstatSync(directoryFd);
85
+ fs.writeFileSync(join(dir, "owner.json"), JSON.stringify({ pid, nonce, startedAt: new Date(now()).toISOString() }));
86
+ } catch (err) {
87
+ // Ownership was proven by the mkdir, not by the moment of cleanup: the
88
+ // directory is removed only if it is still ours (same inode) and holds
89
+ // no other pass's record. Anything else is reported, not deleted.
90
+ let cleanupError, reason;
91
+ const cur = readOwner(dir);
92
+ const same = (() => { try { const current = lstatSync(dir); return identity && current.dev === identity.dev && current.ino === identity.ino; } catch { return false; } })();
93
+ if (!existsSync(dir)) reason = "gone";
94
+ else if (!identity) reason = "unverified";
95
+ else if (!same || (cur && (cur.pid !== pid || cur.nonce !== nonce))) reason = "replaced";
96
+ else { try { fs.rmSync(dir, { recursive: true, force: true }); } catch (e2) { cleanupError = e2; } }
97
+ const removed = reason === undefined && !existsSync(dir);
98
+ err.lockCleanup = {
99
+ path: dir,
100
+ removed,
101
+ ...(reason ? { reason } : {}),
102
+ ...(reason === "replaced" && cur ? { owner: cur } : {}),
103
+ ...(cleanupError ? { error: cleanupError.message } : {}),
104
+ ...(removed || reason === "gone" || reason === "replaced" ? {} : { recovery: recoveryInstruction(dir, undefined, "unknown") }),
105
+ };
106
+ throw err;
107
+ } finally {
108
+ if (directoryFd !== undefined) fs.closeSync(directoryFd);
109
+ }
57
110
  return {
58
111
  path: dir,
59
112
  release: () => {
113
+ if (!existsSync(dir)) return { released: false, reason: "gone" };
60
114
  const cur = readOwner(dir);
61
- if (cur && cur.pid === pid) { try { rmSync(dir, { recursive: true, force: true }); } catch { /* already gone */ } }
115
+ if (!cur) return { released: false, reason: "unknown-owner", recovery: recoveryInstruction(dir, undefined, "unknown") };
116
+ if (cur.pid !== pid || cur.nonce !== nonce) return { released: false, reason: "not-owner", owner: cur };
117
+ let error;
118
+ try { fs.rmSync(dir, { recursive: true, force: true }); } catch (e) { error = e; }
119
+ let remains;
120
+ try { fs.lstatSync(dir); remains = true; }
121
+ catch (err) {
122
+ if (err.code === "ENOENT") remains = false;
123
+ else return { released: false, reason: "remove-failed", error: err.message,
124
+ recovery: "could not verify capture lock removal; inspect the filesystem error and rerun capture" };
125
+ }
126
+ if (!remains) {
127
+ // A removal may throw after changing the filesystem. Absence does
128
+ // not erase that failure from a final-capture receipt.
129
+ if (error) return { released: false, reason: "remove-failed", error: error.message,
130
+ recovery: "the lock is now absent, but removal reported an error; inspect the failure and rerun capture" };
131
+ return { released: true };
132
+ }
133
+ // Our own lock could not be removed. We are the holder and, as far as
134
+ // this process can tell, alive; the operator gets a conditional line.
135
+ const live = pid === process.pid ? "alive" : liveness(pid);
136
+ const recovery = `${dir} is still held by pid ${pid} (this pass, ${live} when it reported this); its removal failed${error ? ` (${error.message})` : ""}; once that process has exited (ps -p ${pid}), remove the lock with: rm -r -- ${shellQuote(dir)} and rerun`;
137
+ return { released: false, reason: "remove-failed", liveness: live, ...(error ? { error: error.message } : {}), recovery };
62
138
  },
63
139
  };
64
140
  }
@@ -16,9 +16,10 @@
16
16
  // snapshots, queue operations, mode flips) yields no docs — but its
17
17
  // native line is still stored verbatim in the turn, so nothing is lost.
18
18
 
19
- import { existsSync, readdirSync } from "node:fs";
20
- import { basename, join } from "node:path";
19
+ import { lstatSync, readdirSync, statSync } from "node:fs";
20
+ import { basename, dirname, join } from "node:path";
21
21
  import { homedir } from "node:os";
22
+ import { nativeDirectory } from "./session-roots.mjs";
22
23
 
23
24
  // Iterate JSONL lines of a buffer without materializing the whole file as
24
25
  // one string — real transcripts reach hundreds of MB (a 789 MB Codex
@@ -53,21 +54,24 @@ function* parsedLines(bytes) {
53
54
  }
54
55
  }
55
56
 
56
- function listJsonlFiles(root, maxDepth) {
57
+ function listJsonlFiles(root, maxDepth, { strict = false } = {}) {
57
58
  const out = [];
58
59
  const walk = (dir, depth) => {
59
60
  let names;
60
61
  try {
61
62
  names = readdirSync(dir, { withFileTypes: true });
62
- } catch {
63
- return;
63
+ } catch (err) {
64
+ // An optional, absent root is normal; an unreadable directory or a
65
+ // subtree that disappeared during discovery is not an empty scan.
66
+ if (!strict && depth === 0 && err.code === "ENOENT") return;
67
+ throw err;
64
68
  }
65
69
  for (const entry of names.sort((a, b) => a.name.localeCompare(b.name))) {
66
70
  const path = join(dir, entry.name);
67
- if (entry.isDirectory()) {
71
+ if (entry.name.endsWith(".jsonl") && !entry.isDirectory()) {
72
+ out.push(path); // privacy rules precede opening/statting source links
73
+ } else if (entry.isDirectory() || (entry.isSymbolicLink() && statSync(path).isDirectory())) {
68
74
  if (depth < maxDepth) walk(path, depth + 1);
69
- } else if (entry.name.endsWith(".jsonl")) {
70
- out.push(path);
71
75
  }
72
76
  }
73
77
  };
@@ -75,18 +79,97 @@ function listJsonlFiles(root, maxDepth) {
75
79
  return out;
76
80
  }
77
81
 
82
+ // Only absence makes a default root optional. existsSync also hides access
83
+ // failures, which would turn an unperformed scan into a false empty result.
84
+ function directoryExists(path) {
85
+ try {
86
+ if (!statSync(path).isDirectory()) throw new Error(`session root is not a directory: ${path}`);
87
+ return true;
88
+ } catch (err) {
89
+ if (err.code === "ENOENT") {
90
+ // ENOENT through a dangling directory link is not an absent runtime.
91
+ for (let part = path; ; part = dirname(part)) {
92
+ try {
93
+ if (lstatSync(part).isSymbolicLink()) throw new Error(`unresolvable session root: ${path}`);
94
+ break;
95
+ } catch (e) { if (e.code !== "ENOENT") throw e; }
96
+ if (dirname(part) === part) break;
97
+ }
98
+ return false;
99
+ }
100
+ throw err;
101
+ }
102
+ }
103
+
78
104
  // ------------------------------------------------------------ claude code
79
105
 
80
- function ccRoots(home = homedir()) {
106
+ function configuredRoot(value, suffix, options) {
107
+ const root = join(nativeDirectory(value, options), suffix);
108
+ // Explicitly relocated storage is not an optional absent default. A
109
+ // missing/unreadable root means we cannot certify its evidence inventory.
110
+ if (!directoryExists(root)) throw new Error(`configured session root does not exist: ${root}`);
111
+ return [root];
112
+ }
113
+
114
+ function ccRoots(home = homedir(), env = process.env, options = {}) {
115
+ if (env.CLAUDE_CONFIG_DIR) return configuredRoot(env.CLAUDE_CONFIG_DIR, "projects", { home, ...options });
81
116
  const roots = [];
82
- for (const name of readdirSync(home).sort()) {
83
- if (!name.startsWith(".claude")) continue;
84
- const projects = join(home, name, "projects");
85
- if (existsSync(projects)) roots.push(projects);
117
+ for (const entry of readdirSync(home, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name))) {
118
+ if (!entry.name.startsWith(".claude")) continue;
119
+ // ~/.claude.json and its backups are normal config FILES, not roots.
120
+ if (!entry.isDirectory() && !entry.isSymbolicLink()) continue;
121
+ if (entry.isSymbolicLink() && !statSync(join(home, entry.name)).isDirectory()) continue;
122
+ const projects = join(home, entry.name, "projects");
123
+ if (directoryExists(projects)) roots.push(projects);
86
124
  }
87
125
  return roots;
88
126
  }
89
127
 
128
+ // Native Claude child transcripts live at
129
+ // projects/<project>/<sessionId>/subagents/agent-*.jsonl, not beside the
130
+ // parent's file. Enumerate only this layout, never arbitrary project files.
131
+ function listCcFiles(root, { strict = false } = {}) {
132
+ const out = [];
133
+ const entries = (dir, optional = false) => {
134
+ try { return readdirSync(dir, { withFileTypes: true }).sort((a, b) => a.name.localeCompare(b.name)); }
135
+ catch (err) { if (optional && err.code === "ENOENT") return []; throw err; }
136
+ };
137
+ const directory = (entry, path) => entry.isDirectory() || (entry.isSymbolicLink() && !entry.name.endsWith(".jsonl") && statSync(path).isDirectory());
138
+ for (const project of entries(root, !strict)) {
139
+ const projectPath = join(root, project.name);
140
+ if (!directory(project, projectPath)) {
141
+ if (project.name.endsWith(".jsonl")) out.push(projectPath);
142
+ continue;
143
+ }
144
+ for (const entry of entries(projectPath)) {
145
+ const path = join(projectPath, entry.name);
146
+ if (!directory(entry, path)) {
147
+ if (entry.name.endsWith(".jsonl")) out.push(path);
148
+ continue;
149
+ }
150
+ // Listing the session directory (rather than treating ENOENT at an
151
+ // assumed subagents path as optional) preserves disappearance errors.
152
+ for (const childDir of entries(path)) {
153
+ if (childDir.name !== "subagents") continue;
154
+ const children = join(path, childDir.name);
155
+ if (!directory(childDir, children)) throw new Error(`session subagents root is not a directory: ${children}`);
156
+ for (const child of entries(children)) {
157
+ if (child.name.endsWith(".jsonl")) out.push(join(children, child.name));
158
+ }
159
+ }
160
+ }
161
+ }
162
+ return out;
163
+ }
164
+
165
+ function ccParentId(path) {
166
+ return basename(dirname(path)) === "subagents" ? basename(dirname(dirname(path))) : null;
167
+ }
168
+ function ccSessionId(path) {
169
+ const parent = ccParentId(path), id = basename(path, ".jsonl");
170
+ return parent ? `${parent}.${id}` : id;
171
+ }
172
+
90
173
  // Cap for the unknown-part fallback: prefix stays searchable, the full
91
174
  // bytes are always in the verbatim blob — this bounds the index, it does
92
175
  // not strip the record.
@@ -178,9 +261,11 @@ export function extractCcText(bytes) {
178
261
 
179
262
  // --------------------------------------------------------------------- pi
180
263
 
181
- function piRoots(home = homedir()) {
264
+ function piRoots(home = homedir(), env = process.env, options = {}) {
265
+ if (env.PI_CODING_AGENT_SESSION_DIR) return configuredRoot(env.PI_CODING_AGENT_SESSION_DIR, "", { home, ...options, tilde: true });
266
+ if (env.PI_CODING_AGENT_DIR) return configuredRoot(env.PI_CODING_AGENT_DIR, "sessions", { home, ...options, tilde: true });
182
267
  const root = join(home, ".pi", "agent", "sessions");
183
- return existsSync(root) ? [root] : [];
268
+ return directoryExists(root) ? [root] : [];
184
269
  }
185
270
 
186
271
  export function extractPiText(bytes) {
@@ -214,9 +299,10 @@ function piSessionId(path) {
214
299
 
215
300
  // ------------------------------------------------------------------ codex
216
301
 
217
- function codexRoots(home = homedir()) {
302
+ function codexRoots(home = homedir(), env = process.env, options = {}) {
303
+ if (env.CODEX_HOME) return configuredRoot(env.CODEX_HOME, "sessions", { home, ...options });
218
304
  const root = join(home, ".codex", "sessions");
219
- return existsSync(root) ? [root] : [];
305
+ return directoryExists(root) ? [root] : [];
220
306
  }
221
307
 
222
308
  export function extractCodexText(bytes) {
@@ -268,21 +354,22 @@ export const SESSION_FORMATS = {
268
354
  cc: {
269
355
  source: "cc",
270
356
  defaultRoots: ccRoots,
271
- listFiles: (roots) => roots.flatMap((r) => listJsonlFiles(r, 1)),
272
- sessionId: (path) => basename(path, ".jsonl"),
357
+ listFiles: (roots, options) => roots.flatMap((r) => listCcFiles(r, options)),
358
+ sessionId: ccSessionId,
359
+ ignoreKeys: (path) => [basename(path, ".jsonl"), ccParentId(path)].filter(Boolean),
273
360
  extractText: extractCcText,
274
361
  },
275
362
  pi: {
276
363
  source: "pi",
277
364
  defaultRoots: piRoots,
278
- listFiles: (roots) => roots.flatMap((r) => listJsonlFiles(r, 1)),
365
+ listFiles: (roots, options) => roots.flatMap((r) => listJsonlFiles(r, 1, options)),
279
366
  sessionId: piSessionId,
280
367
  extractText: extractPiText,
281
368
  },
282
369
  codex: {
283
370
  source: "codex",
284
371
  defaultRoots: codexRoots,
285
- listFiles: (roots) => roots.flatMap((r) => listJsonlFiles(r, 3)),
372
+ listFiles: (roots, options) => roots.flatMap((r) => listJsonlFiles(r, 3, options)),
286
373
  sessionId: codexSessionId,
287
374
  extractText: extractCodexText,
288
375
  },
@@ -0,0 +1,87 @@
1
+ // Independent native record custody. Recipes are relaunch templates, not proof
2
+ // of where a past process wrote. Only the execution-side recorder resolves env.
3
+ import { createHash, randomUUID } from "node:crypto";
4
+ import { lstatSync, mkdirSync, readFileSync, readdirSync, realpathSync, renameSync, writeFileSync } from "node:fs";
5
+ import { basename, dirname, join, resolve } from "node:path";
6
+ import { nativeLaunchLocations } from "./session-roots.mjs";
7
+
8
+ function canonical(home) {
9
+ try { return realpathSync(home); }
10
+ catch (e) {
11
+ if (e.code !== "ENOENT") throw e;
12
+ home = resolve(home);
13
+ return dirname(home) === home ? home : join(canonical(dirname(home)), basename(home));
14
+ }
15
+ }
16
+ // The source-home leaf is an identity, not a redirectable lookup hint. Resolve
17
+ // parent aliases (e.g. /tmp), but never select another home's authority by
18
+ // following a substituted leaf. Retired homes may legitimately be absent.
19
+ function sourceHome(home) {
20
+ const absolute = resolve(home);
21
+ try {
22
+ const stat = lstatSync(absolute);
23
+ if (stat.isSymbolicLink() || !stat.isDirectory()) throw new Error("native record source home was substituted; refusing redirected history authority");
24
+ } catch (e) { if (e.code !== "ENOENT") throw e; }
25
+ return join(canonical(dirname(absolute)), basename(absolute));
26
+ }
27
+ export function nativeHistoryPath(home) {
28
+ home = sourceHome(home);
29
+ return join(dirname(home), ".oats-native-record", createHash("sha256").update(home).digest("hex"));
30
+ }
31
+ function atomic(path, value) {
32
+ const tmp = `${path}.${randomUUID()}.tmp`;
33
+ writeFileSync(tmp, JSON.stringify(value) + "\n", { mode: 0o600, flag: "wx" });
34
+ renameSync(tmp, path);
35
+ }
36
+ function manifest(home) {
37
+ const value = JSON.parse(readFileSync(join(nativeHistoryPath(home), "history.json"), "utf8"));
38
+ if (value.version !== 1 || value.home !== sourceHome(home) || typeof value.completeHistory !== "boolean") throw new Error("invalid native record history authority");
39
+ return value;
40
+ }
41
+ // Called only for a newly scaffolded home, before capability hooks. A legacy
42
+ // start can record new locations but cannot invent authority for earlier starts.
43
+ export function initializeNativeHistory(home, { completeHistory = true } = {}) {
44
+ const dir = nativeHistoryPath(home);
45
+ mkdirSync(dir, { recursive: true, mode: 0o700 });
46
+ try { writeFileSync(join(dir, "history.json"), JSON.stringify({ version: 1, home: sourceHome(home), completeHistory }) + "\n", { mode: 0o600, flag: "wx" }); }
47
+ catch (e) { if (e.code !== "EEXIST") throw e; manifest(home); } // retain earlier launches if a name is reused
48
+ }
49
+ export function prepareNativeStart(home, runtime) {
50
+ try { manifest(home); }
51
+ catch (e) {
52
+ if (e.code !== "ENOENT") throw e;
53
+ initializeNativeHistory(home, { completeHistory: false });
54
+ }
55
+ const id = randomUUID();
56
+ atomic(join(nativeHistoryPath(home), `${id}.json`), { version: 1, id, home: sourceHome(home), runtime, state: "pending" });
57
+ return id;
58
+ }
59
+ // Executed under the SAME environment prefix, cwd and native argv as the
60
+ // harness, after the backend shell's startup. No environment map or argv is
61
+ // serialized. Failure leaves pending evidence and prevents the native exec.
62
+ export function recordNativeStart(home, id, runtime, args, env = process.env) {
63
+ manifest(home);
64
+ if (!/^[a-f0-9-]{36}$/.test(id)) throw new Error("invalid native record start id");
65
+ const path = join(nativeHistoryPath(home), `${id}.json`);
66
+ const pending = JSON.parse(readFileSync(path, "utf8"));
67
+ if (pending.version !== 1 || pending.id !== id || pending.home !== sourceHome(home) || pending.runtime !== runtime || pending.state !== "pending") throw new Error("invalid native record start receipt");
68
+ const locations = nativeLaunchLocations(runtime, { cwd: home, env, args }).map(canonical);
69
+ atomic(path, { version: 1, id, home: sourceHome(home), runtime, state: "started", startedAt: new Date().toISOString(), locations });
70
+ }
71
+ export function historicalSessionRoots(home) {
72
+ const authority = manifest(home);
73
+ if (!authority.completeHistory) throw new Error("native record roots for earlier launches are unknown; legacy history cannot certify complete capture");
74
+ const roots = { cc: [], pi: [], codex: [] };
75
+ for (const name of readdirSync(nativeHistoryPath(home)).sort()) {
76
+ if (name === "history.json") continue;
77
+ if (!/^[a-f0-9-]{36}\.json$/.test(name)) throw new Error("unrecognized or unfinished native record receipt");
78
+ const row = JSON.parse(readFileSync(join(nativeHistoryPath(home), name), "utf8"));
79
+ const source = { claude: "cc", pi: "pi", codex: "codex" }[row.runtime];
80
+ if (row.version !== 1 || row.id !== basename(name, ".json") || row.home !== authority.home || !source || row.state !== "started" || !Array.isArray(row.locations) || row.locations.length === 0) throw new Error("native record launch is pending or its location receipt is invalid");
81
+ for (const path of row.locations) {
82
+ if (typeof path !== "string" || !path.startsWith("/") || path.includes("\0") || resolve(path) !== path) throw new Error("invalid historical native record location");
83
+ if (!roots[source].includes(path)) roots[source].push(path);
84
+ }
85
+ }
86
+ return roots;
87
+ }