@awebai/oats 0.35.1 → 0.35.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,20 @@
1
+ # OATS 0.35.2
2
+
3
+ ## Fixed
4
+
5
+ - **A capture pass runs in bounded memory, whatever the record's size**
6
+ (awebai/oats#456). On a host with a 2.6 GB turn record, the periodic pass
7
+ (`capture.mjs` under `--max-old-space-size=256`) died within seconds every
8
+ time it ran. Each death left a dead-owner capture lock, the cause of #437.
9
+ The pass loaded whole files:
10
+ - the first append to a stream parsed its whole journal;
11
+ - the aw dedupe read the whole aw journal into a set of ids;
12
+ - every changed comm log was read whole (a log over 512 MB failed even
13
+ with a large heap);
14
+ - a lost session offset re-read the whole journal and transcript;
15
+ - a session's new lines were all collected before any was appended.
16
+
17
+ Each is now streamed. On a 1.1 GB synthetic record the same pass peaks at
18
+ 160 MB resident instead of about 2 GB. A host whose capture job was turned
19
+ off can turn it back on: the first pass after a long gap captures the
20
+ backlog in bounded memory.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@awebai/oats",
3
- "version": "0.35.1",
3
+ "version": "0.35.2",
4
4
  "description": "OATS (Open Agent Team Specification) — durable souls, disposable instances, targetable capability packages, and the runtime-neutral oats CLI/kernel.",
5
5
  "keywords": [
6
6
  "agents",
@@ -226,6 +226,22 @@ this package, restart any long-running `capture --watch` process — a
226
226
  daemon holding the old database file open would otherwise keep indexing
227
227
  into an orphaned inode until it restarts.
228
228
 
229
+ ## Memory
230
+
231
+ A capture pass runs in bounded memory, whatever the size of the record.
232
+ Journals are streamed, never parsed whole: the first append to a stream
233
+ validates its journal one line at a time. Comm logs are read in chunks, and
234
+ the aw dedupe looks up only the changed log's own turn ids, in one streamed
235
+ read of the journal. A session's new lines are walked, never collected, and
236
+ appended in 8 MB batches. A lost offset is rebuilt from the journal's tail.
237
+
238
+ So the heap a pass needs grows with one changed comm log's entry count, not
239
+ with the record. A host job can keep a small `--max-old-space-size`;
240
+ `packages/record/test/capture-bounded-memory.test.mjs` captures a record four
241
+ times its 64 MB heap. One cost is not on the heap: the new bytes of one
242
+ session are read as one buffer, so resident memory grows with a single
243
+ session's backlog.
244
+
229
245
  ## Known costs, accepted for v1
230
246
 
231
247
  - Deletion via tombstone is eventual: an offline replica retains bytes until
@@ -9,9 +9,14 @@
9
9
  // account; account and file identity live in each turn's provenance.
10
10
  // Projection is deterministic, so the same entry captured on two machines
11
11
  // dedupes by id. Reconciliation is the capture: scan, project, append what
12
- // is new, batched with a single fsync per pass.
12
+ // is new, in bounded batches.
13
+ //
14
+ // Memory is bounded by one changed log's ids, never by the record or a log's
15
+ // bytes (awebai/oats#456): a log is read in chunks, and what the stream
16
+ // already holds is looked up for that log's ids alone, in one streamed read
17
+ // of the journal, never as the whole stream's id set.
13
18
 
14
- import { existsSync, mkdirSync, readdirSync, readFileSync, statSync, writeFileSync } from "node:fs";
19
+ import { closeSync, existsSync, fstatSync, mkdirSync, openSync, readdirSync, readFileSync, statSync, writeFileSync } from "node:fs";
15
20
  import { basename, dirname, join } from "node:path";
16
21
  import { homedir } from "node:os";
17
22
 
@@ -20,6 +25,7 @@ import {
20
25
  projectInteractionLogEntry,
21
26
  } from "./project-aweb.mjs";
22
27
  import { loadIgnore } from "./ignore.mjs";
28
+ import { bufferLines, fileChunks } from "./file-lines.mjs";
23
29
 
24
30
  export function defaultCommLogDir(home = homedir()) {
25
31
  return join(home, ".config", "aw", "logs");
@@ -37,42 +43,78 @@ export function listCommLogs(dir = defaultCommLogDir()) {
37
43
  .map((name) => join(dir, name));
38
44
  }
39
45
 
40
- function readEntries(path) {
41
- const entries = [];
42
- let skipped = 0;
43
- const text = readFileSync(path, "utf8");
44
- const lines = text.split("\n");
45
- for (let i = 0; i < lines.length; i++) {
46
- const line = lines[i];
47
- if (line.trim() === "") continue;
46
+ // The entries of the first `end` bytes of a JSONL log, one at a time, read
47
+ // in chunks from `fd` (a comm log can exceed V8's maximum string length).
48
+ // A final line without its newline is still an entry when it parses; when it
49
+ // does not, it is a torn tail the client is still writing, never counted. An
50
+ // interior bad line is skipped and counted (`onSkipped`), never fatal.
51
+ function* logEntries(fd, end, onSkipped) {
52
+ for (const line of bufferLines(fileChunks(fd, { end }))) {
53
+ const final = line[line.length - 1] !== 10;
54
+ let entry;
48
55
  try {
49
- entries.push(JSON.parse(line));
56
+ // Inside the try: a line beyond V8's string limit is a bad line like any other.
57
+ const text = line.toString("utf8");
58
+ if (text.trim() === "") continue;
59
+ entry = JSON.parse(text);
50
60
  } catch {
51
- // A torn final line is expected while the client is writing; an
52
- // interior bad line is skipped and counted, never fatal to capture.
53
- if (i < lines.length - 1) skipped++;
61
+ if (!final) onSkipped();
62
+ continue;
54
63
  }
64
+ yield entry;
55
65
  }
56
- return { entries, skipped };
57
66
  }
58
67
 
59
- function projectFile(entries, project) {
60
- const turns = [];
61
- let failed = 0;
62
- for (const entry of entries) {
63
- try {
64
- turns.push(project(entry));
65
- } catch {
66
- // One unprojectable entry must not abort the pass, but it fails
67
- // visibly: counted here and reported by the caller.
68
- failed++;
68
+ const APPEND_BATCH = 2000;
69
+
70
+ // One log file into `streamId`: project every entry, append the turns the
71
+ // stream lacks. With `knownIds` (a caller's own set, kept up to date), that
72
+ // set decides; without, the log is read twice over the same prefix (its size
73
+ // when opened: the client may be appending): once for its turn ids, which one
74
+ // streamed read of the journal narrows to those already there, and once to
75
+ // append the rest in batches.
76
+ function captureLogFile(store, { streamId, path, project, knownIds }) {
77
+ const fd = openSync(path, "r");
78
+ try {
79
+ const end = fstatSync(fd).size;
80
+ let entries = 0, skipped = 0, failed = 0;
81
+ function* turns() {
82
+ entries = 0; skipped = 0; failed = 0;
83
+ for (const entry of logEntries(fd, end, () => skipped++)) {
84
+ entries++;
85
+ let turn;
86
+ try { turn = project(entry); }
87
+ catch {
88
+ // One unprojectable entry must not abort the pass, but it fails
89
+ // visibly: counted here and reported by the caller.
90
+ failed++;
91
+ continue;
92
+ }
93
+ yield turn;
94
+ }
69
95
  }
70
- }
71
- return { turns, failed };
96
+ let known = knownIds;
97
+ if (!known) {
98
+ const candidates = new Set();
99
+ for (const turn of turns()) candidates.add(turn.id);
100
+ known = store.idsAmong(streamId, candidates);
101
+ }
102
+ let appended = 0, batch = [];
103
+ const flush = () => { store.appendBatch(streamId, batch); appended += batch.length; batch = []; };
104
+ for (const turn of turns()) {
105
+ if (known.has(turn.id)) continue;
106
+ known.add(turn.id);
107
+ batch.push(turn);
108
+ if (batch.length >= APPEND_BATCH) flush();
109
+ }
110
+ if (batch.length) flush();
111
+ return { entries, appended, skipped, failed };
112
+ } finally { closeSync(fd); }
72
113
  }
73
114
 
74
115
  // Capture one comm-log file into `<owner>~aw`. The account name is the
75
- // filename stem. `knownIds` carries the stream's ids across files in a pass.
116
+ // filename stem. `knownIds`, when given, is the caller's own id set, kept up to
117
+ // date, and decides instead of a read of the journal.
76
118
  //
77
119
  // Deliberately does NOT consult the ignore list: this is an explicit
78
120
  // "capture this file" command, and the caller has named the file. The
@@ -81,26 +123,8 @@ function projectFile(entries, project) {
81
123
  export function captureCommLog(store, { owner, path, knownIds = null }) {
82
124
  const account = basename(path, ".jsonl");
83
125
  const streamId = awStream(owner);
84
- const ids = knownIds ?? new Set(store.readStream(streamId).map((t) => t.id));
85
- const { entries, skipped } = readEntries(path);
86
- const { turns, failed } = projectFile(entries, (e) =>
87
- projectCommLogEntry(e, { selfName: account }),
88
- );
89
- const fresh = [];
90
- for (const turn of turns) {
91
- if (ids.has(turn.id)) continue;
92
- ids.add(turn.id);
93
- fresh.push(turn);
94
- }
95
- store.appendBatch(streamId, fresh);
96
- return {
97
- account,
98
- stream: streamId,
99
- entries: entries.length,
100
- appended: fresh.length,
101
- skipped,
102
- failed,
103
- };
126
+ const r = captureLogFile(store, { streamId, path, knownIds, project: (e) => projectCommLogEntry(e, { selfName: account }) });
127
+ return { account, stream: streamId, ...r };
104
128
  }
105
129
 
106
130
  // Capture one workspace interaction log into `<owner>~aw`. Like
@@ -108,19 +132,8 @@ export function captureCommLog(store, { owner, path, knownIds = null }) {
108
132
  // explicit per-file command; pass-level entry points enforce the policy.
109
133
  export function captureInteractionLog(store, { owner, path, selfName, workspace, knownIds = null }) {
110
134
  const streamId = awStream(owner);
111
- const ids = knownIds ?? new Set(store.readStream(streamId).map((t) => t.id));
112
- const { entries, skipped } = readEntries(path);
113
- const { turns, failed } = projectFile(entries, (e) =>
114
- projectInteractionLogEntry(e, { selfName, workspace }),
115
- );
116
- const fresh = [];
117
- for (const turn of turns) {
118
- if (ids.has(turn.id)) continue;
119
- ids.add(turn.id);
120
- fresh.push(turn);
121
- }
122
- store.appendBatch(streamId, fresh);
123
- return { stream: streamId, entries: entries.length, appended: fresh.length, skipped, failed };
135
+ const r = captureLogFile(store, { streamId, path, knownIds, project: (e) => projectInteractionLogEntry(e, { selfName, workspace }) });
136
+ return { stream: streamId, ...r };
124
137
  }
125
138
 
126
139
  // Derived seen-files cache (same pattern as capture-cc): skip a log file
@@ -143,17 +156,15 @@ function saveSeenCache(store, cache) {
143
156
  writeFileSync(path, JSON.stringify(cache));
144
157
  }
145
158
 
146
- // One reconciliation pass over every default comm log. The stream's known
147
- // ids are read once and shared across files; unchanged files are skipped.
159
+ // One reconciliation pass over every default comm log. Unchanged files are
160
+ // skipped; each changed one looks up its own ids in the stream.
148
161
  //
149
162
  // Files matching the record's ignore list (`<root>/ignore`, see ignore.mjs)
150
163
  // are skipped before being opened — no turns, no seen-cache entry — and
151
164
  // reported as `{ account, path, ignored: true }` so a pass stays visible
152
165
  // about what it refused to read.
153
166
  export function captureAwLogs(store, { owner, commLogDir, ignore = null } = {}) {
154
- const streamId = awStream(owner);
155
167
  const ign = ignore ?? loadIgnore(store.root);
156
- let knownIds = null; // lazy: only read the journal if some file changed
157
168
  const seen = loadSeenCache(store);
158
169
  const results = [];
159
170
  for (const path of listCommLogs(commLogDir ?? defaultCommLogDir())) {
@@ -170,8 +181,7 @@ export function captureAwLogs(store, { owner, commLogDir, ignore = null } = {})
170
181
  }
171
182
  const prev = seen[path];
172
183
  if (prev && prev.size === stat.size && prev.mtimeMs === stat.mtimeMs) continue;
173
- if (!knownIds) knownIds = new Set(store.readStream(streamId).map((t) => t.id));
174
- results.push(captureCommLog(store, { owner, path, knownIds }));
184
+ results.push(captureCommLog(store, { owner, path }));
175
185
  seen[path] = { size: stat.size, mtimeMs: stat.mtimeMs };
176
186
  }
177
187
  saveSeenCache(store, seen);
@@ -29,6 +29,7 @@ import { jsonlLines, SESSION_FORMATS } from "./formats.mjs";
29
29
  import { loadIgnore } from "./ignore.mjs";
30
30
  import { assertIdentity, assertProtectedDescriptor, digest, identity, readRange, verifySnapshot } from "./session-snapshot.mjs";
31
31
  import { guardCapturedPath } from "./native-history.mjs";
32
+ import { fileChunks } from "./file-lines.mjs";
32
33
  import { isDeepStrictEqual } from "node:util";
33
34
 
34
35
  export const SESSION_STREAM_SOURCE = "cc";
@@ -118,16 +119,16 @@ function readFrom(path, start, size) {
118
119
  }
119
120
  }
120
121
 
121
- // The journal's last captured source line, read from its tail without
122
- // parsing the whole file (backward scan, doubling window). 0 when the
123
- // journal is missing or empty.
124
- function lastJournalLine(store, streamId) {
122
+ // The journal's last whole turn, read from its tail without parsing the whole
123
+ // file (backward scan, doubling window). null when the journal is missing or
124
+ // holds none.
125
+ function lastJournalTurn(store, streamId) {
125
126
  const path = store.journalPath(streamId);
126
127
  let size;
127
128
  try {
128
129
  size = statSync(path).size;
129
130
  } catch (err) {
130
- if (err.code === "ENOENT") return 0;
131
+ if (err.code === "ENOENT") return null;
131
132
  throw err;
132
133
  }
133
134
  let window = 64 * 1024;
@@ -145,12 +146,12 @@ function lastJournalLine(store, streamId) {
145
146
  // window covers the whole file).
146
147
  if (i === 0 && start > 0) break;
147
148
  try {
148
- return JSON.parse(candidates[i]).provenance?.origin?.line ?? 0;
149
+ return JSON.parse(candidates[i]);
149
150
  } catch {
150
151
  continue; // fragment or torn line: look further back
151
152
  }
152
153
  }
153
- if (start === 0) return 0;
154
+ if (start === 0) return null;
154
155
  window *= 2;
155
156
  }
156
157
  }
@@ -160,24 +161,83 @@ function lastJournalLine(store, streamId) {
160
161
  // Honest limit: an in-place REWRITE of already-captured lines is not
161
162
  // detected (only growth is; a shrink triggers a rescan via the size
162
163
  // check in the caller). Transcript writers are append-only in practice.
163
- function offsetFromJournal(store, streamId, sourcePath, final, sourceBytes) {
164
- const turns = store.readStream(streamId);
165
- if (turns.length === 0) return { bytes: 0, line: 0, lastTs: "" };
166
- const last = turns[turns.length - 1];
164
+ // Both are read in bounded memory: the journal's last turn from its tail, the
165
+ // source's lines counted chunk by chunk (awebai/oats#456).
166
+ // `last` is the journal's last turn (lastJournalTurn), null when it has none.
167
+ function offsetFromJournal(last, sourcePath, final, sourceBytes) {
168
+ if (!last) return { bytes: 0, line: 0, lastTs: "" };
167
169
  const lastLine = last.provenance?.origin?.line ?? 0;
168
- const bytes = sourceBytes ?? readFileSync(sourcePath);
169
- let line = 0;
170
- let offset = 0;
171
- while (line < lastLine && offset < bytes.length) {
172
- const nl = bytes.indexOf(10, offset);
173
- if (nl === -1) break;
174
- line++;
175
- offset = nl + 1;
176
- }
170
+ const { line, offset } = sourceBytes ? lineOffset([sourceBytes], lastLine) : lineOffsetOfFile(sourcePath, lastLine);
177
171
  if (final && line < lastLine) throw new Error(`session source is shorter than its captured journal: ${sourcePath}`);
178
172
  return { bytes: offset, line, lastTs: last.ts ?? "" };
179
173
  }
180
174
 
175
+ // The byte offset just past the `lastLine`-th complete line of the bytes in
176
+ // `chunks` (in order), and how many complete lines it found (fewer when the
177
+ // source is shorter).
178
+ function lineOffset(chunks, lastLine) {
179
+ let line = 0, offset = 0;
180
+ for (const chunk of chunks) {
181
+ let from = 0;
182
+ while (line < lastLine) {
183
+ const nl = chunk.indexOf(10, from);
184
+ if (nl === -1) break;
185
+ line++;
186
+ from = nl + 1;
187
+ }
188
+ offset += line < lastLine ? chunk.length : from;
189
+ if (line >= lastLine) break;
190
+ }
191
+ return { line, offset };
192
+ }
193
+
194
+ function lineOffsetOfFile(path, lastLine) {
195
+ const fd = openSync(path, "r");
196
+ try { return lineOffset(fileChunks(fd), lastLine); }
197
+ finally { closeSync(fd); }
198
+ }
199
+
200
+ // Source bytes of turns held before they are appended: what a pass keeps of a
201
+ // transcript at once, whatever its size.
202
+ const FLUSH_BYTES = 8 * 1024 * 1024;
203
+
204
+ // The COMPLETE lines of `chunk`, one at a time, with their stamps:
205
+ // { text, ts, bytes }. `walk` records where the walk ended (`scanned`, bytes)
206
+ // and why it stopped early (`reason`): a torn tail, invalid UTF-8 (decoding
207
+ // replacement characters would change both the verbatim line and its byte
208
+ // offset, possibly treating a later fragment as a record) or a line beyond
209
+ // V8's string limit.
210
+ function* completeLines(chunk, walk) {
211
+ let scanned = 0;
212
+ walk.scanned = 0;
213
+ while (scanned < chunk.length) {
214
+ const nl = chunk.indexOf(10, scanned);
215
+ if (nl === -1) { walk.reason = "torn-tail"; return; }
216
+ const bytes = chunk.subarray(scanned, nl);
217
+ if (!isUtf8(bytes)) { walk.reason = "invalid-utf8"; return; }
218
+ let text;
219
+ try { text = bytes.toString("utf8"); }
220
+ catch (err) {
221
+ if (err.code !== "ERR_STRING_TOO_LONG") throw err;
222
+ walk.reason = "oversized-line";
223
+ return;
224
+ }
225
+ const lineBytes = nl - scanned + 1;
226
+ let ts = "";
227
+ if (text.trim() !== "") {
228
+ try {
229
+ const d = JSON.parse(text);
230
+ if (typeof d?.timestamp === "string") ts = d.timestamp;
231
+ } catch {
232
+ /* unparseable native line: captured verbatim */
233
+ }
234
+ }
235
+ scanned += lineBytes;
236
+ walk.scanned = scanned;
237
+ yield { text, ts, bytes: lineBytes };
238
+ }
239
+ }
240
+
181
241
  // One reconciliation pass for one format: one turn per NEW complete line
182
242
  // of every session file under `roots`. Unstamped leading lines are held
183
243
  // until the file shows its first timestamp (then they carry it forward),
@@ -253,10 +313,12 @@ export function captureSessions(store, { owner, roots, files, format = "cc", ign
253
313
  // journal is the truth; before appending anything, any disagreement
254
314
  // rebuilds the offset from it. Background passes check on growth;
255
315
  // final passes also verify unchanged files before confirming capture.
256
- if (state && (final || stat.size > state.bytes) && state.line !== lastJournalLine(store, streamId)) {
316
+ let journalLast; // the journal's last turn, read from its tail once, when needed
317
+ const lastTurn = () => (journalLast === undefined ? (journalLast = lastJournalTurn(store, streamId)) : journalLast);
318
+ if (state && (final || stat.size > state.bytes) && state.line !== (lastTurn()?.provenance?.origin?.line ?? 0)) {
257
319
  state = null;
258
320
  }
259
- if (!state) state = offsetFromJournal(store, streamId, path, final, sourceBytes);
321
+ if (!state) state = offsetFromJournal(lastTurn(), path, final, sourceBytes);
260
322
  if (stat.size <= state.bytes) {
261
323
  unchanged++;
262
324
  offsets[offKey] = state;
@@ -265,62 +327,38 @@ export function captureSessions(store, { owner, roots, files, format = "cc", ign
265
327
  }
266
328
 
267
329
  const chunk = sourceBytes ? sourceBytes.subarray(state.bytes) : readRange(fd, state.bytes, stat.size, path, capturedPi);
268
- // Phase 1: collect the COMPLETE lines of the chunk with their stamps.
269
- const lines = [];
270
- let scanned = 0;
271
- let reason;
272
- while (scanned < chunk.length) {
273
- const nl = chunk.indexOf(10, scanned);
274
- if (nl === -1) { reason = "torn-tail"; break; }
275
- const bytes = chunk.subarray(scanned, nl);
276
- // Decoding replacement characters would change both the verbatim line
277
- // and its byte offset, possibly treating a later fragment as a record.
278
- if (!isUtf8(bytes)) { reason = "invalid-utf8"; break; }
279
- let text;
280
- try { text = bytes.toString("utf8"); }
281
- catch (err) {
282
- if (err.code !== "ERR_STRING_TOO_LONG") throw err;
283
- reason = "oversized-line";
284
- break;
285
- }
286
- const lineBytes = nl - scanned + 1;
287
- let ts = "";
288
- if (text.trim() !== "") {
289
- try {
290
- const d = JSON.parse(text);
291
- if (typeof d?.timestamp === "string") ts = d.timestamp;
292
- } catch {
293
- /* unparseable native line: captured verbatim below */
294
- }
295
- }
296
- lines.push({ text, ts, bytes: lineBytes });
297
- scanned += lineBytes;
298
- }
299
- if (reason) {
330
+ // Every turn needs a stamp. Leading lines before the file's first stamp
331
+ // carry it backward (deterministic: the file's first stamp is invariant
332
+ // however capture is scheduled); if the file has shown no stamp at all
333
+ // yet, hold everything for a later pass. The lines are walked, never
334
+ // collected: a first capture of a huge transcript is one file's worth of
335
+ // NEW lines, far more than the heap (awebai/oats#456).
336
+ const stopped = (walk) => {
337
+ if (!walk.reason) return;
300
338
  incomplete++;
301
- issues.push({ source: fmt.source, path, reason, offset: state.bytes + scanned });
302
- }
303
- // Phase 2: every turn needs a stamp. Leading lines before the file's
304
- // first stamp carry it backward (deterministic: the file's first
305
- // stamp is invariant however capture is scheduled); if the file has
306
- // shown no stamp at all yet, hold everything for a later pass.
339
+ issues.push({ source: fmt.source, path, reason: walk.reason, offset: state.bytes + walk.scanned });
340
+ };
307
341
  let lastTs = state.lastTs ?? "";
308
342
  if (!lastTs) {
309
- const first = lines.find((l) => l.ts);
310
- if (!first) {
311
- if (lines.length > 0) {
343
+ const scan = {};
344
+ let any = false;
345
+ for (const l of completeLines(chunk, scan)) {
346
+ any = true;
347
+ if (l.ts) { lastTs = l.ts; break; }
348
+ }
349
+ if (!lastTs) {
350
+ stopped(scan);
351
+ if (any) {
312
352
  held++;
313
353
  issues.push({ source: fmt.source, path, reason: "unstamped", offset: state.bytes });
314
354
  }
315
355
  verifySource();
316
356
  continue; // do not advance; retry when a stamp exists
317
357
  }
318
- lastTs = first.ts;
319
358
  }
320
359
  verifySource(); // parsing/recovery may take time; still no journal write yet
321
360
  // Turns flush to the journal in bounded batches, so memory stays flat
322
- // however large the backlog (a first capture of a huge transcript is
323
- // one file's worth of NEW lines). A crash between flushes cannot
361
+ // however large the backlog. A crash between flushes cannot
324
362
  // duplicate: this file's offset is saved only after its final flush,
325
363
  // and a lost offset rebuilds from the journal's own last line number.
326
364
  let fresh = [];
@@ -336,7 +374,8 @@ export function captureSessions(store, { owner, roots, files, format = "cc", ign
336
374
  };
337
375
  let lineNo = state.line;
338
376
  let consumed = 0;
339
- for (const l of lines) {
377
+ const walk = {};
378
+ for (const l of completeLines(chunk, walk)) {
340
379
  if (l.ts) lastTs = l.ts;
341
380
  lineNo++;
342
381
  consumed += l.bytes;
@@ -348,8 +387,9 @@ export function captureSessions(store, { owner, roots, files, format = "cc", ign
348
387
  ),
349
388
  );
350
389
  freshBytes += l.bytes;
351
- if (freshBytes >= 64 * 1024 * 1024) flush();
390
+ if (freshBytes >= FLUSH_BYTES) flush();
352
391
  }
392
+ stopped(walk);
353
393
  const grew = fresh.length > 0 || consumed > 0;
354
394
  flush();
355
395
  offsets[offKey] = { bytes: state.bytes + consumed, line: lineNo, lastTs };
@@ -0,0 +1,38 @@
1
+ // Reading a file in bounded memory: its bytes in chunks, and its lines one at
2
+ // a time. Journals, comm logs and transcripts can be far larger than the heap
3
+ // (awebai/oats#456), so no reader here ever holds more than one chunk and one
4
+ // line.
5
+
6
+ import { readSync } from "node:fs";
7
+
8
+ // The bytes of `fd` from `start` up to `end` (default: the end of the file), in
9
+ // chunks of at most `size` bytes. Each chunk is a fresh buffer, so a caller may
10
+ // keep one past the next read.
11
+ export function* fileChunks(fd, { start = 0, end = Infinity, size = 1 << 20 } = {}) {
12
+ for (let position = start; position < end; ) {
13
+ const chunk = Buffer.allocUnsafe(Math.min(size, end - position));
14
+ const n = readSync(fd, chunk, 0, chunk.length, position);
15
+ if (n === 0) return;
16
+ position += n;
17
+ yield chunk.subarray(0, n);
18
+ }
19
+ }
20
+
21
+ // The lines of `chunks`, one at a time: each complete line WITH its newline,
22
+ // then a final fragment without one, if the bytes end mid-line. A caller tells
23
+ // them apart by the last byte. Fragments are joined only at a newline, so a
24
+ // long line is copied once, not as each chunk arrives.
25
+ export function* bufferLines(chunks) {
26
+ let pieces = [], size = 0;
27
+ for (const chunk of chunks) {
28
+ let from = 0;
29
+ for (let nl = chunk.indexOf(10, from); nl >= 0; nl = chunk.indexOf(10, from)) {
30
+ pieces.push(chunk.subarray(from, nl + 1));
31
+ size += nl + 1 - from;
32
+ yield pieces.length === 1 ? pieces[0] : Buffer.concat(pieces, size);
33
+ pieces = []; size = 0; from = nl + 1;
34
+ }
35
+ if (from < chunk.length) { pieces.push(chunk.subarray(from)); size += chunk.length - from; }
36
+ }
37
+ if (size > 0) yield pieces.length === 1 ? pieces[0] : Buffer.concat(pieces, size);
38
+ }
@@ -13,6 +13,7 @@
13
13
  // - the effective record is the id-deduplicated union across streams,
14
14
  // minus turns hidden by valid tombstones.
15
15
 
16
+ import { bufferLines, fileChunks } from "./file-lines.mjs";
16
17
  import {
17
18
  appendFileSync,
18
19
  closeSync,
@@ -413,34 +414,33 @@ export class RecordStore {
413
414
  let fd;
414
415
  try { fd = openSync(this.journalPath(streamId), "r"); }
415
416
  catch (e) { if (e.code === "ENOENT") return; throw e; }
416
- let pieces = [], size = 0, lineStart = 0;
417
+ let lineStart = 0;
417
418
  try {
418
- for (;;) {
419
- const chunk = Buffer.allocUnsafe(65536);
420
- const n = readSync(fd, chunk, 0, chunk.length, null);
421
- if (n === 0) break;
422
- let from = 0;
423
- for (let nl = chunk.indexOf(10, from); nl >= 0 && nl < n; nl = chunk.indexOf(10, from)) {
424
- pieces.push(chunk.subarray(from, nl + 1));
425
- size += nl + 1 - from;
426
- const line = pieces.length === 1 ? pieces[0] : Buffer.concat(pieces, size);
427
- let parsed;
428
- try { parsed = parseJournal(line); }
429
- catch (e) {
430
- if (e instanceof StoreError) throw new StoreError(`corrupt interior journal line at byte ${lineStart}`);
431
- throw e;
432
- }
433
- lineStart += size;
434
- pieces = []; size = 0; from = nl + 1;
435
- yield* parsed.turns;
419
+ for (const line of bufferLines(fileChunks(fd, { size: 65536 }))) {
420
+ // A final fragment without a newline is a torn tail, even if its
421
+ // JSON is valid. Readers leave it for the owner's append repair.
422
+ if (line[line.length - 1] !== 10) return;
423
+ let parsed;
424
+ try { parsed = parseJournal(line); }
425
+ catch (e) {
426
+ if (e instanceof StoreError) throw new StoreError(`corrupt interior journal line at byte ${lineStart}`);
427
+ throw e;
436
428
  }
437
- if (from < n) { pieces.push(chunk.subarray(from, n)); size += n - from; }
429
+ lineStart += line.length;
430
+ yield* parsed.turns;
438
431
  }
439
- // A final fragment without a newline is a torn tail, even if its
440
- // JSON is valid. Readers leave it for the owner's append repair.
441
432
  } finally { closeSync(fd); }
442
433
  }
443
434
 
435
+ // The ids among `candidates` that the stream holds, in one streamed read of
436
+ // its journal: a dedupe for a few ids never builds the whole stream's set.
437
+ idsAmong(streamId, candidates) {
438
+ const found = new Set();
439
+ if (candidates.size === 0) return found;
440
+ for (const turn of this.iterateStream(streamId)) if (candidates.has(turn.id)) found.add(turn.id);
441
+ return found;
442
+ }
443
+
444
444
  // Is this a session-content stream (`<owner>~<source>.<session-id>`)?
445
445
  // Their journals hold whole conversations and can be large, so bulk
446
446
  // reads exclude them unless asked; access them per-thread instead.
@@ -548,12 +548,14 @@ export class RecordStore {
548
548
  }
549
549
  const path = this.journalPath(streamId);
550
550
  this.withStreamLock(streamId, () => {
551
- // First append to this stream in this instance: parse the whole
551
+ // First append to this stream in this instance: read the whole
552
552
  // journal, so interior corruption throws here instead of silently
553
553
  // collecting appends behind the damage. A torn tail is tolerated
554
- // (parseJournal treats it as final-line-torn) and repaired below.
554
+ // (iterateStream leaves it, as parseJournal treats it as
555
+ // final-line-torn) and repaired below. Streamed, one line at a time:
556
+ // a journal can be far larger than the heap (awebai/oats#456).
555
557
  if (!this.validatedStreams.has(streamId) && existsSync(path)) {
556
- parseJournal(readFileSync(path));
558
+ for (const _turn of this.iterateStream(streamId)) { /* validated as read */ }
557
559
  }
558
560
  this.validatedStreams.add(streamId);
559
561
  repairTail(path);