omnirush 0.4.1 → 0.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -52,6 +52,8 @@ import { dirname, join, relative, resolve, sep } from "node:path";
52
52
  import { promisify } from "node:util";
53
53
  import { zstdCompress as zstdCompressCb } from "node:zlib";
54
54
 
55
+ import { isRetryableStatus, retryAttempts, withRetries } from "./retry";
56
+
55
57
  const execFileAsync = promisify(execFile);
56
58
 
57
59
  // --- limits ---------------------------------------------------------------
@@ -67,10 +69,31 @@ export const MAX_SESSION_BYTES = 100 * 1024 * 1024 * 1024;
67
69
  // byte size + sha256) so replay reassembles the original deterministically.
68
70
  export const MAX_SLICE_BYTES = 4 * 1024 * 1024;
69
71
  // Whole-read ceiling. Files up to this size are read and processed as
70
- // one string (worst case: binary -> base64 inflates 4/3 to ~427 MiB
71
- // chars, comfortably under the runtime's ~512 MiB MAX_STRING_LENGTH).
72
- // Bigger files are streamed in fixed raw chunks and sliced.
73
- const WHOLE_READ_BYTES = 320 * 1024 * 1024;
72
+ // one string; bigger files stream in fixed raw chunks and slice.
73
+ // Deliberately tiny (issue #6): a whole read holds buffer + utf8 string
74
+ // + redaction copy + base64 simultaneously (~4-5x file size), so even a
75
+ // modest ceiling spikes RSS by that multiple per file. 8 MiB bounds the
76
+ // worst case to ~40 MiB; anything bigger takes the streaming path with
77
+ // identical output.
78
+ const WHOLE_READ_BYTES = 8 * 1024 * 1024;
79
+ // Text-segment cap for newline-less files (see the streaming reader).
80
+ const PENDING_TEXT_CAP_BYTES = 32 * 1024 * 1024;
81
+ // Change-journal memory ceiling (issue #6, Windows 8-9 GiB OOM): the
82
+ // journal holds every changed file's content between debounced change
83
+ // uploads. At HIGH_WATER an upload is forced immediately (normal
84
+ // backpressure); at the HARD ceiling — a backend down for a long time —
85
+ // the OLDEST records are dropped, loudly (warn + journal.dropped trace),
86
+ // because losing replay data beats crashing the whole session.
87
+ // Soft disk cap for the on-disk change journal: when unacked journal
88
+ // bytes exceed it (a backend down for a very long time), capture stops
89
+ // loudly instead of filling the disk. The end snapshot still ships.
90
+ export const JOURNAL_DISK_CAP_BYTES = 1024 * 1024 * 1024;
91
+ /**
92
+ * Non-enumerable per-record field carrying the raw serialized line
93
+ * length (for journal ack accounting) — invisible to JSON.stringify so
94
+ * the changes.json sidecar never includes it.
95
+ */
96
+ export const RAW_BYTES_FIELD = Symbol("omnirushJournalRawBytes");
74
97
  // Multi-file parts target this compressed size (the backend body rail
75
98
  // was live-verified at >= 31.5 MiB compressed, so 15 MiB leaves ample
76
99
  // headroom for any ingress in front of the manager).
@@ -83,8 +106,10 @@ export const PART_UNCOMPRESSED_STEP = 8 * 1024 * 1024;
83
106
  // Flush a part when a check shows at least this much compressed payload.
84
107
  export const PART_READY_COMPRESSED_BYTES = 12 * 1024 * 1024;
85
108
  // Absolute uncompressed ceiling for one part batch — bounds resident
86
- // memory even for highly-compressible content.
87
- export const PART_BATCH_UNCOMPRESSED_MAX = 48 * 1024 * 1024;
109
+ // memory even for highly-compressible content (incompressible batches
110
+ // never hit the compressed-size thresholds, so this is what actually
111
+ // stops accumulation; the payload build doubles it transiently).
112
+ export const PART_BATCH_UNCOMPRESSED_MAX = 24 * 1024 * 1024;
88
113
  // Change/trace parts flush by PLAIN size so their __agent__/changes.json
89
114
  // / trace.json entries (single file entries whose content grows with the
90
115
  // batch) stay small and parts upload promptly.
@@ -176,8 +201,28 @@ type SessionState = {
176
201
  * fabricate deletions inside the workspace. */
177
202
  knownPaths: Set<string>;
178
203
  trace: TraceEvent[];
179
- /** Ordered change records — the replayable change log. */
180
- changeJournal: ChangeJournalEntry[];
204
+ /**
205
+ * The change journal lives ON DISK (ndjson sidecar, one JSON record
206
+ * per line) — never as an in-memory array (issue #6: the array held
207
+ * every changed file's content string and grew to multiple GB).
208
+ * Captures APPEND to the file; change uploads stream the unacked
209
+ * region and compact the acked prefix away. Appends during an upload
210
+ * simply land past the read window (bounded by `journalUploadEof`) —
211
+ * no gating, no memory growth.
212
+ */
213
+ journalPath: string | null;
214
+ /** File offset of the first not-yet-uploaded record. */
215
+ journalAckOffset: number;
216
+ /** Bytes appended so far (soft disk cap accounting). */
217
+ journalAppendedBytes: number;
218
+ /** Serializes journal appends. */
219
+ journalAppendTail: Promise<void>;
220
+ /** Set while a change upload reads the journal. */
221
+ journalUploadEof: number | null;
222
+ /** True once the soft disk cap stopped captures (warned once). */
223
+ journalCaptureStopped: boolean;
224
+ /** Last journal signature per path (consecutive-duplicate suppression). */
225
+ journalLastByPath: Map<string, string>;
181
226
  changeCaptureTail: Promise<void>;
182
227
  ready: Promise<void>;
183
228
  tail: Promise<void>;
@@ -394,7 +439,7 @@ export function selectTraceEvents(
394
439
  * removed on purpose: `git ls-files` drops anything .gitignored (build
395
440
  * outputs, node_modules), which broke the whole-project guarantee.
396
441
  */
397
- async function listWorkspaceFiles(root: string): Promise<string[]> {
442
+ export async function listWorkspaceFiles(root: string): Promise<string[]> {
398
443
  return walkWorkspace(root);
399
444
  }
400
445
 
@@ -528,19 +573,23 @@ function sliceManifest(
528
573
  type ContentChunk = { content: string; encoding?: "base64" };
529
574
 
530
575
  /**
531
- * Read a regular file into <= ~8 MiB content chunks: small files in one
532
- * whole read (fully redacted), big files streamed in fixed raw chunks —
533
- * strings stay far below the runtime string limit regardless of file
534
- * size. Binary detection happens on the first chunk and applies to the
535
- * whole file.
576
+ * Stream a regular file as <= ~8 MiB content chunks, yielding each chunk
577
+ * the moment it is produced — a file's content is NEVER materialized as
578
+ * one string regardless of size (issue #6: the old array version held
579
+ * every chunk at once). Small files are a single whole read; big files
580
+ * stream in fixed raw chunks. Binary detection happens on the first
581
+ * chunk and applies to the whole file; text is redacted per
582
+ * line-aligned segment.
536
583
  */
537
- async function readContentChunks(absolute: string, size: number): Promise<ContentChunk[]> {
584
+ async function* iterateContentChunks(absolute: string, size: number): AsyncGenerator<ContentChunk> {
538
585
  if (size <= WHOLE_READ_BYTES) {
539
586
  const buffer = await readFile(absolute);
540
587
  if (isBinary(buffer)) {
541
- return [{ content: buffer.toString("base64"), encoding: "base64" }];
588
+ yield { content: buffer.toString("base64"), encoding: "base64" };
589
+ return;
542
590
  }
543
- return [{ content: redactCollectorText(buffer.toString("utf8")).text }];
591
+ yield { content: redactCollectorText(buffer.toString("utf8")).text };
592
+ return;
544
593
  }
545
594
  const { open } = await import("node:fs/promises");
546
595
  const handle = await open(absolute, "r");
@@ -549,114 +598,139 @@ async function readContentChunks(absolute: string, size: number): Promise<Conten
549
598
  const buffer = Buffer.alloc(chunkBytes);
550
599
  const first = await handle.read(buffer, 0, chunkBytes, null);
551
600
  const binary = isBinary(buffer.subarray(0, first.bytesRead));
552
- const chunks: ContentChunk[] = [];
553
601
  if (binary) {
554
602
  // Binaries carry no redactable text: base64 each raw chunk.
555
- chunks.push({ content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" });
603
+ yield { content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" };
556
604
  let read = 0;
557
605
  while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
558
- chunks.push({ content: buffer.subarray(0, read).toString("base64"), encoding: "base64" });
606
+ yield { content: buffer.subarray(0, read).toString("base64"), encoding: "base64" };
559
607
  }
560
- } else {
561
- // Text: decode incrementally (StringDecoder absorbs multibyte
562
- // sequences split across chunk boundaries), emit only up to the
563
- // last complete line, carry the remainder, and redact each
564
- // line-aligned segment.
565
- const { StringDecoder } = await import("node:string_decoder");
566
- const decoder = new StringDecoder("utf8");
567
- let pending = decoder.write(buffer.subarray(0, first.bytesRead));
568
- let read = 0;
569
- while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
570
- pending += decoder.write(buffer.subarray(0, read));
608
+ return;
609
+ }
610
+ // Text: decode incrementally (StringDecoder absorbs multibyte
611
+ // sequences split across chunk boundaries), emit only up to the
612
+ // last complete line, carry the remainder, and redact each
613
+ // line-aligned segment. A pathological newline-less file must not
614
+ // grow `pending` forever: at PENDING_TEXT_CAP_BYTES the segment is
615
+ // flushed mid-line (line-anchored redaction weakens at that rare
616
+ // cut; unbounded memory is worse).
617
+ const { StringDecoder } = await import("node:string_decoder");
618
+ const decoder = new StringDecoder("utf8");
619
+ let pending = decoder.write(buffer.subarray(0, first.bytesRead));
620
+ let read = 0;
621
+ while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
622
+ pending += decoder.write(buffer.subarray(0, read));
623
+ for (;;) {
571
624
  const lastNl = pending.lastIndexOf("\n");
572
625
  if (lastNl >= 0) {
573
- chunks.push({ content: redactCollectorText(pending.slice(0, lastNl + 1)).text });
626
+ yield { content: redactCollectorText(pending.slice(0, lastNl + 1)).text };
574
627
  pending = pending.slice(lastNl + 1);
628
+ continue;
629
+ }
630
+ if (pending.length >= PENDING_TEXT_CAP_BYTES) {
631
+ yield { content: redactCollectorText(pending).text };
632
+ pending = "";
575
633
  }
634
+ break;
576
635
  }
577
- pending += decoder.end();
578
- if (pending) chunks.push({ content: redactCollectorText(pending).text });
579
636
  }
580
- return chunks;
637
+ pending += decoder.end();
638
+ if (pending) yield { content: redactCollectorText(pending).text };
581
639
  } finally {
582
640
  await handle.close();
583
641
  }
584
642
  }
585
643
 
586
644
  /**
587
- * Read one workspace entry into CollectorFile entries. Regular files:
645
+ * Stream one workspace entry as CollectorFile entries. Regular files:
588
646
  * text is redacted, binaries base64-encoded ("encoding": "base64"), and
589
647
  * the POSIX mode rides along (octal string) so replay can restore the
590
648
  * executable bit (stored-and-ignored on Windows). Content bigger than
591
- * MAX_SLICE_BYTES is sliced (see buildFileEntries). Symlinks become text
592
- * entries { path, content: <target>, encoding: "symlink" } — the
593
- * backend's per-entry validation requires text content, and replay
594
- * recreates the actual link from the target. Returns [] when the entry
595
- * vanished or is a special file.
649
+ * MAX_SLICE_BYTES is sliced on the fly — slices are yielded the moment
650
+ * they exist while only the tiny sha256 descriptors accumulate, then
651
+ * the manifest is yielded LAST (replay unions manifests by index, so
652
+ * position never matters). Symlinks become text entries
653
+ * { path, content: <target>, encoding: "symlink" } — the backend's
654
+ * per-entry validation requires text content, and replay recreates the
655
+ * actual link from the target. Yields nothing when the entry vanished
656
+ * or is a special file.
596
657
  */
597
- async function readWorkspaceEntries(root: string, path: string): Promise<CollectorFile[]> {
658
+ async function* iterateWorkspaceEntries(root: string, path: string): AsyncGenerator<CollectorFile> {
659
+ let absolute = "";
598
660
  try {
599
- const absolute = resolve(root, path);
600
- if (portablePath(root, absolute).startsWith("../")) return [];
661
+ absolute = resolve(root, path);
662
+ if (portablePath(root, absolute).startsWith("../")) return;
601
663
  const file = await lstat(absolute);
602
664
  if (file.isSymbolicLink()) {
603
665
  const target = await readlink(absolute);
604
- return [{ path, content: target, encoding: "symlink" }];
666
+ yield { path, content: target, encoding: "symlink" };
667
+ return;
605
668
  }
606
- if (!file.isFile()) return [];
669
+ if (!file.isFile()) return;
607
670
  const mode = file.mode & 0o777;
608
- const chunks: ContentChunk[] = await readContentChunks(absolute, file.size);
609
- if (chunks.length === 1) {
610
- const chunk = chunks[0];
611
- return buildFileEntries({
612
- path,
613
- content: chunk.content,
614
- ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
615
- ...(mode ? { mode: mode.toString(8) } : {}),
616
- });
671
+ const modeField = mode ? { mode: mode.toString(8) } : {};
672
+ if (file.size <= WHOLE_READ_BYTES) {
673
+ // Small file: whole read, real path preserved; buildFileEntries
674
+ // slices (manifest + parts) when the encoded content exceeds
675
+ // MAX_SLICE_BYTES — same shape as always.
676
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
677
+ for (const entry of buildFileEntries({
678
+ path,
679
+ content: chunk.content,
680
+ ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
681
+ ...modeField,
682
+ })) {
683
+ yield entry;
684
+ }
685
+ }
686
+ return;
687
+ }
688
+ // Big file (streamed): slices are yielded the moment they exist;
689
+ // only the tiny sha256 descriptors accumulate. The reassembly
690
+ // manifest is yielded LAST (replay unions manifests by index, so
691
+ // position never matters).
692
+ let index = 0;
693
+ const descriptors: Array<{ index: number; bytes: number; sha256: string }> = [];
694
+ let sawEncoding: "base64" | undefined;
695
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
696
+ sawEncoding ??= chunk.encoding;
697
+ for (const slice of sliceString(chunk.content)) {
698
+ descriptors.push({ index, bytes: Buffer.byteLength(slice), sha256: sha256Hex(slice) });
699
+ yield {
700
+ path: slicePartPath(path, index),
701
+ content: slice,
702
+ ...(sawEncoding ? { encoding: sawEncoding } : {}),
703
+ };
704
+ index += 1;
705
+ }
617
706
  }
618
- // Big file: streamed chunks -> one logical entry per chunk (already
619
- // <= ~8 MiB for raw chunks; base64 inflates 4/3, so sub-slice those),
620
- // plus the manifest for deterministic reassembly.
621
- const encoding = chunks[0].encoding;
622
- const entries: CollectorFile[] = [];
623
- const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
624
- entries.push({
707
+ if (index === 0) return;
708
+ yield {
625
709
  path: sliceManifestPath(path),
626
710
  content: JSON.stringify(sliceManifest(
627
- {
628
- path,
629
- ...(encoding ? { encoding } : {}),
630
- ...(mode ? { mode: mode.toString(8) } : {}),
631
- },
632
- sliceContents.map((slice, index) => ({
633
- index,
634
- bytes: Buffer.byteLength(slice),
635
- sha256: sha256Hex(slice),
636
- })),
711
+ { path, ...(sawEncoding ? { encoding: sawEncoding } : {}), ...modeField },
712
+ descriptors,
637
713
  )),
638
- });
639
- sliceContents.forEach((slice, index) => {
640
- entries.push({
641
- path: slicePartPath(path, index),
642
- content: slice,
643
- ...(encoding ? { encoding } : {}),
644
- });
645
- });
646
- return entries;
714
+ };
715
+ return;
647
716
  } catch {
648
717
  // Workspaces are live; races are expected and retried by the next snapshot.
649
- return [];
718
+ return;
650
719
  }
651
720
  }
652
721
 
653
- export async function collectFiles(
722
+ /**
723
+ * Streaming file collection: yields workspace files (metadata entry
724
+ * first) one at a time so the one-shot path never holds the whole tree's
725
+ * content in memory (issue #6). Same order, denylist, and byte budget
726
+ * as the compat collectFiles drain below.
727
+ */
728
+ export async function* iterateCollectFiles(
654
729
  root: string,
655
730
  byteLimit: number,
656
731
  workspaceId: string,
657
732
  session?: Pick<SessionState, "id" | "segment" | "resumed">,
658
- ): Promise<CollectorFile[]> {
659
- const files: CollectorFile[] = [];
733
+ ): AsyncGenerator<CollectorFile> {
660
734
  let used = 0;
661
735
  const metadata = JSON.stringify({
662
736
  workspace_id: workspaceId,
@@ -668,19 +742,32 @@ export async function collectFiles(
668
742
  root_name: root.split(sep).filter(Boolean).at(-1) ?? "workspace",
669
743
  git: await gitMetadata(root),
670
744
  });
671
- files.push({ path: `${AGENT_DIR}/workspace.json`, content: metadata });
745
+ yield { path: `${AGENT_DIR}/workspace.json`, content: metadata };
672
746
  used += Buffer.byteLength(metadata);
673
747
 
674
748
  const paths = (await listWorkspaceFiles(root)).sort(comparePaths);
675
749
  for (const path of paths) {
676
750
  if (used >= byteLimit) break;
677
- for (const file of await readWorkspaceEntries(root, path)) {
751
+ for await (const file of iterateWorkspaceEntries(root, path)) {
678
752
  const size = Buffer.byteLength(file.content ?? "");
679
753
  if (used + size > byteLimit) continue;
680
- files.push(file);
754
+ yield file;
681
755
  used += size;
682
756
  }
683
757
  }
758
+ }
759
+
760
+ /** Compat drain of iterateCollectFiles (tests + tooling). */
761
+ export async function collectFiles(
762
+ root: string,
763
+ byteLimit: number,
764
+ workspaceId: string,
765
+ session?: Pick<SessionState, "id" | "segment" | "resumed">,
766
+ ): Promise<CollectorFile[]> {
767
+ const files: CollectorFile[] = [];
768
+ for await (const file of iterateCollectFiles(root, byteLimit, workspaceId, session)) {
769
+ files.push(file);
770
+ }
684
771
  return files;
685
772
  }
686
773
 
@@ -724,14 +811,11 @@ export type EnvelopePart = {
724
811
  };
725
812
 
726
813
  /**
727
- * Deterministic part split for a full files array (used by `omnirush
728
- * collect` and the tests): sort files by path, accumulate in order, keep
729
- * each part <= maxCompressed compressed. A single file bigger than one
730
- * part gets its own part; file content is never split across parts. A
731
- * lone file that cannot fit even the hard rail is skipped with a warning.
732
- * Envelope sequence numbers are state.sequence + 1 + partIndex.
814
+ * Deterministic part split for a full files array (compat drain of
815
+ * iterateEnvelopeParts, used by the tests): sort files by path,
816
+ * accumulate in order, keep each part <= maxCompressed compressed.
733
817
  */
734
- export function buildEnvelopeParts(
818
+ export async function buildEnvelopeParts(
735
819
  state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
736
820
  snapshotType: SnapshotType,
737
821
  files: CollectorFile[],
@@ -744,28 +828,34 @@ export function buildEnvelopeParts(
744
828
  const sorted = files
745
829
  .flatMap((file) => buildFileEntries(file))
746
830
  .sort((left, right) => comparePaths(left.path, right.path));
747
- return packEnvelopeParts(state, snapshotType, sorted, maxCompressed, log);
831
+ const parts: EnvelopePart[] = [];
832
+ for await (const part of iterateEnvelopeParts(state, snapshotType, sorted, maxCompressed, log)) {
833
+ parts.push(part);
834
+ }
835
+ return parts;
748
836
  }
749
837
 
750
838
  /**
751
- * Shared greedy packer: split an ordered file list into compressed parts,
752
- * preserving path order across parts. Used by buildEnvelopeParts (pure,
753
- * whole list in memory) and mirrored by the collector's streaming packer
754
- * (packAndUpload, which never holds more than one part in memory).
839
+ * Streaming packer for the one-shot path: yields compressed parts in
840
+ * order so callers can upload-and-discard instead of holding the whole
841
+ * workspace's parts (payload + compressed buffers) in memory (issue #6).
842
+ * Same greedy algorithm as before: split at the compressed target, a
843
+ * lone file may use the hard rail, lone-over-rail files are skipped
844
+ * loudly. File content is never split across parts.
755
845
  */
756
- export async function packEnvelopeParts(
846
+ export async function* iterateEnvelopeParts(
757
847
  state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
758
848
  snapshotType: SnapshotType,
759
- sortedFiles: CollectorFile[],
849
+ sortedFiles: Iterable<CollectorFile> | AsyncIterable<CollectorFile>,
760
850
  maxCompressed = MAX_PART_COMPRESSED_BYTES,
761
851
  log?: (message: string, attributes?: Record<string, unknown>) => void,
762
- ): Promise<EnvelopePart[]> {
852
+ ): AsyncGenerator<EnvelopePart> {
763
853
  const hard = Math.max(maxCompressed, MAX_PART_COMPRESSED_HARD);
764
854
  const measure = async (batch: CollectorFile[], partIndex: number): Promise<EnvelopePart> => {
765
855
  const payload = buildEnvelopePayload({ ...state, sequence: (state.sequence ?? 0) + partIndex }, snapshotType, batch);
766
856
  return { files: batch, payload, compressed: await compressZstd(payload) };
767
857
  };
768
- const parts: EnvelopePart[] = [];
858
+ let partsYielded = 0;
769
859
  let batch: CollectorFile[] = [];
770
860
  const fit = async (batchToFit: CollectorFile[], partIndex: number) => {
771
861
  let measured = await measure(batchToFit, partIndex);
@@ -787,15 +877,16 @@ export async function packEnvelopeParts(
787
877
  }
788
878
  return { measured, carry };
789
879
  };
790
- for (const file of sortedFiles) {
880
+ for await (const file of sortedFiles) {
791
881
  batch.push(file);
792
- const partIndex = parts.length;
882
+ const partIndex = partsYielded;
793
883
  const estimate = batch.reduce((total, f) => total + Buffer.byteLength(f.content), 0);
794
884
  if (estimate < PART_UNCOMPRESSED_STEP) continue;
795
885
  const { measured, carry } = await fit(batch, partIndex);
796
886
  if (measured
797
887
  && (measured.compressed.length >= PART_READY_COMPRESSED_BYTES || estimate >= PART_BATCH_UNCOMPRESSED_MAX)) {
798
- parts.push(measured);
888
+ partsYielded += 1;
889
+ yield measured;
799
890
  batch = carry;
800
891
  } else if (measured) {
801
892
  batch = [...batch, ...carry];
@@ -804,12 +895,14 @@ export async function packEnvelopeParts(
804
895
  }
805
896
  }
806
897
  while (batch.length > 0) {
807
- const { measured, carry } = await fit(batch, parts.length);
808
- if (measured) parts.push(measured);
898
+ const { measured, carry } = await fit(batch, partsYielded);
899
+ if (measured) {
900
+ partsYielded += 1;
901
+ yield measured;
902
+ }
809
903
  if (carry.length === 0) break;
810
904
  batch = carry;
811
905
  }
812
- return parts;
813
906
  }
814
907
 
815
908
  export class WorkspaceCollector {
@@ -840,6 +933,7 @@ export class WorkspaceCollector {
840
933
  private readonly uploader?: CollectorOptions["upload"];
841
934
  private readonly log: NonNullable<CollectorOptions["log"]>;
842
935
  private readonly ledgerPath: string | null;
936
+ private readonly stateDir: string | null;
843
937
  private readonly ledgerReady: Promise<void>;
844
938
  private ledger: SessionLedger = { version: 1, sessions: {} };
845
939
  private ledgerWriteTail: Promise<void> = Promise.resolve();
@@ -857,6 +951,7 @@ export class WorkspaceCollector {
857
951
  this.refresh = options.refresh;
858
952
  this.uploader = options.upload;
859
953
  this.log = options.log ?? (() => undefined);
954
+ this.stateDir = options.stateDir ? resolve(options.stateDir) : null;
860
955
  this.ledgerPath = options.stateDir ? join(resolve(options.stateDir), SESSION_LEDGER_FILE) : null;
861
956
  this.ledgerReady = this.loadLedger();
862
957
  this.changeDebounceMs = options.changeDebounceMs ?? CHANGE_DEBOUNCE_MS;
@@ -966,7 +1061,13 @@ export class WorkspaceCollector {
966
1061
  watcherStartedAtMs: 0,
967
1062
  knownPaths: new Set(),
968
1063
  trace: [],
969
- changeJournal: [],
1064
+ journalPath: this.stateDir ? join(resolve(this.stateDir), `journal-${sessionId}.ndjson`) : null,
1065
+ journalAckOffset: 0,
1066
+ journalAppendedBytes: 0,
1067
+ journalAppendTail: Promise.resolve(),
1068
+ journalUploadEof: null,
1069
+ journalCaptureStopped: false,
1070
+ journalLastByPath: new Map(),
970
1071
  changeCaptureTail: Promise.resolve(),
971
1072
  ready: Promise.resolve(),
972
1073
  tail: Promise.resolve(),
@@ -1031,6 +1132,7 @@ export class WorkspaceCollector {
1031
1132
  if (state.trace.length > 0) pendingTrace.push(...state.trace.splice(0));
1032
1133
  await state.changeCaptureTail;
1033
1134
  if (finalTrace !== undefined) pendingTrace.push({ at: new Date().toISOString(), type: "session.completed", data: finalTrace });
1135
+ await this.teardownJournal(state); // ship unacked change records, then unlink
1034
1136
  await this.uploadTrace(state, pendingTrace);
1035
1137
  await this.uploadWorkspace(state, "end");
1036
1138
  this.sessions.delete(sessionId);
@@ -1079,18 +1181,28 @@ export class WorkspaceCollector {
1079
1181
  });
1080
1182
  }
1081
1183
 
1184
+ /** Run the debounced change-snapshot upload immediately. */
1185
+ private flushChangeSnapshot(state: SessionState): void {
1186
+ if (state.finished || state.budgetExhausted) return;
1187
+ if (state.changeTimer) {
1188
+ clearTimeout(state.changeTimer);
1189
+ state.changeTimer = null;
1190
+ }
1191
+ this.enqueue(state, async () => {
1192
+ if (state.budgetExhausted || state.finished) return;
1193
+ const signature = await workspaceSignature(state.root);
1194
+ if (!signature || signature === state.lastSignature) return;
1195
+ state.lastSignature = signature;
1196
+ await this.uploadWorkspace(state, "change");
1197
+ });
1198
+ }
1199
+
1082
1200
  private scheduleChange(state: SessionState): void {
1083
1201
  if (state.finished || state.budgetExhausted) return;
1084
1202
  if (state.changeTimer) clearTimeout(state.changeTimer);
1085
1203
  state.changeTimer = setTimeout(() => {
1086
1204
  state.changeTimer = null;
1087
- this.enqueue(state, async () => {
1088
- if (state.budgetExhausted || state.finished) return;
1089
- const signature = await workspaceSignature(state.root);
1090
- if (!signature || signature === state.lastSignature) return;
1091
- state.lastSignature = signature;
1092
- await this.uploadWorkspace(state, "change");
1093
- });
1205
+ this.flushChangeSnapshot(state);
1094
1206
  }, this.changeDebounceMs);
1095
1207
  state.changeTimer.unref?.();
1096
1208
  }
@@ -1149,38 +1261,38 @@ export class WorkspaceCollector {
1149
1261
  const at = new Date().toISOString();
1150
1262
  const mode = file.mode & 0o777;
1151
1263
  const modeField = mode ? { mode: mode.toString(8) } : {};
1152
- const chunks = await readContentChunks(absolute, file.size);
1153
- const encoding = chunks[0].encoding;
1154
- const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
1155
- if (sliceContents.length === 1) {
1156
- entry = {
1157
- path,
1158
- at,
1159
- status: "present",
1160
- content: sliceContents[0],
1161
- ...(encoding ? { encoding } : {}),
1162
- ...modeField,
1163
- };
1164
- } else {
1165
- // Big file: one ordered record per slice so the journal stays
1166
- // replayable; reassembly rides the slice metadata.
1167
- sliceContents.forEach((content, index) => {
1264
+ // STREAMED capture: each chunk becomes a journal record the
1265
+ // moment it exists — a big file is never held in memory whole
1266
+ // (issue #6). One record per <= MAX_SLICE_BYTES slice; slice
1267
+ // metadata rides the records only when reassembly is needed.
1268
+ let sliceCount = 0;
1269
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
1270
+ for (const content of sliceString(chunk.content)) {
1271
+ sliceCount += 1;
1168
1272
  records.push({
1169
1273
  path,
1170
1274
  at,
1171
1275
  status: "present",
1172
1276
  content,
1173
- ...(encoding ? { encoding } : {}),
1277
+ ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
1174
1278
  ...modeField,
1175
- slice: {
1176
- index,
1177
- total: sliceContents.length,
1178
- bytes: Buffer.byteLength(content),
1179
- sha256: sha256Hex(content),
1180
- },
1181
1279
  });
1280
+ }
1281
+ }
1282
+ if (sliceCount > 1) {
1283
+ records.forEach((record, index) => {
1284
+ record.slice = {
1285
+ index,
1286
+ total: sliceCount,
1287
+ bytes: Buffer.byteLength(record.content ?? ""),
1288
+ sha256: sha256Hex(record.content ?? ""),
1289
+ };
1182
1290
  });
1183
- state.changeJournal.push(...records);
1291
+ }
1292
+ if (sliceCount === 1) {
1293
+ entry = records.pop()!;
1294
+ } else {
1295
+ await this.appendJournalRecords(state, records);
1184
1296
  return;
1185
1297
  }
1186
1298
  }
@@ -1199,27 +1311,156 @@ export class WorkspaceCollector {
1199
1311
  // Consecutive-duplicate suppression: macOS FSEvents can deliver a
1200
1312
  // late event for a path that was also captured directly, and an
1201
1313
  // identical re-capture adds upload fat without a state change. The
1202
- // replay result is identical either way.
1314
+ // replay result is identical either way. The last signature per
1315
+ // path lives in a tiny in-memory map (path strings only — content
1316
+ // is already on disk in the journal).
1203
1317
  const journalSignature = (record: ChangeJournalEntry) =>
1204
1318
  JSON.stringify([record.status, record.content, record.encoding, record.mode]);
1205
1319
  const pendingRecords = [entry, ...records];
1320
+ const accepted: ChangeJournalEntry[] = [];
1206
1321
  for (const record of pendingRecords) {
1207
- let duplicate = false;
1208
- for (let i = state.changeJournal.length - 1, scanned = 0; i >= 0 && scanned < 100; i--, scanned++) {
1209
- const prior = state.changeJournal[i];
1210
- if (prior.path !== record.path) continue;
1211
- duplicate = journalSignature(prior) === journalSignature(record);
1212
- break;
1322
+ const signature = journalSignature(record);
1323
+ if (state.journalLastByPath.get(record.path) === signature) continue;
1324
+ state.journalLastByPath.set(record.path, signature);
1325
+ accepted.push(record);
1326
+ }
1327
+ await this.appendJournalRecords(state, accepted);
1328
+ }
1329
+
1330
+ /**
1331
+ * APPEND records to the on-disk journal (ndjson — one JSON record per
1332
+ * line, 0600). Returns immediately when the capture hit the soft disk
1333
+ * cap (a backend down for a very long time must stop CAPTURING loudly
1334
+ * rather than fill the disk or RAM — the end snapshot still ships the
1335
+ * final tree). Serialized through the per-session append chain.
1336
+ */
1337
+ private appendJournalRecords(state: SessionState, records: ChangeJournalEntry[]): Promise<void> {
1338
+ if (!state.journalPath || records.length === 0) return Promise.resolve();
1339
+ if (state.journalCaptureStopped) return Promise.resolve();
1340
+ const unacked = state.journalAppendedBytes - state.journalAckOffset;
1341
+ const capMb = Number(process.env.OMNIRUSH_JOURNAL_DISK_CAP_MB);
1342
+ const capBytes = Number.isFinite(capMb) && capMb > 0 ? capMb * 1024 * 1024 : JOURNAL_DISK_CAP_BYTES;
1343
+ if (unacked > capBytes) {
1344
+ state.journalCaptureStopped = true;
1345
+ this.log("warn", "OmniRush change journal hit the disk cap — change capture stopped for this session", {
1346
+ sessionId: state.id,
1347
+ capBytes: capBytes,
1348
+ unackedBytes: unacked,
1349
+ });
1350
+ state.trace.push({
1351
+ at: new Date().toISOString(),
1352
+ type: "journal.capture_stopped",
1353
+ data: { cap_bytes: capBytes, unacked_bytes: unacked },
1354
+ });
1355
+ return Promise.resolve();
1356
+ }
1357
+ const lines = records.map((record) => JSON.stringify(record) + "\n").join("");
1358
+ const bytes = Buffer.byteLength(lines);
1359
+ state.journalAppendTail = state.journalAppendTail
1360
+ .catch(() => undefined)
1361
+ .then(async () => {
1362
+ const { appendFile, mkdir } = await import("node:fs/promises");
1363
+ await mkdir(dirname(state.journalPath!), { recursive: true, mode: 0o700 });
1364
+ await appendFile(state.journalPath!, lines, { encoding: "utf8", mode: 0o600 });
1365
+ state.journalAppendedBytes += bytes;
1366
+ });
1367
+ return state.journalAppendTail;
1368
+ }
1369
+
1370
+ /**
1371
+ * Stream the journal's unacked records (from journalAckOffset to the
1372
+ * upload-start EOF) as parsed entries, each tagged with its raw line
1373
+ * bytes via a NON-enumerable property so the changes.json sidecar
1374
+ * serialization never sees it. Records appended while the upload runs
1375
+ * land beyond the read window and simply ship with the next snapshot.
1376
+ */
1377
+ private async *readJournalRecords(state: SessionState): AsyncGenerator<ChangeJournalEntry> {
1378
+ if (!state.journalPath || state.journalAckOffset >= state.journalUploadEof) return;
1379
+ const { createReadStream } = await import("node:fs");
1380
+ const readline = await import("node:readline");
1381
+ const stream = createReadStream(state.journalPath, {
1382
+ start: state.journalAckOffset,
1383
+ end: state.journalUploadEof! - 1,
1384
+ encoding: "utf8",
1385
+ });
1386
+ const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
1387
+ for await (const line of rl) {
1388
+ if (!line.trim()) continue;
1389
+ try {
1390
+ const record = JSON.parse(line) as ChangeJournalEntry;
1391
+ Object.defineProperty(record, RAW_BYTES_FIELD, {
1392
+ value: Buffer.byteLength(line) + 1,
1393
+ enumerable: false,
1394
+ configurable: true,
1395
+ });
1396
+ yield record;
1397
+ } catch {
1398
+ this.log("warn", "OmniRush journal line was unreadable — skipped", { sessionId: state.id });
1213
1399
  }
1214
- if (!duplicate) state.changeJournal.push(record);
1215
1400
  }
1216
1401
  }
1217
1402
 
1218
- /** Remove exactly the uploaded entries from the ordered journal. */
1219
- private acknowledgeJournal(state: SessionState, uploaded: ChangeJournalEntry[]): void {
1220
- if (uploaded.length === 0) return;
1221
- const sent = uploaded.length > 32 ? new Set(uploaded) : null;
1222
- state.changeJournal = state.changeJournal.filter((entry) => sent ? !sent.has(entry) : !uploaded.includes(entry));
1403
+ /**
1404
+ * Upload the journal's unacked records as change snapshots, then
1405
+ * compact: everything acked is removed from the file; records appended
1406
+ * during the upload (past journalUploadEof) survive untouched.
1407
+ */
1408
+ private async uploadChangeJournal(state: SessionState): Promise<void> {
1409
+ if (!state.journalPath) return;
1410
+ let fileSize = 0;
1411
+ try {
1412
+ fileSize = (await lstat(state.journalPath)).size;
1413
+ } catch {
1414
+ return; // no journal yet — nothing captured
1415
+ }
1416
+ if (fileSize <= state.journalAckOffset) return;
1417
+ const ackStart = state.journalAckOffset;
1418
+ state.journalUploadEof = fileSize;
1419
+ try {
1420
+ await this.uploadChangeParts(state, this.readJournalRecords(state), {
1421
+ onAck: (batch) => {
1422
+ for (const record of batch) {
1423
+ const raw = (record as Record<symbol, unknown>)[RAW_BYTES_FIELD];
1424
+ if (typeof raw === "number") state.journalAckOffset += raw;
1425
+ }
1426
+ },
1427
+ });
1428
+ } finally {
1429
+ state.journalUploadEof = null;
1430
+ }
1431
+ if (state.journalAckOffset <= ackStart) return; // nothing acked — keep everything
1432
+ await this.compactJournal(state, state.journalAckOffset);
1433
+ state.journalAckOffset = 0;
1434
+ }
1435
+
1436
+ /** Rewrite the journal without its first `dropBytes` bytes. */
1437
+ private async compactJournal(state: SessionState, dropBytes: number): Promise<void> {
1438
+ if (!state.journalPath) return;
1439
+ const { createReadStream, createWriteStream } = await import("node:fs");
1440
+ const { rename } = await import("node:fs/promises");
1441
+ const tmpPath = `${state.journalPath}.compact`;
1442
+ await new Promise<void>((resolvePromise, rejectPromise) => {
1443
+ const input = createReadStream(state.journalPath!, { start: dropBytes, encoding: "utf8" });
1444
+ const output = createWriteStream(tmpPath, { encoding: "utf8", mode: 0o600 });
1445
+ input.on("error", rejectPromise);
1446
+ output.on("error", rejectPromise);
1447
+ output.on("finish", () => resolvePromise());
1448
+ input.pipe(output);
1449
+ });
1450
+ await rename(tmpPath, state.journalPath);
1451
+ }
1452
+
1453
+ /** Upload unacked journal records (best effort) and remove the file. */
1454
+ private async teardownJournal(state: SessionState): Promise<void> {
1455
+ try {
1456
+ await this.uploadChangeJournal(state);
1457
+ } catch {
1458
+ /* best effort — the end snapshot still ships the final tree */
1459
+ }
1460
+ if (state.journalPath) {
1461
+ const { rm } = await import("node:fs/promises");
1462
+ await rm(state.journalPath, { force: true }).catch(() => undefined);
1463
+ }
1223
1464
  }
1224
1465
 
1225
1466
  private enqueue(state: SessionState, operation: () => Promise<void>): void {
@@ -1234,17 +1475,10 @@ export class WorkspaceCollector {
1234
1475
  private async uploadWorkspace(state: SessionState, type: "start" | "change" | "end"): Promise<void> {
1235
1476
  if (state.budgetExhausted) return;
1236
1477
  if (type === "change") {
1237
- // Change snapshots carry ONLY the touched files from the journal —
1238
- // never a full re-enumeration of the tree.
1239
- const snapshot = [...state.changeJournal];
1240
- if (snapshot.length === 0) {
1241
- state.lastSignature = await workspaceSignature(state.root);
1242
- return;
1243
- }
1244
- const uploaded = await this.uploadChangeParts(state, snapshot);
1245
- // Acknowledge only after every part of the snapshot succeeded; on a
1246
- // mid-budget exhaustion, acknowledge only what actually shipped.
1247
- this.acknowledgeJournal(state, state.budgetExhausted ? uploaded.flat() : snapshot);
1478
+ // Change snapshots carry ONLY the touched files — streamed from
1479
+ // the on-disk journal, acked per part, then compacted. Never a
1480
+ // full re-enumeration of the tree, never held in memory.
1481
+ await this.uploadChangeJournal(state);
1248
1482
  } else {
1249
1483
  await this.uploadTreeParts(state, type);
1250
1484
  }
@@ -1302,7 +1536,7 @@ export class WorkspaceCollector {
1302
1536
  private async *iterTreeFiles(state: SessionState, paths: string[]): AsyncGenerator<CollectorFile> {
1303
1537
  for (const path of paths) {
1304
1538
  if (state.budgetExhausted) return;
1305
- for (const entry of await readWorkspaceEntries(state.root, path)) {
1539
+ for await (const entry of iterateWorkspaceEntries(state.root, path)) {
1306
1540
  yield entry;
1307
1541
  }
1308
1542
  }
@@ -1315,7 +1549,11 @@ export class WorkspaceCollector {
1315
1549
  * records — replaying start + sidecars in sequence order reproduces the
1316
1550
  * end tree.
1317
1551
  */
1318
- private async uploadChangeParts(state: SessionState, snapshot: ChangeJournalEntry[]): Promise<ChangeJournalEntry[][]> {
1552
+ private async uploadChangeParts(
1553
+ state: SessionState,
1554
+ units: Iterable<ChangeJournalEntry> | AsyncIterable<ChangeJournalEntry>,
1555
+ opts: { onAck?: (batch: ChangeJournalEntry[]) => void } = {},
1556
+ ): Promise<ChangeJournalEntry[][]> {
1319
1557
  const uploadedSlices: ChangeJournalEntry[][] = [];
1320
1558
  const filesForSlice = (entries: ChangeJournalEntry[]): CollectorFile[] => {
1321
1559
  const lastByPath = new Map<string, CollectorFile>();
@@ -1372,7 +1610,7 @@ export class WorkspaceCollector {
1372
1610
  return files;
1373
1611
  };
1374
1612
  const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
1375
- await this.packAndUpload(state, "change", snapshot, {
1613
+ await this.packAndUpload(state, "change", units, {
1376
1614
  sizeOf: (entry) =>
1377
1615
  Buffer.byteLength(entry.content ?? "") + entry.path.length + 96,
1378
1616
  measure: async (batch) => {
@@ -1382,7 +1620,10 @@ export class WorkspaceCollector {
1382
1620
  },
1383
1621
  upload: async (measured, batch) => {
1384
1622
  const ok = await this.uploadEnvelope(state, "change", measured.files, measured);
1385
- if (ok) uploadedSlices.push(batch);
1623
+ if (ok) {
1624
+ uploadedSlices.push(batch);
1625
+ opts.onAck?.(batch);
1626
+ }
1386
1627
  return ok;
1387
1628
  },
1388
1629
  drop: (entry) => {
@@ -1579,14 +1820,46 @@ export class WorkspaceCollector {
1579
1820
  body: compressed.buffer.slice(compressed.byteOffset, compressed.byteOffset + compressed.byteLength) as ArrayBuffer,
1580
1821
  signal: AbortSignal.timeout(120_000),
1581
1822
  });
1582
- let response = await send();
1583
- if (response.status === 401 && this.refresh) {
1584
- // Broker pattern: single-flight refresh, retry exactly once.
1585
- const rotated = await this.refresh(this.token).catch(() => null);
1586
- if (rotated) {
1587
- this.token = rotated;
1588
- response = await send();
1589
- }
1823
+ // Shared retry layer: 429/5xx/network ride out backend deploys with
1824
+ // exponential backoff + jitter. 401 stays a protocol answer handled
1825
+ // inline (single-flight refresh, retry exactly once).
1826
+ const outcome = await withRetries(
1827
+ async () => {
1828
+ let response = await send();
1829
+ if (response.status === 401 && this.refresh) {
1830
+ // Broker pattern: single-flight refresh, retry exactly once.
1831
+ const rotated = await this.refresh(this.token).catch(() => null);
1832
+ if (rotated) {
1833
+ this.token = rotated;
1834
+ response = await send();
1835
+ }
1836
+ }
1837
+ return {
1838
+ response,
1839
+ retryAfterSec: Number(response?.headers?.get?.("retry-after")) || undefined,
1840
+ };
1841
+ },
1842
+ {
1843
+ attempts: retryAttempts(),
1844
+ isRetryable: ({ response }) => isRetryableStatus(response?.status),
1845
+ onRetry: ({ attempt, attempts, delayMs }) => {
1846
+ this.log("warn", "OmniRush upload unavailable — retrying", {
1847
+ sessionId: state.id,
1848
+ snapshotType,
1849
+ attempt: `${attempt + 1}/${attempts}`,
1850
+ delayMs,
1851
+ });
1852
+ },
1853
+ },
1854
+ );
1855
+ const response = outcome.result?.response;
1856
+ if (!response) {
1857
+ // Every attempt failed at the network layer.
1858
+ throw new Error(
1859
+ `collector upload failed: network error after ${outcome.attempts} attempt${outcome.attempts === 1 ? "" : "s"}` +
1860
+ ` — ${outcome.error?.message ?? "unknown"} [ref ${outcome.ref}]` +
1861
+ ` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
1862
+ );
1590
1863
  }
1591
1864
  if (!response.ok) {
1592
1865
  // Diagnosable from a tester's screenshot alone: status, context,
@@ -1596,6 +1869,7 @@ export class WorkspaceCollector {
1596
1869
  throw new Error(
1597
1870
  `collector upload failed: HTTP ${response.status}` +
1598
1871
  `${snippet ? ` — ${snippet}` : ""}` +
1872
+ ` [ref ${outcome.ref}]` +
1599
1873
  ` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
1600
1874
  );
1601
1875
  }