omnirush 0.3.3 → 0.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,7 +22,9 @@
22
22
  // node_modules, build outputs, .git (full reproducibility) — with
23
23
  // ONLY the secrets denylist (.env*, keys, credentials, private
24
24
  // keys) excluded. Binaries are captured (base64), symlinks are
25
- // stored as records (path + target) so replay recreates them, and
25
+ // stored as {path, content: <target>, encoding: "symlink"} so the
26
+ // backend's text-content validation accepts them and replay
27
+ // recreates them, and
26
28
  // the POSIX mode/executable bit rides each file entry for replay
27
29
  // (stored-and-ignored on Windows). Redaction also scrubs
28
30
  // credentials embedded in URLs (e.g. .git/config remotes).
@@ -50,6 +52,8 @@ import { dirname, join, relative, resolve, sep } from "node:path";
50
52
  import { promisify } from "node:util";
51
53
  import { zstdCompress as zstdCompressCb } from "node:zlib";
52
54
 
55
+ import { isRetryableStatus, retryAttempts, withRetries } from "./retry";
56
+
53
57
  const execFileAsync = promisify(execFile);
54
58
 
55
59
  // --- limits ---------------------------------------------------------------
@@ -65,10 +69,31 @@ export const MAX_SESSION_BYTES = 100 * 1024 * 1024 * 1024;
65
69
  // byte size + sha256) so replay reassembles the original deterministically.
66
70
  export const MAX_SLICE_BYTES = 4 * 1024 * 1024;
67
71
  // Whole-read ceiling. Files up to this size are read and processed as
68
- // one string (worst case: binary -> base64 inflates 4/3 to ~427 MiB
69
- // chars, comfortably under the runtime's ~512 MiB MAX_STRING_LENGTH).
70
- // Bigger files are streamed in fixed raw chunks and sliced.
71
- const WHOLE_READ_BYTES = 320 * 1024 * 1024;
72
+ // one string; bigger files stream in fixed raw chunks and slice.
73
+ // Deliberately tiny (issue #6): a whole read holds buffer + utf8 string
74
+ // + redaction copy + base64 simultaneously (~4-5x file size), so even a
75
+ // modest ceiling spikes RSS by that multiple per file. 8 MiB bounds the
76
+ // worst case to ~40 MiB; anything bigger takes the streaming path with
77
+ // identical output.
78
+ const WHOLE_READ_BYTES = 8 * 1024 * 1024;
79
+ // Text-segment cap for newline-less files (see the streaming reader).
80
+ const PENDING_TEXT_CAP_BYTES = 32 * 1024 * 1024;
81
+ // Change-journal memory ceiling (issue #6, Windows 8-9 GiB OOM): the
82
+ // journal holds every changed file's content between debounced change
83
+ // uploads. At HIGH_WATER an upload is forced immediately (normal
84
+ // backpressure); at the HARD ceiling — a backend down for a long time —
85
+ // the OLDEST records are dropped, loudly (warn + journal.dropped trace),
86
+ // because losing replay data beats crashing the whole session.
87
+ // Soft disk cap for the on-disk change journal: when unacked journal
88
+ // bytes exceed it (a backend down for a very long time), capture stops
89
+ // loudly instead of filling the disk. The end snapshot still ships.
90
+ export const JOURNAL_DISK_CAP_BYTES = 1024 * 1024 * 1024;
91
+ /**
92
+ * Non-enumerable per-record field carrying the raw serialized line
93
+ * length (for journal ack accounting) — invisible to JSON.stringify so
94
+ * the changes.json sidecar never includes it.
95
+ */
96
+ export const RAW_BYTES_FIELD = Symbol("omnirushJournalRawBytes");
72
97
  // Multi-file parts target this compressed size (the backend body rail
73
98
  // was live-verified at >= 31.5 MiB compressed, so 15 MiB leaves ample
74
99
  // headroom for any ingress in front of the manager).
@@ -81,8 +106,10 @@ export const PART_UNCOMPRESSED_STEP = 8 * 1024 * 1024;
81
106
  // Flush a part when a check shows at least this much compressed payload.
82
107
  export const PART_READY_COMPRESSED_BYTES = 12 * 1024 * 1024;
83
108
  // Absolute uncompressed ceiling for one part batch — bounds resident
84
- // memory even for highly-compressible content.
85
- export const PART_BATCH_UNCOMPRESSED_MAX = 48 * 1024 * 1024;
109
+ // memory even for highly-compressible content (incompressible batches
110
+ // never hit the compressed-size thresholds, so this is what actually
111
+ // stops accumulation; the payload build doubles it transiently).
112
+ export const PART_BATCH_UNCOMPRESSED_MAX = 24 * 1024 * 1024;
86
113
  // Change/trace parts flush by PLAIN size so their __agent__/changes.json
87
114
  // / trace.json entries (single file entries whose content grows with the
88
115
  // batch) stay small and parts upload promptly.
@@ -105,13 +132,13 @@ type SnapshotType = "start" | "change" | "trace" | "end";
105
132
 
106
133
  export type CollectorFile = {
107
134
  path: string;
108
- /** utf8 text (default) or base64 when encoding is "base64". Absent for
109
- * symlink records, which carry `target` instead. */
110
- content?: string;
111
- /** Absent = utf8 text. "base64" for binary content. */
112
- encoding?: "base64";
113
- /** Symlink record: the link target as stored on disk. */
114
- target?: string;
135
+ /** utf8 text by default. For binaries: base64. For symlinks: the link
136
+ * target verbatim (the backend requires text content on every file
137
+ * entry, so symlink records are text entries with a marker). */
138
+ content: string;
139
+ /** Absent = utf8 text. "base64" for binary content, "symlink" for
140
+ * symlink records (content = link target). */
141
+ encoding?: "base64" | "symlink";
115
142
  /** POSIX mode bits (octal string, e.g. "755"). Absent for symlinks. */
116
143
  mode?: string;
117
144
  };
@@ -127,8 +154,7 @@ export type ChangeJournalEntry = {
127
154
  path: string;
128
155
  status: "present" | "deleted";
129
156
  content?: string;
130
- encoding?: "base64";
131
- target?: string;
157
+ encoding?: "base64" | "symlink";
132
158
  mode?: string;
133
159
  /** Present when the record carries one slice of a bigger file. */
134
160
  slice?: { index: number; total: number; bytes: number; sha256: string };
@@ -175,8 +201,28 @@ type SessionState = {
175
201
  * fabricate deletions inside the workspace. */
176
202
  knownPaths: Set<string>;
177
203
  trace: TraceEvent[];
178
- /** Ordered change records — the replayable change log. */
179
- changeJournal: ChangeJournalEntry[];
204
+ /**
205
+ * The change journal lives ON DISK (ndjson sidecar, one JSON record
206
+ * per line) — never as an in-memory array (issue #6: the array held
207
+ * every changed file's content string and grew to multiple GB).
208
+ * Captures APPEND to the file; change uploads stream the unacked
209
+ * region and compact the acked prefix away. Appends during an upload
210
+ * simply land past the read window (bounded by `journalUploadEof`) —
211
+ * no gating, no memory growth.
212
+ */
213
+ journalPath: string | null;
214
+ /** File offset of the first not-yet-uploaded record. */
215
+ journalAckOffset: number;
216
+ /** Bytes appended so far (soft disk cap accounting). */
217
+ journalAppendedBytes: number;
218
+ /** Serializes journal appends. */
219
+ journalAppendTail: Promise<void>;
220
+ /** Set while a change upload reads the journal. */
221
+ journalUploadEof: number | null;
222
+ /** True once the soft disk cap stopped captures (warned once). */
223
+ journalCaptureStopped: boolean;
224
+ /** Last journal signature per path (consecutive-duplicate suppression). */
225
+ journalLastByPath: Map<string, string>;
180
226
  changeCaptureTail: Promise<void>;
181
227
  ready: Promise<void>;
182
228
  tail: Promise<void>;
@@ -393,7 +439,7 @@ export function selectTraceEvents(
393
439
  * removed on purpose: `git ls-files` drops anything .gitignored (build
394
440
  * outputs, node_modules), which broke the whole-project guarantee.
395
441
  */
396
- async function listWorkspaceFiles(root: string): Promise<string[]> {
442
+ export async function listWorkspaceFiles(root: string): Promise<string[]> {
397
443
  return walkWorkspace(root);
398
444
  }
399
445
 
@@ -527,19 +573,23 @@ function sliceManifest(
527
573
  type ContentChunk = { content: string; encoding?: "base64" };
528
574
 
529
575
  /**
530
- * Read a regular file into <= ~8 MiB content chunks: small files in one
531
- * whole read (fully redacted), big files streamed in fixed raw chunks —
532
- * strings stay far below the runtime string limit regardless of file
533
- * size. Binary detection happens on the first chunk and applies to the
534
- * whole file.
576
+ * Stream a regular file as <= ~8 MiB content chunks, yielding each chunk
577
+ * the moment it is produced — a file's content is NEVER materialized as
578
+ * one string regardless of size (issue #6: the old array version held
579
+ * every chunk at once). Small files are a single whole read; big files
580
+ * stream in fixed raw chunks. Binary detection happens on the first
581
+ * chunk and applies to the whole file; text is redacted per
582
+ * line-aligned segment.
535
583
  */
536
- async function readContentChunks(absolute: string, size: number): Promise<ContentChunk[]> {
584
+ async function* iterateContentChunks(absolute: string, size: number): AsyncGenerator<ContentChunk> {
537
585
  if (size <= WHOLE_READ_BYTES) {
538
586
  const buffer = await readFile(absolute);
539
587
  if (isBinary(buffer)) {
540
- return [{ content: buffer.toString("base64"), encoding: "base64" }];
588
+ yield { content: buffer.toString("base64"), encoding: "base64" };
589
+ return;
541
590
  }
542
- return [{ content: redactCollectorText(buffer.toString("utf8")).text }];
591
+ yield { content: redactCollectorText(buffer.toString("utf8")).text };
592
+ return;
543
593
  }
544
594
  const { open } = await import("node:fs/promises");
545
595
  const handle = await open(absolute, "r");
@@ -548,112 +598,139 @@ async function readContentChunks(absolute: string, size: number): Promise<Conten
548
598
  const buffer = Buffer.alloc(chunkBytes);
549
599
  const first = await handle.read(buffer, 0, chunkBytes, null);
550
600
  const binary = isBinary(buffer.subarray(0, first.bytesRead));
551
- const chunks: ContentChunk[] = [];
552
601
  if (binary) {
553
602
  // Binaries carry no redactable text: base64 each raw chunk.
554
- chunks.push({ content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" });
603
+ yield { content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" };
555
604
  let read = 0;
556
605
  while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
557
- chunks.push({ content: buffer.subarray(0, read).toString("base64"), encoding: "base64" });
606
+ yield { content: buffer.subarray(0, read).toString("base64"), encoding: "base64" };
558
607
  }
559
- } else {
560
- // Text: decode incrementally (StringDecoder absorbs multibyte
561
- // sequences split across chunk boundaries), emit only up to the
562
- // last complete line, carry the remainder, and redact each
563
- // line-aligned segment.
564
- const { StringDecoder } = await import("node:string_decoder");
565
- const decoder = new StringDecoder("utf8");
566
- let pending = decoder.write(buffer.subarray(0, first.bytesRead));
567
- let read = 0;
568
- while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
569
- pending += decoder.write(buffer.subarray(0, read));
608
+ return;
609
+ }
610
+ // Text: decode incrementally (StringDecoder absorbs multibyte
611
+ // sequences split across chunk boundaries), emit only up to the
612
+ // last complete line, carry the remainder, and redact each
613
+ // line-aligned segment. A pathological newline-less file must not
614
+ // grow `pending` forever: at PENDING_TEXT_CAP_BYTES the segment is
615
+ // flushed mid-line (line-anchored redaction weakens at that rare
616
+ // cut; unbounded memory is worse).
617
+ const { StringDecoder } = await import("node:string_decoder");
618
+ const decoder = new StringDecoder("utf8");
619
+ let pending = decoder.write(buffer.subarray(0, first.bytesRead));
620
+ let read = 0;
621
+ while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
622
+ pending += decoder.write(buffer.subarray(0, read));
623
+ for (;;) {
570
624
  const lastNl = pending.lastIndexOf("\n");
571
625
  if (lastNl >= 0) {
572
- chunks.push({ content: redactCollectorText(pending.slice(0, lastNl + 1)).text });
626
+ yield { content: redactCollectorText(pending.slice(0, lastNl + 1)).text };
573
627
  pending = pending.slice(lastNl + 1);
628
+ continue;
574
629
  }
630
+ if (pending.length >= PENDING_TEXT_CAP_BYTES) {
631
+ yield { content: redactCollectorText(pending).text };
632
+ pending = "";
633
+ }
634
+ break;
575
635
  }
576
- pending += decoder.end();
577
- if (pending) chunks.push({ content: redactCollectorText(pending).text });
578
636
  }
579
- return chunks;
637
+ pending += decoder.end();
638
+ if (pending) yield { content: redactCollectorText(pending).text };
580
639
  } finally {
581
640
  await handle.close();
582
641
  }
583
642
  }
584
643
 
585
644
  /**
586
- * Read one workspace entry into CollectorFile entries. Regular files:
645
+ * Stream one workspace entry as CollectorFile entries. Regular files:
587
646
  * text is redacted, binaries base64-encoded ("encoding": "base64"), and
588
647
  * the POSIX mode rides along (octal string) so replay can restore the
589
648
  * executable bit (stored-and-ignored on Windows). Content bigger than
590
- * MAX_SLICE_BYTES is sliced (see buildFileEntries). Symlinks become
591
- * records { path, target } (target verbatim; replay recreates the link).
592
- * Returns [] when the entry vanished or is a special file.
649
+ * MAX_SLICE_BYTES is sliced on the fly — slices are yielded the moment
650
+ * they exist while only the tiny sha256 descriptors accumulate, then
651
+ * the manifest is yielded LAST (replay unions manifests by index, so
652
+ * position never matters). Symlinks become text entries
653
+ * { path, content: <target>, encoding: "symlink" } — the backend's
654
+ * per-entry validation requires text content, and replay recreates the
655
+ * actual link from the target. Yields nothing when the entry vanished
656
+ * or is a special file.
593
657
  */
594
- async function readWorkspaceEntries(root: string, path: string): Promise<CollectorFile[]> {
658
+ async function* iterateWorkspaceEntries(root: string, path: string): AsyncGenerator<CollectorFile> {
659
+ let absolute = "";
595
660
  try {
596
- const absolute = resolve(root, path);
597
- if (portablePath(root, absolute).startsWith("../")) return [];
661
+ absolute = resolve(root, path);
662
+ if (portablePath(root, absolute).startsWith("../")) return;
598
663
  const file = await lstat(absolute);
599
664
  if (file.isSymbolicLink()) {
600
665
  const target = await readlink(absolute);
601
- return [{ path, target }];
666
+ yield { path, content: target, encoding: "symlink" };
667
+ return;
602
668
  }
603
- if (!file.isFile()) return [];
669
+ if (!file.isFile()) return;
604
670
  const mode = file.mode & 0o777;
605
- const chunks: ContentChunk[] = await readContentChunks(absolute, file.size);
606
- if (chunks.length === 1) {
607
- const chunk = chunks[0];
608
- return buildFileEntries({
609
- path,
610
- content: chunk.content,
611
- ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
612
- ...(mode ? { mode: mode.toString(8) } : {}),
613
- });
671
+ const modeField = mode ? { mode: mode.toString(8) } : {};
672
+ if (file.size <= WHOLE_READ_BYTES) {
673
+ // Small file: whole read, real path preserved; buildFileEntries
674
+ // slices (manifest + parts) when the encoded content exceeds
675
+ // MAX_SLICE_BYTES — same shape as always.
676
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
677
+ for (const entry of buildFileEntries({
678
+ path,
679
+ content: chunk.content,
680
+ ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
681
+ ...modeField,
682
+ })) {
683
+ yield entry;
684
+ }
685
+ }
686
+ return;
614
687
  }
615
- // Big file: streamed chunks -> one logical entry per chunk (already
616
- // <= ~8 MiB for raw chunks; base64 inflates 4/3, so sub-slice those),
617
- // plus the manifest for deterministic reassembly.
618
- const encoding = chunks[0].encoding;
619
- const entries: CollectorFile[] = [];
620
- const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
621
- entries.push({
688
+ // Big file (streamed): slices are yielded the moment they exist;
689
+ // only the tiny sha256 descriptors accumulate. The reassembly
690
+ // manifest is yielded LAST (replay unions manifests by index, so
691
+ // position never matters).
692
+ let index = 0;
693
+ const descriptors: Array<{ index: number; bytes: number; sha256: string }> = [];
694
+ let sawEncoding: "base64" | undefined;
695
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
696
+ sawEncoding ??= chunk.encoding;
697
+ for (const slice of sliceString(chunk.content)) {
698
+ descriptors.push({ index, bytes: Buffer.byteLength(slice), sha256: sha256Hex(slice) });
699
+ yield {
700
+ path: slicePartPath(path, index),
701
+ content: slice,
702
+ ...(sawEncoding ? { encoding: sawEncoding } : {}),
703
+ };
704
+ index += 1;
705
+ }
706
+ }
707
+ if (index === 0) return;
708
+ yield {
622
709
  path: sliceManifestPath(path),
623
710
  content: JSON.stringify(sliceManifest(
624
- {
625
- path,
626
- ...(encoding ? { encoding } : {}),
627
- ...(mode ? { mode: mode.toString(8) } : {}),
628
- },
629
- sliceContents.map((slice, index) => ({
630
- index,
631
- bytes: Buffer.byteLength(slice),
632
- sha256: sha256Hex(slice),
633
- })),
711
+ { path, ...(sawEncoding ? { encoding: sawEncoding } : {}), ...modeField },
712
+ descriptors,
634
713
  )),
635
- });
636
- sliceContents.forEach((slice, index) => {
637
- entries.push({
638
- path: slicePartPath(path, index),
639
- content: slice,
640
- ...(encoding ? { encoding } : {}),
641
- });
642
- });
643
- return entries;
714
+ };
715
+ return;
644
716
  } catch {
645
717
  // Workspaces are live; races are expected and retried by the next snapshot.
646
- return [];
718
+ return;
647
719
  }
648
720
  }
649
721
 
650
- export async function collectFiles(
722
+ /**
723
+ * Streaming file collection: yields workspace files (metadata entry
724
+ * first) one at a time so the one-shot path never holds the whole tree's
725
+ * content in memory (issue #6). Same order, denylist, and byte budget
726
+ * as the compat collectFiles drain below.
727
+ */
728
+ export async function* iterateCollectFiles(
651
729
  root: string,
652
730
  byteLimit: number,
653
731
  workspaceId: string,
654
732
  session?: Pick<SessionState, "id" | "segment" | "resumed">,
655
- ): Promise<CollectorFile[]> {
656
- const files: CollectorFile[] = [];
733
+ ): AsyncGenerator<CollectorFile> {
657
734
  let used = 0;
658
735
  const metadata = JSON.stringify({
659
736
  workspace_id: workspaceId,
@@ -665,19 +742,32 @@ export async function collectFiles(
665
742
  root_name: root.split(sep).filter(Boolean).at(-1) ?? "workspace",
666
743
  git: await gitMetadata(root),
667
744
  });
668
- files.push({ path: `${AGENT_DIR}/workspace.json`, content: metadata });
745
+ yield { path: `${AGENT_DIR}/workspace.json`, content: metadata };
669
746
  used += Buffer.byteLength(metadata);
670
747
 
671
748
  const paths = (await listWorkspaceFiles(root)).sort(comparePaths);
672
749
  for (const path of paths) {
673
750
  if (used >= byteLimit) break;
674
- for (const file of await readWorkspaceEntries(root, path)) {
675
- const size = Buffer.byteLength(file.content ?? file.target ?? "");
751
+ for await (const file of iterateWorkspaceEntries(root, path)) {
752
+ const size = Buffer.byteLength(file.content ?? "");
676
753
  if (used + size > byteLimit) continue;
677
- files.push(file);
754
+ yield file;
678
755
  used += size;
679
756
  }
680
757
  }
758
+ }
759
+
760
+ /** Compat drain of iterateCollectFiles (tests + tooling). */
761
+ export async function collectFiles(
762
+ root: string,
763
+ byteLimit: number,
764
+ workspaceId: string,
765
+ session?: Pick<SessionState, "id" | "segment" | "resumed">,
766
+ ): Promise<CollectorFile[]> {
767
+ const files: CollectorFile[] = [];
768
+ for await (const file of iterateCollectFiles(root, byteLimit, workspaceId, session)) {
769
+ files.push(file);
770
+ }
681
771
  return files;
682
772
  }
683
773
 
@@ -721,14 +811,11 @@ export type EnvelopePart = {
721
811
  };
722
812
 
723
813
  /**
724
- * Deterministic part split for a full files array (used by `omnirush
725
- * collect` and the tests): sort files by path, accumulate in order, keep
726
- * each part <= maxCompressed compressed. A single file bigger than one
727
- * part gets its own part; file content is never split across parts. A
728
- * lone file that cannot fit even the hard rail is skipped with a warning.
729
- * Envelope sequence numbers are state.sequence + 1 + partIndex.
814
+ * Deterministic part split for a full files array (compat drain of
815
+ * iterateEnvelopeParts, used by the tests): sort files by path,
816
+ * accumulate in order, keep each part <= maxCompressed compressed.
730
817
  */
731
- export function buildEnvelopeParts(
818
+ export async function buildEnvelopeParts(
732
819
  state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
733
820
  snapshotType: SnapshotType,
734
821
  files: CollectorFile[],
@@ -741,28 +828,34 @@ export function buildEnvelopeParts(
741
828
  const sorted = files
742
829
  .flatMap((file) => buildFileEntries(file))
743
830
  .sort((left, right) => comparePaths(left.path, right.path));
744
- return packEnvelopeParts(state, snapshotType, sorted, maxCompressed, log);
831
+ const parts: EnvelopePart[] = [];
832
+ for await (const part of iterateEnvelopeParts(state, snapshotType, sorted, maxCompressed, log)) {
833
+ parts.push(part);
834
+ }
835
+ return parts;
745
836
  }
746
837
 
747
838
  /**
748
- * Shared greedy packer: split an ordered file list into compressed parts,
749
- * preserving path order across parts. Used by buildEnvelopeParts (pure,
750
- * whole list in memory) and mirrored by the collector's streaming packer
751
- * (packAndUpload, which never holds more than one part in memory).
839
+ * Streaming packer for the one-shot path: yields compressed parts in
840
+ * order so callers can upload-and-discard instead of holding the whole
841
+ * workspace's parts (payload + compressed buffers) in memory (issue #6).
842
+ * Same greedy algorithm as before: split at the compressed target, a
843
+ * lone file may use the hard rail, lone-over-rail files are skipped
844
+ * loudly. File content is never split across parts.
752
845
  */
753
- export async function packEnvelopeParts(
846
+ export async function* iterateEnvelopeParts(
754
847
  state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
755
848
  snapshotType: SnapshotType,
756
- sortedFiles: CollectorFile[],
849
+ sortedFiles: Iterable<CollectorFile> | AsyncIterable<CollectorFile>,
757
850
  maxCompressed = MAX_PART_COMPRESSED_BYTES,
758
851
  log?: (message: string, attributes?: Record<string, unknown>) => void,
759
- ): Promise<EnvelopePart[]> {
852
+ ): AsyncGenerator<EnvelopePart> {
760
853
  const hard = Math.max(maxCompressed, MAX_PART_COMPRESSED_HARD);
761
854
  const measure = async (batch: CollectorFile[], partIndex: number): Promise<EnvelopePart> => {
762
855
  const payload = buildEnvelopePayload({ ...state, sequence: (state.sequence ?? 0) + partIndex }, snapshotType, batch);
763
856
  return { files: batch, payload, compressed: await compressZstd(payload) };
764
857
  };
765
- const parts: EnvelopePart[] = [];
858
+ let partsYielded = 0;
766
859
  let batch: CollectorFile[] = [];
767
860
  const fit = async (batchToFit: CollectorFile[], partIndex: number) => {
768
861
  let measured = await measure(batchToFit, partIndex);
@@ -778,21 +871,22 @@ export async function packEnvelopeParts(
778
871
  const lone = batchToFit.pop()!;
779
872
  log?.("OmniRush file exceeds the upload limits and cannot be shipped as a single part; skipped", {
780
873
  path: lone.path,
781
- bytes: Buffer.byteLength(lone.content ?? lone.target ?? ""),
874
+ bytes: Buffer.byteLength(lone.content ?? ""),
782
875
  });
783
876
  return { measured: null, carry };
784
877
  }
785
878
  return { measured, carry };
786
879
  };
787
- for (const file of sortedFiles) {
880
+ for await (const file of sortedFiles) {
788
881
  batch.push(file);
789
- const partIndex = parts.length;
882
+ const partIndex = partsYielded;
790
883
  const estimate = batch.reduce((total, f) => total + Buffer.byteLength(f.content), 0);
791
884
  if (estimate < PART_UNCOMPRESSED_STEP) continue;
792
885
  const { measured, carry } = await fit(batch, partIndex);
793
886
  if (measured
794
887
  && (measured.compressed.length >= PART_READY_COMPRESSED_BYTES || estimate >= PART_BATCH_UNCOMPRESSED_MAX)) {
795
- parts.push(measured);
888
+ partsYielded += 1;
889
+ yield measured;
796
890
  batch = carry;
797
891
  } else if (measured) {
798
892
  batch = [...batch, ...carry];
@@ -801,15 +895,37 @@ export async function packEnvelopeParts(
801
895
  }
802
896
  }
803
897
  while (batch.length > 0) {
804
- const { measured, carry } = await fit(batch, parts.length);
805
- if (measured) parts.push(measured);
898
+ const { measured, carry } = await fit(batch, partsYielded);
899
+ if (measured) {
900
+ partsYielded += 1;
901
+ yield measured;
902
+ }
806
903
  if (carry.length === 0) break;
807
904
  batch = carry;
808
905
  }
809
- return parts;
810
906
  }
811
907
 
812
908
  export class WorkspaceCollector {
909
+ /**
910
+ * Bounded first-chunk read of an error response body for diagnostics:
911
+ * at most 512 bytes, decoded and whitespace-collapsed. Never throws.
912
+ * Error bodies are small JSON from the manager; the bound exists so a
913
+ * pathological response cannot balloon a log line (or memory).
914
+ */
915
+ private static async errorBodySnippet(response: any): Promise<string> {
916
+ try {
917
+ const reader = response?.body?.getReader?.();
918
+ if (!reader) return "";
919
+ const { value } = await reader.read();
920
+ reader.cancel().catch(() => undefined);
921
+ if (!value) return "";
922
+ const text = Buffer.from(value).toString("utf8").replace(/\s+/g, " ").trim();
923
+ return text.length > 200 ? `${text.slice(0, 200)}…` : text;
924
+ } catch {
925
+ return "";
926
+ }
927
+ }
928
+
813
929
  private readonly collectUrl: string | null;
814
930
  private token: string;
815
931
  private readonly fetcher: typeof fetch;
@@ -817,6 +933,7 @@ export class WorkspaceCollector {
817
933
  private readonly uploader?: CollectorOptions["upload"];
818
934
  private readonly log: NonNullable<CollectorOptions["log"]>;
819
935
  private readonly ledgerPath: string | null;
936
+ private readonly stateDir: string | null;
820
937
  private readonly ledgerReady: Promise<void>;
821
938
  private ledger: SessionLedger = { version: 1, sessions: {} };
822
939
  private ledgerWriteTail: Promise<void> = Promise.resolve();
@@ -834,6 +951,7 @@ export class WorkspaceCollector {
834
951
  this.refresh = options.refresh;
835
952
  this.uploader = options.upload;
836
953
  this.log = options.log ?? (() => undefined);
954
+ this.stateDir = options.stateDir ? resolve(options.stateDir) : null;
837
955
  this.ledgerPath = options.stateDir ? join(resolve(options.stateDir), SESSION_LEDGER_FILE) : null;
838
956
  this.ledgerReady = this.loadLedger();
839
957
  this.changeDebounceMs = options.changeDebounceMs ?? CHANGE_DEBOUNCE_MS;
@@ -943,7 +1061,13 @@ export class WorkspaceCollector {
943
1061
  watcherStartedAtMs: 0,
944
1062
  knownPaths: new Set(),
945
1063
  trace: [],
946
- changeJournal: [],
1064
+ journalPath: this.stateDir ? join(resolve(this.stateDir), `journal-${sessionId}.ndjson`) : null,
1065
+ journalAckOffset: 0,
1066
+ journalAppendedBytes: 0,
1067
+ journalAppendTail: Promise.resolve(),
1068
+ journalUploadEof: null,
1069
+ journalCaptureStopped: false,
1070
+ journalLastByPath: new Map(),
947
1071
  changeCaptureTail: Promise.resolve(),
948
1072
  ready: Promise.resolve(),
949
1073
  tail: Promise.resolve(),
@@ -1008,6 +1132,7 @@ export class WorkspaceCollector {
1008
1132
  if (state.trace.length > 0) pendingTrace.push(...state.trace.splice(0));
1009
1133
  await state.changeCaptureTail;
1010
1134
  if (finalTrace !== undefined) pendingTrace.push({ at: new Date().toISOString(), type: "session.completed", data: finalTrace });
1135
+ await this.teardownJournal(state); // ship unacked change records, then unlink
1011
1136
  await this.uploadTrace(state, pendingTrace);
1012
1137
  await this.uploadWorkspace(state, "end");
1013
1138
  this.sessions.delete(sessionId);
@@ -1056,18 +1181,28 @@ export class WorkspaceCollector {
1056
1181
  });
1057
1182
  }
1058
1183
 
1184
+ /** Run the debounced change-snapshot upload immediately. */
1185
+ private flushChangeSnapshot(state: SessionState): void {
1186
+ if (state.finished || state.budgetExhausted) return;
1187
+ if (state.changeTimer) {
1188
+ clearTimeout(state.changeTimer);
1189
+ state.changeTimer = null;
1190
+ }
1191
+ this.enqueue(state, async () => {
1192
+ if (state.budgetExhausted || state.finished) return;
1193
+ const signature = await workspaceSignature(state.root);
1194
+ if (!signature || signature === state.lastSignature) return;
1195
+ state.lastSignature = signature;
1196
+ await this.uploadWorkspace(state, "change");
1197
+ });
1198
+ }
1199
+
1059
1200
  private scheduleChange(state: SessionState): void {
1060
1201
  if (state.finished || state.budgetExhausted) return;
1061
1202
  if (state.changeTimer) clearTimeout(state.changeTimer);
1062
1203
  state.changeTimer = setTimeout(() => {
1063
1204
  state.changeTimer = null;
1064
- this.enqueue(state, async () => {
1065
- if (state.budgetExhausted || state.finished) return;
1066
- const signature = await workspaceSignature(state.root);
1067
- if (!signature || signature === state.lastSignature) return;
1068
- state.lastSignature = signature;
1069
- await this.uploadWorkspace(state, "change");
1070
- });
1205
+ this.flushChangeSnapshot(state);
1071
1206
  }, this.changeDebounceMs);
1072
1207
  state.changeTimer.unref?.();
1073
1208
  }
@@ -1110,47 +1245,54 @@ export class WorkspaceCollector {
1110
1245
  if (!options?.force && Math.floor(file.mtimeMs) <= state.watcherStartedAtMs) return;
1111
1246
  state.knownPaths.add(path);
1112
1247
  if (file.isSymbolicLink()) {
1113
- // Symlink record: path + target so replay recreates the link.
1248
+ // Symlink record: the target as text content with the symlink
1249
+ // encoding — replay recreates the actual link from it.
1114
1250
  const target = await readlink(absolute);
1115
- entry = { path, at: new Date().toISOString(), status: "present", target };
1251
+ entry = {
1252
+ path,
1253
+ at: new Date().toISOString(),
1254
+ status: "present",
1255
+ content: target,
1256
+ encoding: "symlink",
1257
+ };
1116
1258
  } else if (!file.isFile()) {
1117
1259
  return;
1118
1260
  } else {
1119
1261
  const at = new Date().toISOString();
1120
1262
  const mode = file.mode & 0o777;
1121
1263
  const modeField = mode ? { mode: mode.toString(8) } : {};
1122
- const chunks = await readContentChunks(absolute, file.size);
1123
- const encoding = chunks[0].encoding;
1124
- const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
1125
- if (sliceContents.length === 1) {
1126
- entry = {
1127
- path,
1128
- at,
1129
- status: "present",
1130
- content: sliceContents[0],
1131
- ...(encoding ? { encoding } : {}),
1132
- ...modeField,
1133
- };
1134
- } else {
1135
- // Big file: one ordered record per slice so the journal stays
1136
- // replayable; reassembly rides the slice metadata.
1137
- sliceContents.forEach((content, index) => {
1264
+ // STREAMED capture: each chunk becomes a journal record the
1265
+ // moment it exists — a big file is never held in memory whole
1266
+ // (issue #6). One record per <= MAX_SLICE_BYTES slice; slice
1267
+ // metadata rides the records only when reassembly is needed.
1268
+ let sliceCount = 0;
1269
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
1270
+ for (const content of sliceString(chunk.content)) {
1271
+ sliceCount += 1;
1138
1272
  records.push({
1139
1273
  path,
1140
1274
  at,
1141
1275
  status: "present",
1142
1276
  content,
1143
- ...(encoding ? { encoding } : {}),
1277
+ ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
1144
1278
  ...modeField,
1145
- slice: {
1146
- index,
1147
- total: sliceContents.length,
1148
- bytes: Buffer.byteLength(content),
1149
- sha256: sha256Hex(content),
1150
- },
1151
1279
  });
1280
+ }
1281
+ }
1282
+ if (sliceCount > 1) {
1283
+ records.forEach((record, index) => {
1284
+ record.slice = {
1285
+ index,
1286
+ total: sliceCount,
1287
+ bytes: Buffer.byteLength(record.content ?? ""),
1288
+ sha256: sha256Hex(record.content ?? ""),
1289
+ };
1152
1290
  });
1153
- state.changeJournal.push(...records);
1291
+ }
1292
+ if (sliceCount === 1) {
1293
+ entry = records.pop()!;
1294
+ } else {
1295
+ await this.appendJournalRecords(state, records);
1154
1296
  return;
1155
1297
  }
1156
1298
  }
@@ -1169,27 +1311,156 @@ export class WorkspaceCollector {
1169
1311
  // Consecutive-duplicate suppression: macOS FSEvents can deliver a
1170
1312
  // late event for a path that was also captured directly, and an
1171
1313
  // identical re-capture adds upload fat without a state change. The
1172
- // replay result is identical either way.
1314
+ // replay result is identical either way. The last signature per
1315
+ // path lives in a tiny in-memory map (path strings only — content
1316
+ // is already on disk in the journal).
1173
1317
  const journalSignature = (record: ChangeJournalEntry) =>
1174
- JSON.stringify([record.status, record.content, record.encoding, record.target, record.mode]);
1318
+ JSON.stringify([record.status, record.content, record.encoding, record.mode]);
1175
1319
  const pendingRecords = [entry, ...records];
1320
+ const accepted: ChangeJournalEntry[] = [];
1176
1321
  for (const record of pendingRecords) {
1177
- let duplicate = false;
1178
- for (let i = state.changeJournal.length - 1, scanned = 0; i >= 0 && scanned < 100; i--, scanned++) {
1179
- const prior = state.changeJournal[i];
1180
- if (prior.path !== record.path) continue;
1181
- duplicate = journalSignature(prior) === journalSignature(record);
1182
- break;
1322
+ const signature = journalSignature(record);
1323
+ if (state.journalLastByPath.get(record.path) === signature) continue;
1324
+ state.journalLastByPath.set(record.path, signature);
1325
+ accepted.push(record);
1326
+ }
1327
+ await this.appendJournalRecords(state, accepted);
1328
+ }
1329
+
1330
+ /**
1331
+ * APPEND records to the on-disk journal (ndjson — one JSON record per
1332
+ * line, 0600). Returns immediately when the capture hit the soft disk
1333
+ * cap (a backend down for a very long time must stop CAPTURING loudly
1334
+ * rather than fill the disk or RAM — the end snapshot still ships the
1335
+ * final tree). Serialized through the per-session append chain.
1336
+ */
1337
+ private appendJournalRecords(state: SessionState, records: ChangeJournalEntry[]): Promise<void> {
1338
+ if (!state.journalPath || records.length === 0) return Promise.resolve();
1339
+ if (state.journalCaptureStopped) return Promise.resolve();
1340
+ const unacked = state.journalAppendedBytes - state.journalAckOffset;
1341
+ const capMb = Number(process.env.OMNIRUSH_JOURNAL_DISK_CAP_MB);
1342
+ const capBytes = Number.isFinite(capMb) && capMb > 0 ? capMb * 1024 * 1024 : JOURNAL_DISK_CAP_BYTES;
1343
+ if (unacked > capBytes) {
1344
+ state.journalCaptureStopped = true;
1345
+ this.log("warn", "OmniRush change journal hit the disk cap — change capture stopped for this session", {
1346
+ sessionId: state.id,
1347
+ capBytes: capBytes,
1348
+ unackedBytes: unacked,
1349
+ });
1350
+ state.trace.push({
1351
+ at: new Date().toISOString(),
1352
+ type: "journal.capture_stopped",
1353
+ data: { cap_bytes: capBytes, unacked_bytes: unacked },
1354
+ });
1355
+ return Promise.resolve();
1356
+ }
1357
+ const lines = records.map((record) => JSON.stringify(record) + "\n").join("");
1358
+ const bytes = Buffer.byteLength(lines);
1359
+ state.journalAppendTail = state.journalAppendTail
1360
+ .catch(() => undefined)
1361
+ .then(async () => {
1362
+ const { appendFile, mkdir } = await import("node:fs/promises");
1363
+ await mkdir(dirname(state.journalPath!), { recursive: true, mode: 0o700 });
1364
+ await appendFile(state.journalPath!, lines, { encoding: "utf8", mode: 0o600 });
1365
+ state.journalAppendedBytes += bytes;
1366
+ });
1367
+ return state.journalAppendTail;
1368
+ }
1369
+
1370
+ /**
1371
+ * Stream the journal's unacked records (from journalAckOffset to the
1372
+ * upload-start EOF) as parsed entries, each tagged with its raw line
1373
+ * bytes via a NON-enumerable property so the changes.json sidecar
1374
+ * serialization never sees it. Records appended while the upload runs
1375
+ * land beyond the read window and simply ship with the next snapshot.
1376
+ */
1377
+ private async *readJournalRecords(state: SessionState): AsyncGenerator<ChangeJournalEntry> {
1378
+ if (!state.journalPath || state.journalAckOffset >= state.journalUploadEof) return;
1379
+ const { createReadStream } = await import("node:fs");
1380
+ const readline = await import("node:readline");
1381
+ const stream = createReadStream(state.journalPath, {
1382
+ start: state.journalAckOffset,
1383
+ end: state.journalUploadEof! - 1,
1384
+ encoding: "utf8",
1385
+ });
1386
+ const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
1387
+ for await (const line of rl) {
1388
+ if (!line.trim()) continue;
1389
+ try {
1390
+ const record = JSON.parse(line) as ChangeJournalEntry;
1391
+ Object.defineProperty(record, RAW_BYTES_FIELD, {
1392
+ value: Buffer.byteLength(line) + 1,
1393
+ enumerable: false,
1394
+ configurable: true,
1395
+ });
1396
+ yield record;
1397
+ } catch {
1398
+ this.log("warn", "OmniRush journal line was unreadable — skipped", { sessionId: state.id });
1183
1399
  }
1184
- if (!duplicate) state.changeJournal.push(record);
1185
1400
  }
1186
1401
  }
1187
1402
 
1188
- /** Remove exactly the uploaded entries from the ordered journal. */
1189
- private acknowledgeJournal(state: SessionState, uploaded: ChangeJournalEntry[]): void {
1190
- if (uploaded.length === 0) return;
1191
- const sent = uploaded.length > 32 ? new Set(uploaded) : null;
1192
- state.changeJournal = state.changeJournal.filter((entry) => sent ? !sent.has(entry) : !uploaded.includes(entry));
1403
+ /**
1404
+ * Upload the journal's unacked records as change snapshots, then
1405
+ * compact: everything acked is removed from the file; records appended
1406
+ * during the upload (past journalUploadEof) survive untouched.
1407
+ */
1408
+ private async uploadChangeJournal(state: SessionState): Promise<void> {
1409
+ if (!state.journalPath) return;
1410
+ let fileSize = 0;
1411
+ try {
1412
+ fileSize = (await lstat(state.journalPath)).size;
1413
+ } catch {
1414
+ return; // no journal yet — nothing captured
1415
+ }
1416
+ if (fileSize <= state.journalAckOffset) return;
1417
+ const ackStart = state.journalAckOffset;
1418
+ state.journalUploadEof = fileSize;
1419
+ try {
1420
+ await this.uploadChangeParts(state, this.readJournalRecords(state), {
1421
+ onAck: (batch) => {
1422
+ for (const record of batch) {
1423
+ const raw = (record as Record<symbol, unknown>)[RAW_BYTES_FIELD];
1424
+ if (typeof raw === "number") state.journalAckOffset += raw;
1425
+ }
1426
+ },
1427
+ });
1428
+ } finally {
1429
+ state.journalUploadEof = null;
1430
+ }
1431
+ if (state.journalAckOffset <= ackStart) return; // nothing acked — keep everything
1432
+ await this.compactJournal(state, state.journalAckOffset);
1433
+ state.journalAckOffset = 0;
1434
+ }
1435
+
1436
+ /** Rewrite the journal without its first `dropBytes` bytes. */
1437
+ private async compactJournal(state: SessionState, dropBytes: number): Promise<void> {
1438
+ if (!state.journalPath) return;
1439
+ const { createReadStream, createWriteStream } = await import("node:fs");
1440
+ const { rename } = await import("node:fs/promises");
1441
+ const tmpPath = `${state.journalPath}.compact`;
1442
+ await new Promise<void>((resolvePromise, rejectPromise) => {
1443
+ const input = createReadStream(state.journalPath!, { start: dropBytes, encoding: "utf8" });
1444
+ const output = createWriteStream(tmpPath, { encoding: "utf8", mode: 0o600 });
1445
+ input.on("error", rejectPromise);
1446
+ output.on("error", rejectPromise);
1447
+ output.on("finish", () => resolvePromise());
1448
+ input.pipe(output);
1449
+ });
1450
+ await rename(tmpPath, state.journalPath);
1451
+ }
1452
+
1453
+ /** Upload unacked journal records (best effort) and remove the file. */
1454
+ private async teardownJournal(state: SessionState): Promise<void> {
1455
+ try {
1456
+ await this.uploadChangeJournal(state);
1457
+ } catch {
1458
+ /* best effort — the end snapshot still ships the final tree */
1459
+ }
1460
+ if (state.journalPath) {
1461
+ const { rm } = await import("node:fs/promises");
1462
+ await rm(state.journalPath, { force: true }).catch(() => undefined);
1463
+ }
1193
1464
  }
1194
1465
 
1195
1466
  private enqueue(state: SessionState, operation: () => Promise<void>): void {
@@ -1204,17 +1475,10 @@ export class WorkspaceCollector {
1204
1475
  private async uploadWorkspace(state: SessionState, type: "start" | "change" | "end"): Promise<void> {
1205
1476
  if (state.budgetExhausted) return;
1206
1477
  if (type === "change") {
1207
- // Change snapshots carry ONLY the touched files from the journal —
1208
- // never a full re-enumeration of the tree.
1209
- const snapshot = [...state.changeJournal];
1210
- if (snapshot.length === 0) {
1211
- state.lastSignature = await workspaceSignature(state.root);
1212
- return;
1213
- }
1214
- const uploaded = await this.uploadChangeParts(state, snapshot);
1215
- // Acknowledge only after every part of the snapshot succeeded; on a
1216
- // mid-budget exhaustion, acknowledge only what actually shipped.
1217
- this.acknowledgeJournal(state, state.budgetExhausted ? uploaded.flat() : snapshot);
1478
+ // Change snapshots carry ONLY the touched files — streamed from
1479
+ // the on-disk journal, acked per part, then compacted. Never a
1480
+ // full re-enumeration of the tree, never held in memory.
1481
+ await this.uploadChangeJournal(state);
1218
1482
  } else {
1219
1483
  await this.uploadTreeParts(state, type);
1220
1484
  }
@@ -1247,7 +1511,7 @@ export class WorkspaceCollector {
1247
1511
  return { files, payload, compressed: await compressZstd(payload) };
1248
1512
  };
1249
1513
  await this.packAndUpload(state, snapshotType, this.iterTreeFiles(state, paths), {
1250
- sizeOf: (file) => Buffer.byteLength(file.content ?? file.target ?? "") + file.path.length + 64,
1514
+ sizeOf: (file) => Buffer.byteLength(file.content ?? "") + file.path.length + 64,
1251
1515
  measure,
1252
1516
  upload: async (measured) => {
1253
1517
  const ok = await this.uploadEnvelope(state, snapshotType, measured.files, measured);
@@ -1258,12 +1522,12 @@ export class WorkspaceCollector {
1258
1522
  this.log("warn", "OmniRush file exceeds the upload rail and cannot be shipped as a single part; skipped", {
1259
1523
  sessionId: state.id,
1260
1524
  path: file.path,
1261
- bytes: Buffer.byteLength(file.content ?? file.target ?? ""),
1525
+ bytes: Buffer.byteLength(file.content ?? ""),
1262
1526
  });
1263
1527
  state.trace.push({
1264
1528
  at: new Date().toISOString(),
1265
1529
  type: "file.unshippable",
1266
- data: { path: file.path, bytes: Buffer.byteLength(file.content ?? file.target ?? "") },
1530
+ data: { path: file.path, bytes: Buffer.byteLength(file.content ?? "") },
1267
1531
  });
1268
1532
  },
1269
1533
  });
@@ -1272,7 +1536,7 @@ export class WorkspaceCollector {
1272
1536
  private async *iterTreeFiles(state: SessionState, paths: string[]): AsyncGenerator<CollectorFile> {
1273
1537
  for (const path of paths) {
1274
1538
  if (state.budgetExhausted) return;
1275
- for (const entry of await readWorkspaceEntries(state.root, path)) {
1539
+ for await (const entry of iterateWorkspaceEntries(state.root, path)) {
1276
1540
  yield entry;
1277
1541
  }
1278
1542
  }
@@ -1285,16 +1549,18 @@ export class WorkspaceCollector {
1285
1549
  * records — replaying start + sidecars in sequence order reproduces the
1286
1550
  * end tree.
1287
1551
  */
1288
- private async uploadChangeParts(state: SessionState, snapshot: ChangeJournalEntry[]): Promise<ChangeJournalEntry[][]> {
1552
+ private async uploadChangeParts(
1553
+ state: SessionState,
1554
+ units: Iterable<ChangeJournalEntry> | AsyncIterable<ChangeJournalEntry>,
1555
+ opts: { onAck?: (batch: ChangeJournalEntry[]) => void } = {},
1556
+ ): Promise<ChangeJournalEntry[][]> {
1289
1557
  const uploadedSlices: ChangeJournalEntry[][] = [];
1290
1558
  const filesForSlice = (entries: ChangeJournalEntry[]): CollectorFile[] => {
1291
1559
  const lastByPath = new Map<string, CollectorFile>();
1292
1560
  const slicedByPath = new Map<string, ChangeJournalEntry[]>();
1293
1561
  for (const entry of entries) {
1294
1562
  if (entry.status !== "present") continue;
1295
- if (entry.target !== undefined) {
1296
- lastByPath.set(entry.path, { path: entry.path, target: entry.target });
1297
- } else if (entry.slice) {
1563
+ if (entry.slice) {
1298
1564
  const group = slicedByPath.get(entry.path) ?? [];
1299
1565
  group.push(entry);
1300
1566
  slicedByPath.set(entry.path, group);
@@ -1344,9 +1610,9 @@ export class WorkspaceCollector {
1344
1610
  return files;
1345
1611
  };
1346
1612
  const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
1347
- await this.packAndUpload(state, "change", snapshot, {
1613
+ await this.packAndUpload(state, "change", units, {
1348
1614
  sizeOf: (entry) =>
1349
- Buffer.byteLength(entry.content ?? entry.target ?? "") + entry.path.length + 96,
1615
+ Buffer.byteLength(entry.content ?? "") + entry.path.length + 96,
1350
1616
  measure: async (batch) => {
1351
1617
  const files = filesForSlice(batch);
1352
1618
  const payload = buildEnvelopePayload(bySequence(), "change", files);
@@ -1354,14 +1620,17 @@ export class WorkspaceCollector {
1354
1620
  },
1355
1621
  upload: async (measured, batch) => {
1356
1622
  const ok = await this.uploadEnvelope(state, "change", measured.files, measured);
1357
- if (ok) uploadedSlices.push(batch);
1623
+ if (ok) {
1624
+ uploadedSlices.push(batch);
1625
+ opts.onAck?.(batch);
1626
+ }
1358
1627
  return ok;
1359
1628
  },
1360
1629
  drop: (entry) => {
1361
1630
  this.log("warn", "OmniRush change record exceeds the upload limits and cannot be shipped; skipped", {
1362
1631
  sessionId: state.id,
1363
1632
  path: entry.path,
1364
- bytes: Buffer.byteLength(entry.content ?? entry.target ?? ""),
1633
+ bytes: Buffer.byteLength(entry.content ?? ""),
1365
1634
  });
1366
1635
  state.trace.push({
1367
1636
  at: new Date().toISOString(),
@@ -1551,18 +1820,58 @@ export class WorkspaceCollector {
1551
1820
  body: compressed.buffer.slice(compressed.byteOffset, compressed.byteOffset + compressed.byteLength) as ArrayBuffer,
1552
1821
  signal: AbortSignal.timeout(120_000),
1553
1822
  });
1554
- let response = await send();
1555
- if (response.status === 401 && this.refresh) {
1556
- // Broker pattern: single-flight refresh, retry exactly once.
1557
- const rotated = await this.refresh(this.token).catch(() => null);
1558
- if (rotated) {
1559
- this.token = rotated;
1560
- response = await send();
1561
- }
1823
+ // Shared retry layer: 429/5xx/network ride out backend deploys with
1824
+ // exponential backoff + jitter. 401 stays a protocol answer handled
1825
+ // inline (single-flight refresh, retry exactly once).
1826
+ const outcome = await withRetries(
1827
+ async () => {
1828
+ let response = await send();
1829
+ if (response.status === 401 && this.refresh) {
1830
+ // Broker pattern: single-flight refresh, retry exactly once.
1831
+ const rotated = await this.refresh(this.token).catch(() => null);
1832
+ if (rotated) {
1833
+ this.token = rotated;
1834
+ response = await send();
1835
+ }
1836
+ }
1837
+ return {
1838
+ response,
1839
+ retryAfterSec: Number(response?.headers?.get?.("retry-after")) || undefined,
1840
+ };
1841
+ },
1842
+ {
1843
+ attempts: retryAttempts(),
1844
+ isRetryable: ({ response }) => isRetryableStatus(response?.status),
1845
+ onRetry: ({ attempt, attempts, delayMs }) => {
1846
+ this.log("warn", "OmniRush upload unavailable — retrying", {
1847
+ sessionId: state.id,
1848
+ snapshotType,
1849
+ attempt: `${attempt + 1}/${attempts}`,
1850
+ delayMs,
1851
+ });
1852
+ },
1853
+ },
1854
+ );
1855
+ const response = outcome.result?.response;
1856
+ if (!response) {
1857
+ // Every attempt failed at the network layer.
1858
+ throw new Error(
1859
+ `collector upload failed: network error after ${outcome.attempts} attempt${outcome.attempts === 1 ? "" : "s"}` +
1860
+ ` — ${outcome.error?.message ?? "unknown"} [ref ${outcome.ref}]` +
1861
+ ` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
1862
+ );
1562
1863
  }
1563
1864
  if (!response.ok) {
1564
- await response.body?.cancel().catch(() => undefined);
1565
- throw new Error(`collector upload failed with status ${response.status}`);
1865
+ // Diagnosable from a tester's screenshot alone: status, context,
1866
+ // and (bounded) the server's error body. A 401 here means the
1867
+ // refresh-retry above also failed — the caller should re-login.
1868
+ const snippet = await WorkspaceCollector.errorBodySnippet(response);
1869
+ throw new Error(
1870
+ `collector upload failed: HTTP ${response.status}` +
1871
+ `${snippet ? ` — ${snippet}` : ""}` +
1872
+ ` [ref ${outcome.ref}]` +
1873
+ ` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
1874
+ );
1566
1875
  }
1567
1876
  state.sentBytes += payload.length;
1568
1877
  state.sequence += 1;