omnirush 0.4.1 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -52,6 +52,9 @@ import { dirname, join, relative, resolve, sep } from "node:path";
52
52
  import { promisify } from "node:util";
53
53
  import { zstdCompress as zstdCompressCb } from "node:zlib";
54
54
 
55
+ import { isRetryableStatus, retryAttempts, withRetries } from "./retry";
56
+ import { buildTraceArtifact, parsePiSession } from "./trace-format";
57
+
55
58
  const execFileAsync = promisify(execFile);
56
59
 
57
60
  // --- limits ---------------------------------------------------------------
@@ -67,10 +70,31 @@ export const MAX_SESSION_BYTES = 100 * 1024 * 1024 * 1024;
67
70
  // byte size + sha256) so replay reassembles the original deterministically.
68
71
  export const MAX_SLICE_BYTES = 4 * 1024 * 1024;
69
72
  // Whole-read ceiling. Files up to this size are read and processed as
70
- // one string (worst case: binary -> base64 inflates 4/3 to ~427 MiB
71
- // chars, comfortably under the runtime's ~512 MiB MAX_STRING_LENGTH).
72
- // Bigger files are streamed in fixed raw chunks and sliced.
73
- const WHOLE_READ_BYTES = 320 * 1024 * 1024;
73
+ // one string; bigger files stream in fixed raw chunks and slice.
74
+ // Deliberately tiny (issue #6): a whole read holds buffer + utf8 string
75
+ // + redaction copy + base64 simultaneously (~4-5x file size), so even a
76
+ // modest ceiling spikes RSS by that multiple per file. 8 MiB bounds the
77
+ // worst case to ~40 MiB; anything bigger takes the streaming path with
78
+ // identical output.
79
+ const WHOLE_READ_BYTES = 8 * 1024 * 1024;
80
+ // Text-segment cap for newline-less files (see the streaming reader).
81
+ const PENDING_TEXT_CAP_BYTES = 32 * 1024 * 1024;
82
+ // Change-journal memory ceiling (issue #6, Windows 8-9 GiB OOM): the
83
+ // journal holds every changed file's content between debounced change
84
+ // uploads. At HIGH_WATER an upload is forced immediately (normal
85
+ // backpressure); at the HARD ceiling — a backend down for a long time —
86
+ // the OLDEST records are dropped, loudly (warn + journal.dropped trace),
87
+ // because losing replay data beats crashing the whole session.
88
+ // Soft disk cap for the on-disk change journal: when unacked journal
89
+ // bytes exceed it (a backend down for a very long time), capture stops
90
+ // loudly instead of filling the disk. The end snapshot still ships.
91
+ export const JOURNAL_DISK_CAP_BYTES = 1024 * 1024 * 1024;
92
+ /**
93
+ * Non-enumerable per-record field carrying the raw serialized line
94
+ * length (for journal ack accounting) — invisible to JSON.stringify so
95
+ * the changes.json sidecar never includes it.
96
+ */
97
+ export const RAW_BYTES_FIELD = Symbol("omnirushJournalRawBytes");
74
98
  // Multi-file parts target this compressed size (the backend body rail
75
99
  // was live-verified at >= 31.5 MiB compressed, so 15 MiB leaves ample
76
100
  // headroom for any ingress in front of the manager).
@@ -83,8 +107,10 @@ export const PART_UNCOMPRESSED_STEP = 8 * 1024 * 1024;
83
107
  // Flush a part when a check shows at least this much compressed payload.
84
108
  export const PART_READY_COMPRESSED_BYTES = 12 * 1024 * 1024;
85
109
  // Absolute uncompressed ceiling for one part batch — bounds resident
86
- // memory even for highly-compressible content.
87
- export const PART_BATCH_UNCOMPRESSED_MAX = 48 * 1024 * 1024;
110
+ // memory even for highly-compressible content (incompressible batches
111
+ // never hit the compressed-size thresholds, so this is what actually
112
+ // stops accumulation; the payload build doubles it transiently).
113
+ export const PART_BATCH_UNCOMPRESSED_MAX = 24 * 1024 * 1024;
88
114
  // Change/trace parts flush by PLAIN size so their __agent__/changes.json
89
115
  // / trace.json entries (single file entries whose content grows with the
90
116
  // batch) stay small and parts upload promptly.
@@ -141,6 +167,9 @@ export type SessionLedgerRecord = {
141
167
  sentBytes?: number;
142
168
  lastMessageId?: string;
143
169
  lastSeenAt: string;
170
+ /** Per-session-FILE transcript capture state (issue: trace format spec).
171
+ * Keys are absolute .jsonl paths; a size/mtime change re-captures. */
172
+ traces?: Record<string, { size: number; mtimeMs: number }>;
144
173
  };
145
174
 
146
175
  export type SessionLedger = {
@@ -176,8 +205,30 @@ type SessionState = {
176
205
  * fabricate deletions inside the workspace. */
177
206
  knownPaths: Set<string>;
178
207
  trace: TraceEvent[];
179
- /** Ordered change records — the replayable change log. */
180
- changeJournal: ChangeJournalEntry[];
208
+ /**
209
+ * The change journal lives ON DISK (ndjson sidecar, one JSON record
210
+ * per line) — never as an in-memory array (issue #6: the array held
211
+ * every changed file's content string and grew to multiple GB).
212
+ * Captures APPEND to the file; change uploads stream the unacked
213
+ * region and compact the acked prefix away. Appends during an upload
214
+ * simply land past the read window (bounded by `journalUploadEof`) —
215
+ * no gating, no memory growth.
216
+ */
217
+ journalPath: string | null;
218
+ /** File offset of the first not-yet-uploaded record. */
219
+ journalAckOffset: number;
220
+ /** Bytes appended so far (soft disk cap accounting). */
221
+ journalAppendedBytes: number;
222
+ /** Serializes journal appends. */
223
+ journalAppendTail: Promise<void>;
224
+ /** Set while a change upload reads the journal. */
225
+ journalUploadEof: number | null;
226
+ /** True once the soft disk cap stopped captures (warned once). */
227
+ journalCaptureStopped: boolean;
228
+ /** Last journal signature per path (consecutive-duplicate suppression). */
229
+ journalLastByPath: Map<string, string>;
230
+ /** Transcript capture state carried across runs via the ledger. */
231
+ transcripts: Record<string, { size: number; mtimeMs: number }>;
181
232
  changeCaptureTail: Promise<void>;
182
233
  ready: Promise<void>;
183
234
  tail: Promise<void>;
@@ -186,6 +237,13 @@ type SessionState = {
186
237
  export type CollectorOptions = {
187
238
  gatewayUrl?: string;
188
239
  accessToken?: string;
240
+ /** pi agent config dir (~/.pi/agent): session transcripts live under
241
+ * <agentDir>/sessions/<cwd-slug>/. Required for trace capture. */
242
+ agentDir?: string;
243
+ /** Resolves the signed-in user id for trace headers (best effort). */
244
+ identityProvider?: () => Promise<{ userId: string | null }>;
245
+ /** Stable client identifier for trace headers (e.g. hostname). */
246
+ clientId?: string;
189
247
  fetch?: typeof fetch;
190
248
  /** Returns the rotated access token, or null when refresh failed. */
191
249
  refresh?: (tokenUsed: string) => Promise<string | null>;
@@ -394,7 +452,7 @@ export function selectTraceEvents(
394
452
  * removed on purpose: `git ls-files` drops anything .gitignored (build
395
453
  * outputs, node_modules), which broke the whole-project guarantee.
396
454
  */
397
- async function listWorkspaceFiles(root: string): Promise<string[]> {
455
+ export async function listWorkspaceFiles(root: string): Promise<string[]> {
398
456
  return walkWorkspace(root);
399
457
  }
400
458
 
@@ -528,19 +586,23 @@ function sliceManifest(
528
586
  type ContentChunk = { content: string; encoding?: "base64" };
529
587
 
530
588
  /**
531
- * Read a regular file into <= ~8 MiB content chunks: small files in one
532
- * whole read (fully redacted), big files streamed in fixed raw chunks —
533
- * strings stay far below the runtime string limit regardless of file
534
- * size. Binary detection happens on the first chunk and applies to the
535
- * whole file.
589
+ * Stream a regular file as <= ~8 MiB content chunks, yielding each chunk
590
+ * the moment it is produced — a file's content is NEVER materialized as
591
+ * one string regardless of size (issue #6: the old array version held
592
+ * every chunk at once). Small files are a single whole read; big files
593
+ * stream in fixed raw chunks. Binary detection happens on the first
594
+ * chunk and applies to the whole file; text is redacted per
595
+ * line-aligned segment.
536
596
  */
537
- async function readContentChunks(absolute: string, size: number): Promise<ContentChunk[]> {
597
+ async function* iterateContentChunks(absolute: string, size: number): AsyncGenerator<ContentChunk> {
538
598
  if (size <= WHOLE_READ_BYTES) {
539
599
  const buffer = await readFile(absolute);
540
600
  if (isBinary(buffer)) {
541
- return [{ content: buffer.toString("base64"), encoding: "base64" }];
601
+ yield { content: buffer.toString("base64"), encoding: "base64" };
602
+ return;
542
603
  }
543
- return [{ content: redactCollectorText(buffer.toString("utf8")).text }];
604
+ yield { content: redactCollectorText(buffer.toString("utf8")).text };
605
+ return;
544
606
  }
545
607
  const { open } = await import("node:fs/promises");
546
608
  const handle = await open(absolute, "r");
@@ -549,114 +611,139 @@ async function readContentChunks(absolute: string, size: number): Promise<Conten
549
611
  const buffer = Buffer.alloc(chunkBytes);
550
612
  const first = await handle.read(buffer, 0, chunkBytes, null);
551
613
  const binary = isBinary(buffer.subarray(0, first.bytesRead));
552
- const chunks: ContentChunk[] = [];
553
614
  if (binary) {
554
615
  // Binaries carry no redactable text: base64 each raw chunk.
555
- chunks.push({ content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" });
616
+ yield { content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" };
556
617
  let read = 0;
557
618
  while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
558
- chunks.push({ content: buffer.subarray(0, read).toString("base64"), encoding: "base64" });
619
+ yield { content: buffer.subarray(0, read).toString("base64"), encoding: "base64" };
559
620
  }
560
- } else {
561
- // Text: decode incrementally (StringDecoder absorbs multibyte
562
- // sequences split across chunk boundaries), emit only up to the
563
- // last complete line, carry the remainder, and redact each
564
- // line-aligned segment.
565
- const { StringDecoder } = await import("node:string_decoder");
566
- const decoder = new StringDecoder("utf8");
567
- let pending = decoder.write(buffer.subarray(0, first.bytesRead));
568
- let read = 0;
569
- while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
570
- pending += decoder.write(buffer.subarray(0, read));
621
+ return;
622
+ }
623
+ // Text: decode incrementally (StringDecoder absorbs multibyte
624
+ // sequences split across chunk boundaries), emit only up to the
625
+ // last complete line, carry the remainder, and redact each
626
+ // line-aligned segment. A pathological newline-less file must not
627
+ // grow `pending` forever: at PENDING_TEXT_CAP_BYTES the segment is
628
+ // flushed mid-line (line-anchored redaction weakens at that rare
629
+ // cut; unbounded memory is worse).
630
+ const { StringDecoder } = await import("node:string_decoder");
631
+ const decoder = new StringDecoder("utf8");
632
+ let pending = decoder.write(buffer.subarray(0, first.bytesRead));
633
+ let read = 0;
634
+ while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
635
+ pending += decoder.write(buffer.subarray(0, read));
636
+ for (;;) {
571
637
  const lastNl = pending.lastIndexOf("\n");
572
638
  if (lastNl >= 0) {
573
- chunks.push({ content: redactCollectorText(pending.slice(0, lastNl + 1)).text });
639
+ yield { content: redactCollectorText(pending.slice(0, lastNl + 1)).text };
574
640
  pending = pending.slice(lastNl + 1);
641
+ continue;
642
+ }
643
+ if (pending.length >= PENDING_TEXT_CAP_BYTES) {
644
+ yield { content: redactCollectorText(pending).text };
645
+ pending = "";
575
646
  }
647
+ break;
576
648
  }
577
- pending += decoder.end();
578
- if (pending) chunks.push({ content: redactCollectorText(pending).text });
579
649
  }
580
- return chunks;
650
+ pending += decoder.end();
651
+ if (pending) yield { content: redactCollectorText(pending).text };
581
652
  } finally {
582
653
  await handle.close();
583
654
  }
584
655
  }
585
656
 
586
657
  /**
587
- * Read one workspace entry into CollectorFile entries. Regular files:
658
+ * Stream one workspace entry as CollectorFile entries. Regular files:
588
659
  * text is redacted, binaries base64-encoded ("encoding": "base64"), and
589
660
  * the POSIX mode rides along (octal string) so replay can restore the
590
661
  * executable bit (stored-and-ignored on Windows). Content bigger than
591
- * MAX_SLICE_BYTES is sliced (see buildFileEntries). Symlinks become text
592
- * entries { path, content: <target>, encoding: "symlink" } — the
593
- * backend's per-entry validation requires text content, and replay
594
- * recreates the actual link from the target. Returns [] when the entry
595
- * vanished or is a special file.
662
+ * MAX_SLICE_BYTES is sliced on the fly — slices are yielded the moment
663
+ * they exist while only the tiny sha256 descriptors accumulate, then
664
+ * the manifest is yielded LAST (replay unions manifests by index, so
665
+ * position never matters). Symlinks become text entries
666
+ * { path, content: <target>, encoding: "symlink" } — the backend's
667
+ * per-entry validation requires text content, and replay recreates the
668
+ * actual link from the target. Yields nothing when the entry vanished
669
+ * or is a special file.
596
670
  */
597
- async function readWorkspaceEntries(root: string, path: string): Promise<CollectorFile[]> {
671
+ async function* iterateWorkspaceEntries(root: string, path: string): AsyncGenerator<CollectorFile> {
672
+ let absolute = "";
598
673
  try {
599
- const absolute = resolve(root, path);
600
- if (portablePath(root, absolute).startsWith("../")) return [];
674
+ absolute = resolve(root, path);
675
+ if (portablePath(root, absolute).startsWith("../")) return;
601
676
  const file = await lstat(absolute);
602
677
  if (file.isSymbolicLink()) {
603
678
  const target = await readlink(absolute);
604
- return [{ path, content: target, encoding: "symlink" }];
679
+ yield { path, content: target, encoding: "symlink" };
680
+ return;
605
681
  }
606
- if (!file.isFile()) return [];
682
+ if (!file.isFile()) return;
607
683
  const mode = file.mode & 0o777;
608
- const chunks: ContentChunk[] = await readContentChunks(absolute, file.size);
609
- if (chunks.length === 1) {
610
- const chunk = chunks[0];
611
- return buildFileEntries({
612
- path,
613
- content: chunk.content,
614
- ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
615
- ...(mode ? { mode: mode.toString(8) } : {}),
616
- });
684
+ const modeField = mode ? { mode: mode.toString(8) } : {};
685
+ if (file.size <= WHOLE_READ_BYTES) {
686
+ // Small file: whole read, real path preserved; buildFileEntries
687
+ // slices (manifest + parts) when the encoded content exceeds
688
+ // MAX_SLICE_BYTES — same shape as always.
689
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
690
+ for (const entry of buildFileEntries({
691
+ path,
692
+ content: chunk.content,
693
+ ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
694
+ ...modeField,
695
+ })) {
696
+ yield entry;
697
+ }
698
+ }
699
+ return;
700
+ }
701
+ // Big file (streamed): slices are yielded the moment they exist;
702
+ // only the tiny sha256 descriptors accumulate. The reassembly
703
+ // manifest is yielded LAST (replay unions manifests by index, so
704
+ // position never matters).
705
+ let index = 0;
706
+ const descriptors: Array<{ index: number; bytes: number; sha256: string }> = [];
707
+ let sawEncoding: "base64" | undefined;
708
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
709
+ sawEncoding ??= chunk.encoding;
710
+ for (const slice of sliceString(chunk.content)) {
711
+ descriptors.push({ index, bytes: Buffer.byteLength(slice), sha256: sha256Hex(slice) });
712
+ yield {
713
+ path: slicePartPath(path, index),
714
+ content: slice,
715
+ ...(sawEncoding ? { encoding: sawEncoding } : {}),
716
+ };
717
+ index += 1;
718
+ }
617
719
  }
618
- // Big file: streamed chunks -> one logical entry per chunk (already
619
- // <= ~8 MiB for raw chunks; base64 inflates 4/3, so sub-slice those),
620
- // plus the manifest for deterministic reassembly.
621
- const encoding = chunks[0].encoding;
622
- const entries: CollectorFile[] = [];
623
- const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
624
- entries.push({
720
+ if (index === 0) return;
721
+ yield {
625
722
  path: sliceManifestPath(path),
626
723
  content: JSON.stringify(sliceManifest(
627
- {
628
- path,
629
- ...(encoding ? { encoding } : {}),
630
- ...(mode ? { mode: mode.toString(8) } : {}),
631
- },
632
- sliceContents.map((slice, index) => ({
633
- index,
634
- bytes: Buffer.byteLength(slice),
635
- sha256: sha256Hex(slice),
636
- })),
724
+ { path, ...(sawEncoding ? { encoding: sawEncoding } : {}), ...modeField },
725
+ descriptors,
637
726
  )),
638
- });
639
- sliceContents.forEach((slice, index) => {
640
- entries.push({
641
- path: slicePartPath(path, index),
642
- content: slice,
643
- ...(encoding ? { encoding } : {}),
644
- });
645
- });
646
- return entries;
727
+ };
728
+ return;
647
729
  } catch {
648
730
  // Workspaces are live; races are expected and retried by the next snapshot.
649
- return [];
731
+ return;
650
732
  }
651
733
  }
652
734
 
653
- export async function collectFiles(
735
+ /**
736
+ * Streaming file collection: yields workspace files (metadata entry
737
+ * first) one at a time so the one-shot path never holds the whole tree's
738
+ * content in memory (issue #6). Same order, denylist, and byte budget
739
+ * as the compat collectFiles drain below.
740
+ */
741
+ export async function* iterateCollectFiles(
654
742
  root: string,
655
743
  byteLimit: number,
656
744
  workspaceId: string,
657
745
  session?: Pick<SessionState, "id" | "segment" | "resumed">,
658
- ): Promise<CollectorFile[]> {
659
- const files: CollectorFile[] = [];
746
+ ): AsyncGenerator<CollectorFile> {
660
747
  let used = 0;
661
748
  const metadata = JSON.stringify({
662
749
  workspace_id: workspaceId,
@@ -668,19 +755,32 @@ export async function collectFiles(
668
755
  root_name: root.split(sep).filter(Boolean).at(-1) ?? "workspace",
669
756
  git: await gitMetadata(root),
670
757
  });
671
- files.push({ path: `${AGENT_DIR}/workspace.json`, content: metadata });
758
+ yield { path: `${AGENT_DIR}/workspace.json`, content: metadata };
672
759
  used += Buffer.byteLength(metadata);
673
760
 
674
761
  const paths = (await listWorkspaceFiles(root)).sort(comparePaths);
675
762
  for (const path of paths) {
676
763
  if (used >= byteLimit) break;
677
- for (const file of await readWorkspaceEntries(root, path)) {
764
+ for await (const file of iterateWorkspaceEntries(root, path)) {
678
765
  const size = Buffer.byteLength(file.content ?? "");
679
766
  if (used + size > byteLimit) continue;
680
- files.push(file);
767
+ yield file;
681
768
  used += size;
682
769
  }
683
770
  }
771
+ }
772
+
773
+ /** Compat drain of iterateCollectFiles (tests + tooling). */
774
+ export async function collectFiles(
775
+ root: string,
776
+ byteLimit: number,
777
+ workspaceId: string,
778
+ session?: Pick<SessionState, "id" | "segment" | "resumed">,
779
+ ): Promise<CollectorFile[]> {
780
+ const files: CollectorFile[] = [];
781
+ for await (const file of iterateCollectFiles(root, byteLimit, workspaceId, session)) {
782
+ files.push(file);
783
+ }
684
784
  return files;
685
785
  }
686
786
 
@@ -724,14 +824,11 @@ export type EnvelopePart = {
724
824
  };
725
825
 
726
826
  /**
727
- * Deterministic part split for a full files array (used by `omnirush
728
- * collect` and the tests): sort files by path, accumulate in order, keep
729
- * each part <= maxCompressed compressed. A single file bigger than one
730
- * part gets its own part; file content is never split across parts. A
731
- * lone file that cannot fit even the hard rail is skipped with a warning.
732
- * Envelope sequence numbers are state.sequence + 1 + partIndex.
827
+ * Deterministic part split for a full files array (compat drain of
828
+ * iterateEnvelopeParts, used by the tests): sort files by path,
829
+ * accumulate in order, keep each part <= maxCompressed compressed.
733
830
  */
734
- export function buildEnvelopeParts(
831
+ export async function buildEnvelopeParts(
735
832
  state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
736
833
  snapshotType: SnapshotType,
737
834
  files: CollectorFile[],
@@ -744,28 +841,34 @@ export function buildEnvelopeParts(
744
841
  const sorted = files
745
842
  .flatMap((file) => buildFileEntries(file))
746
843
  .sort((left, right) => comparePaths(left.path, right.path));
747
- return packEnvelopeParts(state, snapshotType, sorted, maxCompressed, log);
844
+ const parts: EnvelopePart[] = [];
845
+ for await (const part of iterateEnvelopeParts(state, snapshotType, sorted, maxCompressed, log)) {
846
+ parts.push(part);
847
+ }
848
+ return parts;
748
849
  }
749
850
 
750
851
  /**
751
- * Shared greedy packer: split an ordered file list into compressed parts,
752
- * preserving path order across parts. Used by buildEnvelopeParts (pure,
753
- * whole list in memory) and mirrored by the collector's streaming packer
754
- * (packAndUpload, which never holds more than one part in memory).
852
+ * Streaming packer for the one-shot path: yields compressed parts in
853
+ * order so callers can upload-and-discard instead of holding the whole
854
+ * workspace's parts (payload + compressed buffers) in memory (issue #6).
855
+ * Same greedy algorithm as before: split at the compressed target, a
856
+ * lone file may use the hard rail, lone-over-rail files are skipped
857
+ * loudly. File content is never split across parts.
755
858
  */
756
- export async function packEnvelopeParts(
859
+ export async function* iterateEnvelopeParts(
757
860
  state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
758
861
  snapshotType: SnapshotType,
759
- sortedFiles: CollectorFile[],
862
+ sortedFiles: Iterable<CollectorFile> | AsyncIterable<CollectorFile>,
760
863
  maxCompressed = MAX_PART_COMPRESSED_BYTES,
761
864
  log?: (message: string, attributes?: Record<string, unknown>) => void,
762
- ): Promise<EnvelopePart[]> {
865
+ ): AsyncGenerator<EnvelopePart> {
763
866
  const hard = Math.max(maxCompressed, MAX_PART_COMPRESSED_HARD);
764
867
  const measure = async (batch: CollectorFile[], partIndex: number): Promise<EnvelopePart> => {
765
868
  const payload = buildEnvelopePayload({ ...state, sequence: (state.sequence ?? 0) + partIndex }, snapshotType, batch);
766
869
  return { files: batch, payload, compressed: await compressZstd(payload) };
767
870
  };
768
- const parts: EnvelopePart[] = [];
871
+ let partsYielded = 0;
769
872
  let batch: CollectorFile[] = [];
770
873
  const fit = async (batchToFit: CollectorFile[], partIndex: number) => {
771
874
  let measured = await measure(batchToFit, partIndex);
@@ -787,15 +890,16 @@ export async function packEnvelopeParts(
787
890
  }
788
891
  return { measured, carry };
789
892
  };
790
- for (const file of sortedFiles) {
893
+ for await (const file of sortedFiles) {
791
894
  batch.push(file);
792
- const partIndex = parts.length;
895
+ const partIndex = partsYielded;
793
896
  const estimate = batch.reduce((total, f) => total + Buffer.byteLength(f.content), 0);
794
897
  if (estimate < PART_UNCOMPRESSED_STEP) continue;
795
898
  const { measured, carry } = await fit(batch, partIndex);
796
899
  if (measured
797
900
  && (measured.compressed.length >= PART_READY_COMPRESSED_BYTES || estimate >= PART_BATCH_UNCOMPRESSED_MAX)) {
798
- parts.push(measured);
901
+ partsYielded += 1;
902
+ yield measured;
799
903
  batch = carry;
800
904
  } else if (measured) {
801
905
  batch = [...batch, ...carry];
@@ -804,12 +908,14 @@ export async function packEnvelopeParts(
804
908
  }
805
909
  }
806
910
  while (batch.length > 0) {
807
- const { measured, carry } = await fit(batch, parts.length);
808
- if (measured) parts.push(measured);
911
+ const { measured, carry } = await fit(batch, partsYielded);
912
+ if (measured) {
913
+ partsYielded += 1;
914
+ yield measured;
915
+ }
809
916
  if (carry.length === 0) break;
810
917
  batch = carry;
811
918
  }
812
- return parts;
813
919
  }
814
920
 
815
921
  export class WorkspaceCollector {
@@ -840,6 +946,12 @@ export class WorkspaceCollector {
840
946
  private readonly uploader?: CollectorOptions["upload"];
841
947
  private readonly log: NonNullable<CollectorOptions["log"]>;
842
948
  private readonly ledgerPath: string | null;
949
+ private readonly stateDir: string | null;
950
+ private readonly agentDir: string | null;
951
+ private readonly identityProvider?: CollectorOptions["identityProvider"];
952
+ private readonly clientId?: string;
953
+ private identity: { userId: string | null } | null = null;
954
+ private identityTried = false;
843
955
  private readonly ledgerReady: Promise<void>;
844
956
  private ledger: SessionLedger = { version: 1, sessions: {} };
845
957
  private ledgerWriteTail: Promise<void> = Promise.resolve();
@@ -857,6 +969,10 @@ export class WorkspaceCollector {
857
969
  this.refresh = options.refresh;
858
970
  this.uploader = options.upload;
859
971
  this.log = options.log ?? (() => undefined);
972
+ this.stateDir = options.stateDir ? resolve(options.stateDir) : null;
973
+ this.agentDir = options.agentDir ? resolve(options.agentDir) : null;
974
+ this.identityProvider = options.identityProvider;
975
+ this.clientId = options.clientId;
860
976
  this.ledgerPath = options.stateDir ? join(resolve(options.stateDir), SESSION_LEDGER_FILE) : null;
861
977
  this.ledgerReady = this.loadLedger();
862
978
  this.changeDebounceMs = options.changeDebounceMs ?? CHANGE_DEBOUNCE_MS;
@@ -898,6 +1014,9 @@ export class WorkspaceCollector {
898
1014
  state.sentBytes = previous?.sentBytes ?? 0;
899
1015
  state.lastMessageId = previous?.lastMessageId;
900
1016
  state.resumed = Boolean(previous);
1017
+ if (previous?.traces && typeof previous.traces === "object") {
1018
+ state.transcripts = { ...previous.traces };
1019
+ }
901
1020
  if (state.resumed) {
902
1021
  state.trace.push({
903
1022
  at: new Date().toISOString(),
@@ -922,6 +1041,7 @@ export class WorkspaceCollector {
922
1041
  nextSequence: state.sequence,
923
1042
  sentBytes: state.sentBytes,
924
1043
  ...(state.lastMessageId ? { lastMessageId: state.lastMessageId } : {}),
1044
+ ...(Object.keys(state.transcripts).length > 0 ? { traces: state.transcripts } : {}),
925
1045
  lastSeenAt: new Date().toISOString(),
926
1046
  };
927
1047
  await this.saveLedger();
@@ -966,7 +1086,14 @@ export class WorkspaceCollector {
966
1086
  watcherStartedAtMs: 0,
967
1087
  knownPaths: new Set(),
968
1088
  trace: [],
969
- changeJournal: [],
1089
+ journalPath: this.stateDir ? join(resolve(this.stateDir), `journal-${sessionId}.ndjson`) : null,
1090
+ journalAckOffset: 0,
1091
+ journalAppendedBytes: 0,
1092
+ journalAppendTail: Promise.resolve(),
1093
+ journalUploadEof: null,
1094
+ journalCaptureStopped: false,
1095
+ journalLastByPath: new Map(),
1096
+ transcripts: {},
970
1097
  changeCaptureTail: Promise.resolve(),
971
1098
  ready: Promise.resolve(),
972
1099
  tail: Promise.resolve(),
@@ -1031,6 +1158,7 @@ export class WorkspaceCollector {
1031
1158
  if (state.trace.length > 0) pendingTrace.push(...state.trace.splice(0));
1032
1159
  await state.changeCaptureTail;
1033
1160
  if (finalTrace !== undefined) pendingTrace.push({ at: new Date().toISOString(), type: "session.completed", data: finalTrace });
1161
+ await this.teardownJournal(state); // ship unacked change records, then unlink
1034
1162
  await this.uploadTrace(state, pendingTrace);
1035
1163
  await this.uploadWorkspace(state, "end");
1036
1164
  this.sessions.delete(sessionId);
@@ -1079,18 +1207,28 @@ export class WorkspaceCollector {
1079
1207
  });
1080
1208
  }
1081
1209
 
1210
+ /** Run the debounced change-snapshot upload immediately. */
1211
+ private flushChangeSnapshot(state: SessionState): void {
1212
+ if (state.finished || state.budgetExhausted) return;
1213
+ if (state.changeTimer) {
1214
+ clearTimeout(state.changeTimer);
1215
+ state.changeTimer = null;
1216
+ }
1217
+ this.enqueue(state, async () => {
1218
+ if (state.budgetExhausted || state.finished) return;
1219
+ const signature = await workspaceSignature(state.root);
1220
+ if (!signature || signature === state.lastSignature) return;
1221
+ state.lastSignature = signature;
1222
+ await this.uploadWorkspace(state, "change");
1223
+ });
1224
+ }
1225
+
1082
1226
  private scheduleChange(state: SessionState): void {
1083
1227
  if (state.finished || state.budgetExhausted) return;
1084
1228
  if (state.changeTimer) clearTimeout(state.changeTimer);
1085
1229
  state.changeTimer = setTimeout(() => {
1086
1230
  state.changeTimer = null;
1087
- this.enqueue(state, async () => {
1088
- if (state.budgetExhausted || state.finished) return;
1089
- const signature = await workspaceSignature(state.root);
1090
- if (!signature || signature === state.lastSignature) return;
1091
- state.lastSignature = signature;
1092
- await this.uploadWorkspace(state, "change");
1093
- });
1231
+ this.flushChangeSnapshot(state);
1094
1232
  }, this.changeDebounceMs);
1095
1233
  state.changeTimer.unref?.();
1096
1234
  }
@@ -1149,38 +1287,38 @@ export class WorkspaceCollector {
1149
1287
  const at = new Date().toISOString();
1150
1288
  const mode = file.mode & 0o777;
1151
1289
  const modeField = mode ? { mode: mode.toString(8) } : {};
1152
- const chunks = await readContentChunks(absolute, file.size);
1153
- const encoding = chunks[0].encoding;
1154
- const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
1155
- if (sliceContents.length === 1) {
1156
- entry = {
1157
- path,
1158
- at,
1159
- status: "present",
1160
- content: sliceContents[0],
1161
- ...(encoding ? { encoding } : {}),
1162
- ...modeField,
1163
- };
1164
- } else {
1165
- // Big file: one ordered record per slice so the journal stays
1166
- // replayable; reassembly rides the slice metadata.
1167
- sliceContents.forEach((content, index) => {
1290
+ // STREAMED capture: each chunk becomes a journal record the
1291
+ // moment it exists — a big file is never held in memory whole
1292
+ // (issue #6). One record per <= MAX_SLICE_BYTES slice; slice
1293
+ // metadata rides the records only when reassembly is needed.
1294
+ let sliceCount = 0;
1295
+ for await (const chunk of iterateContentChunks(absolute, file.size)) {
1296
+ for (const content of sliceString(chunk.content)) {
1297
+ sliceCount += 1;
1168
1298
  records.push({
1169
1299
  path,
1170
1300
  at,
1171
1301
  status: "present",
1172
1302
  content,
1173
- ...(encoding ? { encoding } : {}),
1303
+ ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
1174
1304
  ...modeField,
1175
- slice: {
1176
- index,
1177
- total: sliceContents.length,
1178
- bytes: Buffer.byteLength(content),
1179
- sha256: sha256Hex(content),
1180
- },
1181
1305
  });
1306
+ }
1307
+ }
1308
+ if (sliceCount > 1) {
1309
+ records.forEach((record, index) => {
1310
+ record.slice = {
1311
+ index,
1312
+ total: sliceCount,
1313
+ bytes: Buffer.byteLength(record.content ?? ""),
1314
+ sha256: sha256Hex(record.content ?? ""),
1315
+ };
1182
1316
  });
1183
- state.changeJournal.push(...records);
1317
+ }
1318
+ if (sliceCount === 1) {
1319
+ entry = records.pop()!;
1320
+ } else {
1321
+ await this.appendJournalRecords(state, records);
1184
1322
  return;
1185
1323
  }
1186
1324
  }
@@ -1199,27 +1337,156 @@ export class WorkspaceCollector {
1199
1337
  // Consecutive-duplicate suppression: macOS FSEvents can deliver a
1200
1338
  // late event for a path that was also captured directly, and an
1201
1339
  // identical re-capture adds upload fat without a state change. The
1202
- // replay result is identical either way.
1340
+ // replay result is identical either way. The last signature per
1341
+ // path lives in a tiny in-memory map (path strings only — content
1342
+ // is already on disk in the journal).
1203
1343
  const journalSignature = (record: ChangeJournalEntry) =>
1204
1344
  JSON.stringify([record.status, record.content, record.encoding, record.mode]);
1205
1345
  const pendingRecords = [entry, ...records];
1346
+ const accepted: ChangeJournalEntry[] = [];
1206
1347
  for (const record of pendingRecords) {
1207
- let duplicate = false;
1208
- for (let i = state.changeJournal.length - 1, scanned = 0; i >= 0 && scanned < 100; i--, scanned++) {
1209
- const prior = state.changeJournal[i];
1210
- if (prior.path !== record.path) continue;
1211
- duplicate = journalSignature(prior) === journalSignature(record);
1212
- break;
1348
+ const signature = journalSignature(record);
1349
+ if (state.journalLastByPath.get(record.path) === signature) continue;
1350
+ state.journalLastByPath.set(record.path, signature);
1351
+ accepted.push(record);
1352
+ }
1353
+ await this.appendJournalRecords(state, accepted);
1354
+ }
1355
+
1356
+ /**
1357
+ * APPEND records to the on-disk journal (ndjson — one JSON record per
1358
+ * line, 0600). Returns immediately when the capture hit the soft disk
1359
+ * cap (a backend down for a very long time must stop CAPTURING loudly
1360
+ * rather than fill the disk or RAM — the end snapshot still ships the
1361
+ * final tree). Serialized through the per-session append chain.
1362
+ */
1363
+ private appendJournalRecords(state: SessionState, records: ChangeJournalEntry[]): Promise<void> {
1364
+ if (!state.journalPath || records.length === 0) return Promise.resolve();
1365
+ if (state.journalCaptureStopped) return Promise.resolve();
1366
+ const unacked = state.journalAppendedBytes - state.journalAckOffset;
1367
+ const capMb = Number(process.env.OMNIRUSH_JOURNAL_DISK_CAP_MB);
1368
+ const capBytes = Number.isFinite(capMb) && capMb > 0 ? capMb * 1024 * 1024 : JOURNAL_DISK_CAP_BYTES;
1369
+ if (unacked > capBytes) {
1370
+ state.journalCaptureStopped = true;
1371
+ this.log("warn", "OmniRush change journal hit the disk cap — change capture stopped for this session", {
1372
+ sessionId: state.id,
1373
+ capBytes: capBytes,
1374
+ unackedBytes: unacked,
1375
+ });
1376
+ state.trace.push({
1377
+ at: new Date().toISOString(),
1378
+ type: "journal.capture_stopped",
1379
+ data: { cap_bytes: capBytes, unacked_bytes: unacked },
1380
+ });
1381
+ return Promise.resolve();
1382
+ }
1383
+ const lines = records.map((record) => JSON.stringify(record) + "\n").join("");
1384
+ const bytes = Buffer.byteLength(lines);
1385
+ state.journalAppendTail = state.journalAppendTail
1386
+ .catch(() => undefined)
1387
+ .then(async () => {
1388
+ const { appendFile, mkdir } = await import("node:fs/promises");
1389
+ await mkdir(dirname(state.journalPath!), { recursive: true, mode: 0o700 });
1390
+ await appendFile(state.journalPath!, lines, { encoding: "utf8", mode: 0o600 });
1391
+ state.journalAppendedBytes += bytes;
1392
+ });
1393
+ return state.journalAppendTail;
1394
+ }
1395
+
1396
+ /**
1397
+ * Stream the journal's unacked records (from journalAckOffset to the
1398
+ * upload-start EOF) as parsed entries, each tagged with its raw line
1399
+ * bytes via a NON-enumerable property so the changes.json sidecar
1400
+ * serialization never sees it. Records appended while the upload runs
1401
+ * land beyond the read window and simply ship with the next snapshot.
1402
+ */
1403
+ private async *readJournalRecords(state: SessionState): AsyncGenerator<ChangeJournalEntry> {
1404
+ if (!state.journalPath || state.journalAckOffset >= state.journalUploadEof) return;
1405
+ const { createReadStream } = await import("node:fs");
1406
+ const readline = await import("node:readline");
1407
+ const stream = createReadStream(state.journalPath, {
1408
+ start: state.journalAckOffset,
1409
+ end: state.journalUploadEof! - 1,
1410
+ encoding: "utf8",
1411
+ });
1412
+ const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
1413
+ for await (const line of rl) {
1414
+ if (!line.trim()) continue;
1415
+ try {
1416
+ const record = JSON.parse(line) as ChangeJournalEntry;
1417
+ Object.defineProperty(record, RAW_BYTES_FIELD, {
1418
+ value: Buffer.byteLength(line) + 1,
1419
+ enumerable: false,
1420
+ configurable: true,
1421
+ });
1422
+ yield record;
1423
+ } catch {
1424
+ this.log("warn", "OmniRush journal line was unreadable — skipped", { sessionId: state.id });
1213
1425
  }
1214
- if (!duplicate) state.changeJournal.push(record);
1215
1426
  }
1216
1427
  }
1217
1428
 
1218
- /** Remove exactly the uploaded entries from the ordered journal. */
1219
- private acknowledgeJournal(state: SessionState, uploaded: ChangeJournalEntry[]): void {
1220
- if (uploaded.length === 0) return;
1221
- const sent = uploaded.length > 32 ? new Set(uploaded) : null;
1222
- state.changeJournal = state.changeJournal.filter((entry) => sent ? !sent.has(entry) : !uploaded.includes(entry));
1429
+ /**
1430
+ * Upload the journal's unacked records as change snapshots, then
1431
+ * compact: everything acked is removed from the file; records appended
1432
+ * during the upload (past journalUploadEof) survive untouched.
1433
+ */
1434
+ private async uploadChangeJournal(state: SessionState): Promise<void> {
1435
+ if (!state.journalPath) return;
1436
+ let fileSize = 0;
1437
+ try {
1438
+ fileSize = (await lstat(state.journalPath)).size;
1439
+ } catch {
1440
+ return; // no journal yet — nothing captured
1441
+ }
1442
+ if (fileSize <= state.journalAckOffset) return;
1443
+ const ackStart = state.journalAckOffset;
1444
+ state.journalUploadEof = fileSize;
1445
+ try {
1446
+ await this.uploadChangeParts(state, this.readJournalRecords(state), {
1447
+ onAck: (batch) => {
1448
+ for (const record of batch) {
1449
+ const raw = (record as Record<symbol, unknown>)[RAW_BYTES_FIELD];
1450
+ if (typeof raw === "number") state.journalAckOffset += raw;
1451
+ }
1452
+ },
1453
+ });
1454
+ } finally {
1455
+ state.journalUploadEof = null;
1456
+ }
1457
+ if (state.journalAckOffset <= ackStart) return; // nothing acked — keep everything
1458
+ await this.compactJournal(state, state.journalAckOffset);
1459
+ state.journalAckOffset = 0;
1460
+ }
1461
+
1462
+ /** Rewrite the journal without its first `dropBytes` bytes. */
1463
+ private async compactJournal(state: SessionState, dropBytes: number): Promise<void> {
1464
+ if (!state.journalPath) return;
1465
+ const { createReadStream, createWriteStream } = await import("node:fs");
1466
+ const { rename } = await import("node:fs/promises");
1467
+ const tmpPath = `${state.journalPath}.compact`;
1468
+ await new Promise<void>((resolvePromise, rejectPromise) => {
1469
+ const input = createReadStream(state.journalPath!, { start: dropBytes, encoding: "utf8" });
1470
+ const output = createWriteStream(tmpPath, { encoding: "utf8", mode: 0o600 });
1471
+ input.on("error", rejectPromise);
1472
+ output.on("error", rejectPromise);
1473
+ output.on("finish", () => resolvePromise());
1474
+ input.pipe(output);
1475
+ });
1476
+ await rename(tmpPath, state.journalPath);
1477
+ }
1478
+
1479
+ /** Upload unacked journal records (best effort) and remove the file. */
1480
+ private async teardownJournal(state: SessionState): Promise<void> {
1481
+ try {
1482
+ await this.uploadChangeJournal(state);
1483
+ } catch {
1484
+ /* best effort — the end snapshot still ships the final tree */
1485
+ }
1486
+ if (state.journalPath) {
1487
+ const { rm } = await import("node:fs/promises");
1488
+ await rm(state.journalPath, { force: true }).catch(() => undefined);
1489
+ }
1223
1490
  }
1224
1491
 
1225
1492
  private enqueue(state: SessionState, operation: () => Promise<void>): void {
@@ -1234,17 +1501,10 @@ export class WorkspaceCollector {
1234
1501
  private async uploadWorkspace(state: SessionState, type: "start" | "change" | "end"): Promise<void> {
1235
1502
  if (state.budgetExhausted) return;
1236
1503
  if (type === "change") {
1237
- // Change snapshots carry ONLY the touched files from the journal —
1238
- // never a full re-enumeration of the tree.
1239
- const snapshot = [...state.changeJournal];
1240
- if (snapshot.length === 0) {
1241
- state.lastSignature = await workspaceSignature(state.root);
1242
- return;
1243
- }
1244
- const uploaded = await this.uploadChangeParts(state, snapshot);
1245
- // Acknowledge only after every part of the snapshot succeeded; on a
1246
- // mid-budget exhaustion, acknowledge only what actually shipped.
1247
- this.acknowledgeJournal(state, state.budgetExhausted ? uploaded.flat() : snapshot);
1504
+ // Change snapshots carry ONLY the touched files — streamed from
1505
+ // the on-disk journal, acked per part, then compacted. Never a
1506
+ // full re-enumeration of the tree, never held in memory.
1507
+ await this.uploadChangeJournal(state);
1248
1508
  } else {
1249
1509
  await this.uploadTreeParts(state, type);
1250
1510
  }
@@ -1302,7 +1562,7 @@ export class WorkspaceCollector {
1302
1562
  private async *iterTreeFiles(state: SessionState, paths: string[]): AsyncGenerator<CollectorFile> {
1303
1563
  for (const path of paths) {
1304
1564
  if (state.budgetExhausted) return;
1305
- for (const entry of await readWorkspaceEntries(state.root, path)) {
1565
+ for await (const entry of iterateWorkspaceEntries(state.root, path)) {
1306
1566
  yield entry;
1307
1567
  }
1308
1568
  }
@@ -1315,7 +1575,11 @@ export class WorkspaceCollector {
1315
1575
  * records — replaying start + sidecars in sequence order reproduces the
1316
1576
  * end tree.
1317
1577
  */
1318
- private async uploadChangeParts(state: SessionState, snapshot: ChangeJournalEntry[]): Promise<ChangeJournalEntry[][]> {
1578
+ private async uploadChangeParts(
1579
+ state: SessionState,
1580
+ units: Iterable<ChangeJournalEntry> | AsyncIterable<ChangeJournalEntry>,
1581
+ opts: { onAck?: (batch: ChangeJournalEntry[]) => void } = {},
1582
+ ): Promise<ChangeJournalEntry[][]> {
1319
1583
  const uploadedSlices: ChangeJournalEntry[][] = [];
1320
1584
  const filesForSlice = (entries: ChangeJournalEntry[]): CollectorFile[] => {
1321
1585
  const lastByPath = new Map<string, CollectorFile>();
@@ -1372,7 +1636,7 @@ export class WorkspaceCollector {
1372
1636
  return files;
1373
1637
  };
1374
1638
  const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
1375
- await this.packAndUpload(state, "change", snapshot, {
1639
+ await this.packAndUpload(state, "change", units, {
1376
1640
  sizeOf: (entry) =>
1377
1641
  Buffer.byteLength(entry.content ?? "") + entry.path.length + 96,
1378
1642
  measure: async (batch) => {
@@ -1382,7 +1646,10 @@ export class WorkspaceCollector {
1382
1646
  },
1383
1647
  upload: async (measured, batch) => {
1384
1648
  const ok = await this.uploadEnvelope(state, "change", measured.files, measured);
1385
- if (ok) uploadedSlices.push(batch);
1649
+ if (ok) {
1650
+ uploadedSlices.push(batch);
1651
+ opts.onAck?.(batch);
1652
+ }
1386
1653
  return ok;
1387
1654
  },
1388
1655
  drop: (entry) => {
@@ -1487,7 +1754,12 @@ export class WorkspaceCollector {
1487
1754
 
1488
1755
  /** Chunked trace upload: event subsets per part, ordered by sequence. */
1489
1756
  private async uploadTrace(state: SessionState, traceEvents: TraceEvent[]): Promise<void> {
1490
- if (state.budgetExhausted || traceEvents.length === 0) return;
1757
+ if (state.budgetExhausted || traceEvents.length === 0) {
1758
+ // No lifecycle events this round — transcript artifacts may still
1759
+ // be due (new or changed session files).
1760
+ await this.uploadSessionTranscripts(state);
1761
+ return;
1762
+ }
1491
1763
  const remaining = this.sessionBudgetBytes - state.sentBytes;
1492
1764
  if (remaining <= 1024) {
1493
1765
  this.exhaustSessionBudget(state);
@@ -1538,6 +1810,128 @@ export class WorkspaceCollector {
1538
1810
  });
1539
1811
  },
1540
1812
  }, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
1813
+ // Full-conversation transcript artifacts (trace format spec) ride
1814
+ // the same trace uploads — lifecycle events stay in trace.json, the
1815
+ // transcripts land as __agent__/traces/<session>.jsonl.
1816
+ await this.uploadSessionTranscripts(state);
1817
+ }
1818
+
1819
+ /**
1820
+ * The pi agent writes the FULL session transcript as JSONL under
1821
+ * <agentDir>/sessions/<cwd-slug>/<ts>_<session-id>.jsonl — one file
1822
+ * per run, SUBAGENTS as their own files in the same dir. Every file
1823
+ * (new or changed since the last capture) becomes its own trace
1824
+ * artifact (header + full conversation, redacted) uploaded as
1825
+ * __agent__/traces/<session>.jsonl and linked to this session via
1826
+ * parent_session_id. Detection state rides the persistent ledger.
1827
+ */
1828
+ private async uploadSessionTranscripts(state: SessionState): Promise<void> {
1829
+ if (!this.agentDir || !state.id) return;
1830
+ const { readdir, stat } = await import("node:fs/promises");
1831
+ const sessionsRoot = join(this.agentDir, "sessions");
1832
+ let sessionDir: string | null = null;
1833
+ try {
1834
+ for (const entry of await readdir(sessionsRoot, { withFileTypes: true })) {
1835
+ if (!entry.isDirectory()) continue;
1836
+ const dir = join(sessionsRoot, entry.name);
1837
+ const files = await readdir(dir).catch(() => [] as string[]);
1838
+ if (files.some((f) => f.endsWith(`_${state.id}.jsonl`))) {
1839
+ sessionDir = dir;
1840
+ break;
1841
+ }
1842
+ }
1843
+ } catch {
1844
+ return; // no sessions dir — nothing to capture
1845
+ }
1846
+ if (!sessionDir) return;
1847
+
1848
+ const artifacts: Array<{ path: string; content: string; fileKey: string; size: number; mtimeMs: number }> = [];
1849
+ let files: string[] = [];
1850
+ try {
1851
+ files = (await readdir(sessionDir)).filter((f) => f.endsWith(".jsonl")).sort();
1852
+ } catch {
1853
+ return;
1854
+ }
1855
+ for (const file of files) {
1856
+ const filePath = join(sessionDir, file);
1857
+ let size = 0;
1858
+ let mtimeMs = 0;
1859
+ try {
1860
+ const stats = await stat(filePath);
1861
+ size = stats.size;
1862
+ mtimeMs = stats.mtimeMs;
1863
+ } catch {
1864
+ continue;
1865
+ }
1866
+ const known = state.transcripts[filePath];
1867
+ if (known && known.size === size && known.mtimeMs === mtimeMs) continue;
1868
+ let text: string;
1869
+ try {
1870
+ const { readFile } = await import("node:fs/promises");
1871
+ text = await readFile(filePath, "utf8");
1872
+ } catch {
1873
+ continue;
1874
+ }
1875
+ const parsed = parsePiSession(text);
1876
+ const artifactSessionId = parsed.sessionId ?? file.replace(/^\d{4}-\d{2}-\d{2}T[\d-]+Z_/, "").replace(/\.jsonl$/, "");
1877
+ if (!artifactSessionId || parsed.messageRecords.length === 0) {
1878
+ // Still mark empty/HEAD-only files seen so they are not retried.
1879
+ state.transcripts[filePath] = { size, mtimeMs };
1880
+ continue;
1881
+ }
1882
+ const identity = await this.resolveIdentity();
1883
+ const artifact = buildTraceArtifact({
1884
+ sessionId: artifactSessionId,
1885
+ parentSessionId: state.id,
1886
+ userId: identity?.userId ?? null,
1887
+ clientId: this.clientId ?? null,
1888
+ agentId: "omnirush-cli",
1889
+ parsed,
1890
+ redact: (value) => redactCollectorText(value).text,
1891
+ });
1892
+ artifacts.push({
1893
+ path: `${AGENT_DIR}/traces/${artifactSessionId}.jsonl`,
1894
+ content: JSON.stringify(artifact),
1895
+ fileKey: filePath,
1896
+ size,
1897
+ mtimeMs,
1898
+ });
1899
+ state.transcripts[filePath] = { size, mtimeMs };
1900
+ }
1901
+ if (artifacts.length === 0) {
1902
+ await this.persistSession(state).catch(() => undefined);
1903
+ return;
1904
+ }
1905
+ const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
1906
+ await this.packAndUpload(state, "trace", artifacts, {
1907
+ sizeOf: (artifact) => Buffer.byteLength(artifact.content) + artifact.path.length + 64,
1908
+ measure: async (batch) => {
1909
+ const payload = buildEnvelopePayload(bySequence(), "trace", batch);
1910
+ return { files: batch, payload, compressed: await compressZstd(payload) };
1911
+ },
1912
+ upload: async (measured) => this.uploadEnvelope(state, "trace", measured.files, measured),
1913
+ drop: (artifact) => {
1914
+ this.log("warn", "OmniRush transcript artifact exceeds the upload limits; skipped", {
1915
+ sessionId: state.id,
1916
+ path: artifact.path,
1917
+ bytes: Buffer.byteLength(artifact.content),
1918
+ });
1919
+ },
1920
+ }, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
1921
+ await this.persistSession(state).catch(() => undefined);
1922
+ }
1923
+
1924
+ /** Best-effort /device/me identity, fetched once per collector. */
1925
+ private async resolveIdentity(): Promise<{ userId: string | null } | null> {
1926
+ if (this.identityTried) return this.identity;
1927
+ this.identityTried = true;
1928
+ if (!this.identityProvider) return null;
1929
+ try {
1930
+ this.identity = await this.identityProvider();
1931
+ } catch {
1932
+ this.identity = null;
1933
+ }
1934
+ return this.identity;
1541
1935
  }
1542
1936
 
1543
1937
  /**
@@ -1579,14 +1973,46 @@ export class WorkspaceCollector {
1579
1973
  body: compressed.buffer.slice(compressed.byteOffset, compressed.byteOffset + compressed.byteLength) as ArrayBuffer,
1580
1974
  signal: AbortSignal.timeout(120_000),
1581
1975
  });
1582
- let response = await send();
1583
- if (response.status === 401 && this.refresh) {
1584
- // Broker pattern: single-flight refresh, retry exactly once.
1585
- const rotated = await this.refresh(this.token).catch(() => null);
1586
- if (rotated) {
1587
- this.token = rotated;
1588
- response = await send();
1589
- }
1976
+ // Shared retry layer: 429/5xx/network ride out backend deploys with
1977
+ // exponential backoff + jitter. 401 stays a protocol answer handled
1978
+ // inline (single-flight refresh, retry exactly once).
1979
+ const outcome = await withRetries(
1980
+ async () => {
1981
+ let response = await send();
1982
+ if (response.status === 401 && this.refresh) {
1983
+ // Broker pattern: single-flight refresh, retry exactly once.
1984
+ const rotated = await this.refresh(this.token).catch(() => null);
1985
+ if (rotated) {
1986
+ this.token = rotated;
1987
+ response = await send();
1988
+ }
1989
+ }
1990
+ return {
1991
+ response,
1992
+ retryAfterSec: Number(response?.headers?.get?.("retry-after")) || undefined,
1993
+ };
1994
+ },
1995
+ {
1996
+ attempts: retryAttempts(),
1997
+ isRetryable: ({ response }) => isRetryableStatus(response?.status),
1998
+ onRetry: ({ attempt, attempts, delayMs }) => {
1999
+ this.log("warn", "OmniRush upload unavailable — retrying", {
2000
+ sessionId: state.id,
2001
+ snapshotType,
2002
+ attempt: `${attempt + 1}/${attempts}`,
2003
+ delayMs,
2004
+ });
2005
+ },
2006
+ },
2007
+ );
2008
+ const response = outcome.result?.response;
2009
+ if (!response) {
2010
+ // Every attempt failed at the network layer.
2011
+ throw new Error(
2012
+ `collector upload failed: network error after ${outcome.attempts} attempt${outcome.attempts === 1 ? "" : "s"}` +
2013
+ ` — ${outcome.error?.message ?? "unknown"} [ref ${outcome.ref}]` +
2014
+ ` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
2015
+ );
1590
2016
  }
1591
2017
  if (!response.ok) {
1592
2018
  // Diagnosable from a tester's screenshot alone: status, context,
@@ -1596,6 +2022,7 @@ export class WorkspaceCollector {
1596
2022
  throw new Error(
1597
2023
  `collector upload failed: HTTP ${response.status}` +
1598
2024
  `${snippet ? ` — ${snippet}` : ""}` +
2025
+ ` [ref ${outcome.ref}]` +
1599
2026
  ` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
1600
2027
  );
1601
2028
  }