omnirush 0.4.1 → 0.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/collect-once.ts +77 -39
- package/assets/extensions/omnirush/auth.js +56 -17
- package/assets/extensions/omnirush/collector-lib.ts +452 -178
- package/assets/extensions/omnirush/retry.js +144 -0
- package/assets/extensions/omnirush/sota-lib.ts +64 -0
- package/assets/extensions/omnirush/sota.ts +159 -28
- package/package.json +1 -1
- package/scripts/postinstall.js +25 -8
- package/src/bin.js +253 -42
- package/src/lib.js +13 -0
|
@@ -52,6 +52,8 @@ import { dirname, join, relative, resolve, sep } from "node:path";
|
|
|
52
52
|
import { promisify } from "node:util";
|
|
53
53
|
import { zstdCompress as zstdCompressCb } from "node:zlib";
|
|
54
54
|
|
|
55
|
+
import { isRetryableStatus, retryAttempts, withRetries } from "./retry";
|
|
56
|
+
|
|
55
57
|
const execFileAsync = promisify(execFile);
|
|
56
58
|
|
|
57
59
|
// --- limits ---------------------------------------------------------------
|
|
@@ -67,10 +69,31 @@ export const MAX_SESSION_BYTES = 100 * 1024 * 1024 * 1024;
|
|
|
67
69
|
// byte size + sha256) so replay reassembles the original deterministically.
|
|
68
70
|
export const MAX_SLICE_BYTES = 4 * 1024 * 1024;
|
|
69
71
|
// Whole-read ceiling. Files up to this size are read and processed as
|
|
70
|
-
// one string
|
|
71
|
-
//
|
|
72
|
-
//
|
|
73
|
-
|
|
72
|
+
// one string; bigger files stream in fixed raw chunks and slice.
|
|
73
|
+
// Deliberately tiny (issue #6): a whole read holds buffer + utf8 string
|
|
74
|
+
// + redaction copy + base64 simultaneously (~4-5x file size), so even a
|
|
75
|
+
// modest ceiling spikes RSS by that multiple per file. 8 MiB bounds the
|
|
76
|
+
// worst case to ~40 MiB; anything bigger takes the streaming path with
|
|
77
|
+
// identical output.
|
|
78
|
+
const WHOLE_READ_BYTES = 8 * 1024 * 1024;
|
|
79
|
+
// Text-segment cap for newline-less files (see the streaming reader).
|
|
80
|
+
const PENDING_TEXT_CAP_BYTES = 32 * 1024 * 1024;
|
|
81
|
+
// Change-journal memory ceiling (issue #6, Windows 8-9 GiB OOM): the
|
|
82
|
+
// journal holds every changed file's content between debounced change
|
|
83
|
+
// uploads. At HIGH_WATER an upload is forced immediately (normal
|
|
84
|
+
// backpressure); at the HARD ceiling — a backend down for a long time —
|
|
85
|
+
// the OLDEST records are dropped, loudly (warn + journal.dropped trace),
|
|
86
|
+
// because losing replay data beats crashing the whole session.
|
|
87
|
+
// Soft disk cap for the on-disk change journal: when unacked journal
|
|
88
|
+
// bytes exceed it (a backend down for a very long time), capture stops
|
|
89
|
+
// loudly instead of filling the disk. The end snapshot still ships.
|
|
90
|
+
export const JOURNAL_DISK_CAP_BYTES = 1024 * 1024 * 1024;
|
|
91
|
+
/**
|
|
92
|
+
* Non-enumerable per-record field carrying the raw serialized line
|
|
93
|
+
* length (for journal ack accounting) — invisible to JSON.stringify so
|
|
94
|
+
* the changes.json sidecar never includes it.
|
|
95
|
+
*/
|
|
96
|
+
export const RAW_BYTES_FIELD = Symbol("omnirushJournalRawBytes");
|
|
74
97
|
// Multi-file parts target this compressed size (the backend body rail
|
|
75
98
|
// was live-verified at >= 31.5 MiB compressed, so 15 MiB leaves ample
|
|
76
99
|
// headroom for any ingress in front of the manager).
|
|
@@ -83,8 +106,10 @@ export const PART_UNCOMPRESSED_STEP = 8 * 1024 * 1024;
|
|
|
83
106
|
// Flush a part when a check shows at least this much compressed payload.
|
|
84
107
|
export const PART_READY_COMPRESSED_BYTES = 12 * 1024 * 1024;
|
|
85
108
|
// Absolute uncompressed ceiling for one part batch — bounds resident
|
|
86
|
-
// memory even for highly-compressible content
|
|
87
|
-
|
|
109
|
+
// memory even for highly-compressible content (incompressible batches
|
|
110
|
+
// never hit the compressed-size thresholds, so this is what actually
|
|
111
|
+
// stops accumulation; the payload build doubles it transiently).
|
|
112
|
+
export const PART_BATCH_UNCOMPRESSED_MAX = 24 * 1024 * 1024;
|
|
88
113
|
// Change/trace parts flush by PLAIN size so their __agent__/changes.json
|
|
89
114
|
// / trace.json entries (single file entries whose content grows with the
|
|
90
115
|
// batch) stay small and parts upload promptly.
|
|
@@ -176,8 +201,28 @@ type SessionState = {
|
|
|
176
201
|
* fabricate deletions inside the workspace. */
|
|
177
202
|
knownPaths: Set<string>;
|
|
178
203
|
trace: TraceEvent[];
|
|
179
|
-
/**
|
|
180
|
-
|
|
204
|
+
/**
|
|
205
|
+
* The change journal lives ON DISK (ndjson sidecar, one JSON record
|
|
206
|
+
* per line) — never as an in-memory array (issue #6: the array held
|
|
207
|
+
* every changed file's content string and grew to multiple GB).
|
|
208
|
+
* Captures APPEND to the file; change uploads stream the unacked
|
|
209
|
+
* region and compact the acked prefix away. Appends during an upload
|
|
210
|
+
* simply land past the read window (bounded by `journalUploadEof`) —
|
|
211
|
+
* no gating, no memory growth.
|
|
212
|
+
*/
|
|
213
|
+
journalPath: string | null;
|
|
214
|
+
/** File offset of the first not-yet-uploaded record. */
|
|
215
|
+
journalAckOffset: number;
|
|
216
|
+
/** Bytes appended so far (soft disk cap accounting). */
|
|
217
|
+
journalAppendedBytes: number;
|
|
218
|
+
/** Serializes journal appends. */
|
|
219
|
+
journalAppendTail: Promise<void>;
|
|
220
|
+
/** Set while a change upload reads the journal. */
|
|
221
|
+
journalUploadEof: number | null;
|
|
222
|
+
/** True once the soft disk cap stopped captures (warned once). */
|
|
223
|
+
journalCaptureStopped: boolean;
|
|
224
|
+
/** Last journal signature per path (consecutive-duplicate suppression). */
|
|
225
|
+
journalLastByPath: Map<string, string>;
|
|
181
226
|
changeCaptureTail: Promise<void>;
|
|
182
227
|
ready: Promise<void>;
|
|
183
228
|
tail: Promise<void>;
|
|
@@ -394,7 +439,7 @@ export function selectTraceEvents(
|
|
|
394
439
|
* removed on purpose: `git ls-files` drops anything .gitignored (build
|
|
395
440
|
* outputs, node_modules), which broke the whole-project guarantee.
|
|
396
441
|
*/
|
|
397
|
-
async function listWorkspaceFiles(root: string): Promise<string[]> {
|
|
442
|
+
export async function listWorkspaceFiles(root: string): Promise<string[]> {
|
|
398
443
|
return walkWorkspace(root);
|
|
399
444
|
}
|
|
400
445
|
|
|
@@ -528,19 +573,23 @@ function sliceManifest(
|
|
|
528
573
|
type ContentChunk = { content: string; encoding?: "base64" };
|
|
529
574
|
|
|
530
575
|
/**
|
|
531
|
-
*
|
|
532
|
-
*
|
|
533
|
-
*
|
|
534
|
-
*
|
|
535
|
-
*
|
|
576
|
+
* Stream a regular file as <= ~8 MiB content chunks, yielding each chunk
|
|
577
|
+
* the moment it is produced — a file's content is NEVER materialized as
|
|
578
|
+
* one string regardless of size (issue #6: the old array version held
|
|
579
|
+
* every chunk at once). Small files are a single whole read; big files
|
|
580
|
+
* stream in fixed raw chunks. Binary detection happens on the first
|
|
581
|
+
* chunk and applies to the whole file; text is redacted per
|
|
582
|
+
* line-aligned segment.
|
|
536
583
|
*/
|
|
537
|
-
async function
|
|
584
|
+
async function* iterateContentChunks(absolute: string, size: number): AsyncGenerator<ContentChunk> {
|
|
538
585
|
if (size <= WHOLE_READ_BYTES) {
|
|
539
586
|
const buffer = await readFile(absolute);
|
|
540
587
|
if (isBinary(buffer)) {
|
|
541
|
-
|
|
588
|
+
yield { content: buffer.toString("base64"), encoding: "base64" };
|
|
589
|
+
return;
|
|
542
590
|
}
|
|
543
|
-
|
|
591
|
+
yield { content: redactCollectorText(buffer.toString("utf8")).text };
|
|
592
|
+
return;
|
|
544
593
|
}
|
|
545
594
|
const { open } = await import("node:fs/promises");
|
|
546
595
|
const handle = await open(absolute, "r");
|
|
@@ -549,114 +598,139 @@ async function readContentChunks(absolute: string, size: number): Promise<Conten
|
|
|
549
598
|
const buffer = Buffer.alloc(chunkBytes);
|
|
550
599
|
const first = await handle.read(buffer, 0, chunkBytes, null);
|
|
551
600
|
const binary = isBinary(buffer.subarray(0, first.bytesRead));
|
|
552
|
-
const chunks: ContentChunk[] = [];
|
|
553
601
|
if (binary) {
|
|
554
602
|
// Binaries carry no redactable text: base64 each raw chunk.
|
|
555
|
-
|
|
603
|
+
yield { content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" };
|
|
556
604
|
let read = 0;
|
|
557
605
|
while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
|
|
558
|
-
|
|
606
|
+
yield { content: buffer.subarray(0, read).toString("base64"), encoding: "base64" };
|
|
559
607
|
}
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
608
|
+
return;
|
|
609
|
+
}
|
|
610
|
+
// Text: decode incrementally (StringDecoder absorbs multibyte
|
|
611
|
+
// sequences split across chunk boundaries), emit only up to the
|
|
612
|
+
// last complete line, carry the remainder, and redact each
|
|
613
|
+
// line-aligned segment. A pathological newline-less file must not
|
|
614
|
+
// grow `pending` forever: at PENDING_TEXT_CAP_BYTES the segment is
|
|
615
|
+
// flushed mid-line (line-anchored redaction weakens at that rare
|
|
616
|
+
// cut; unbounded memory is worse).
|
|
617
|
+
const { StringDecoder } = await import("node:string_decoder");
|
|
618
|
+
const decoder = new StringDecoder("utf8");
|
|
619
|
+
let pending = decoder.write(buffer.subarray(0, first.bytesRead));
|
|
620
|
+
let read = 0;
|
|
621
|
+
while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
|
|
622
|
+
pending += decoder.write(buffer.subarray(0, read));
|
|
623
|
+
for (;;) {
|
|
571
624
|
const lastNl = pending.lastIndexOf("\n");
|
|
572
625
|
if (lastNl >= 0) {
|
|
573
|
-
|
|
626
|
+
yield { content: redactCollectorText(pending.slice(0, lastNl + 1)).text };
|
|
574
627
|
pending = pending.slice(lastNl + 1);
|
|
628
|
+
continue;
|
|
629
|
+
}
|
|
630
|
+
if (pending.length >= PENDING_TEXT_CAP_BYTES) {
|
|
631
|
+
yield { content: redactCollectorText(pending).text };
|
|
632
|
+
pending = "";
|
|
575
633
|
}
|
|
634
|
+
break;
|
|
576
635
|
}
|
|
577
|
-
pending += decoder.end();
|
|
578
|
-
if (pending) chunks.push({ content: redactCollectorText(pending).text });
|
|
579
636
|
}
|
|
580
|
-
|
|
637
|
+
pending += decoder.end();
|
|
638
|
+
if (pending) yield { content: redactCollectorText(pending).text };
|
|
581
639
|
} finally {
|
|
582
640
|
await handle.close();
|
|
583
641
|
}
|
|
584
642
|
}
|
|
585
643
|
|
|
586
644
|
/**
|
|
587
|
-
*
|
|
645
|
+
* Stream one workspace entry as CollectorFile entries. Regular files:
|
|
588
646
|
* text is redacted, binaries base64-encoded ("encoding": "base64"), and
|
|
589
647
|
* the POSIX mode rides along (octal string) so replay can restore the
|
|
590
648
|
* executable bit (stored-and-ignored on Windows). Content bigger than
|
|
591
|
-
* MAX_SLICE_BYTES is sliced
|
|
592
|
-
*
|
|
593
|
-
*
|
|
594
|
-
*
|
|
595
|
-
*
|
|
649
|
+
* MAX_SLICE_BYTES is sliced on the fly — slices are yielded the moment
|
|
650
|
+
* they exist while only the tiny sha256 descriptors accumulate, then
|
|
651
|
+
* the manifest is yielded LAST (replay unions manifests by index, so
|
|
652
|
+
* position never matters). Symlinks become text entries
|
|
653
|
+
* { path, content: <target>, encoding: "symlink" } — the backend's
|
|
654
|
+
* per-entry validation requires text content, and replay recreates the
|
|
655
|
+
* actual link from the target. Yields nothing when the entry vanished
|
|
656
|
+
* or is a special file.
|
|
596
657
|
*/
|
|
597
|
-
async function
|
|
658
|
+
async function* iterateWorkspaceEntries(root: string, path: string): AsyncGenerator<CollectorFile> {
|
|
659
|
+
let absolute = "";
|
|
598
660
|
try {
|
|
599
|
-
|
|
600
|
-
if (portablePath(root, absolute).startsWith("../")) return
|
|
661
|
+
absolute = resolve(root, path);
|
|
662
|
+
if (portablePath(root, absolute).startsWith("../")) return;
|
|
601
663
|
const file = await lstat(absolute);
|
|
602
664
|
if (file.isSymbolicLink()) {
|
|
603
665
|
const target = await readlink(absolute);
|
|
604
|
-
|
|
666
|
+
yield { path, content: target, encoding: "symlink" };
|
|
667
|
+
return;
|
|
605
668
|
}
|
|
606
|
-
if (!file.isFile()) return
|
|
669
|
+
if (!file.isFile()) return;
|
|
607
670
|
const mode = file.mode & 0o777;
|
|
608
|
-
const
|
|
609
|
-
if (
|
|
610
|
-
|
|
611
|
-
|
|
612
|
-
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
|
|
616
|
-
|
|
671
|
+
const modeField = mode ? { mode: mode.toString(8) } : {};
|
|
672
|
+
if (file.size <= WHOLE_READ_BYTES) {
|
|
673
|
+
// Small file: whole read, real path preserved; buildFileEntries
|
|
674
|
+
// slices (manifest + parts) when the encoded content exceeds
|
|
675
|
+
// MAX_SLICE_BYTES — same shape as always.
|
|
676
|
+
for await (const chunk of iterateContentChunks(absolute, file.size)) {
|
|
677
|
+
for (const entry of buildFileEntries({
|
|
678
|
+
path,
|
|
679
|
+
content: chunk.content,
|
|
680
|
+
...(chunk.encoding ? { encoding: chunk.encoding } : {}),
|
|
681
|
+
...modeField,
|
|
682
|
+
})) {
|
|
683
|
+
yield entry;
|
|
684
|
+
}
|
|
685
|
+
}
|
|
686
|
+
return;
|
|
687
|
+
}
|
|
688
|
+
// Big file (streamed): slices are yielded the moment they exist;
|
|
689
|
+
// only the tiny sha256 descriptors accumulate. The reassembly
|
|
690
|
+
// manifest is yielded LAST (replay unions manifests by index, so
|
|
691
|
+
// position never matters).
|
|
692
|
+
let index = 0;
|
|
693
|
+
const descriptors: Array<{ index: number; bytes: number; sha256: string }> = [];
|
|
694
|
+
let sawEncoding: "base64" | undefined;
|
|
695
|
+
for await (const chunk of iterateContentChunks(absolute, file.size)) {
|
|
696
|
+
sawEncoding ??= chunk.encoding;
|
|
697
|
+
for (const slice of sliceString(chunk.content)) {
|
|
698
|
+
descriptors.push({ index, bytes: Buffer.byteLength(slice), sha256: sha256Hex(slice) });
|
|
699
|
+
yield {
|
|
700
|
+
path: slicePartPath(path, index),
|
|
701
|
+
content: slice,
|
|
702
|
+
...(sawEncoding ? { encoding: sawEncoding } : {}),
|
|
703
|
+
};
|
|
704
|
+
index += 1;
|
|
705
|
+
}
|
|
617
706
|
}
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
// plus the manifest for deterministic reassembly.
|
|
621
|
-
const encoding = chunks[0].encoding;
|
|
622
|
-
const entries: CollectorFile[] = [];
|
|
623
|
-
const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
|
|
624
|
-
entries.push({
|
|
707
|
+
if (index === 0) return;
|
|
708
|
+
yield {
|
|
625
709
|
path: sliceManifestPath(path),
|
|
626
710
|
content: JSON.stringify(sliceManifest(
|
|
627
|
-
{
|
|
628
|
-
|
|
629
|
-
...(encoding ? { encoding } : {}),
|
|
630
|
-
...(mode ? { mode: mode.toString(8) } : {}),
|
|
631
|
-
},
|
|
632
|
-
sliceContents.map((slice, index) => ({
|
|
633
|
-
index,
|
|
634
|
-
bytes: Buffer.byteLength(slice),
|
|
635
|
-
sha256: sha256Hex(slice),
|
|
636
|
-
})),
|
|
711
|
+
{ path, ...(sawEncoding ? { encoding: sawEncoding } : {}), ...modeField },
|
|
712
|
+
descriptors,
|
|
637
713
|
)),
|
|
638
|
-
}
|
|
639
|
-
|
|
640
|
-
entries.push({
|
|
641
|
-
path: slicePartPath(path, index),
|
|
642
|
-
content: slice,
|
|
643
|
-
...(encoding ? { encoding } : {}),
|
|
644
|
-
});
|
|
645
|
-
});
|
|
646
|
-
return entries;
|
|
714
|
+
};
|
|
715
|
+
return;
|
|
647
716
|
} catch {
|
|
648
717
|
// Workspaces are live; races are expected and retried by the next snapshot.
|
|
649
|
-
return
|
|
718
|
+
return;
|
|
650
719
|
}
|
|
651
720
|
}
|
|
652
721
|
|
|
653
|
-
|
|
722
|
+
/**
|
|
723
|
+
* Streaming file collection: yields workspace files (metadata entry
|
|
724
|
+
* first) one at a time so the one-shot path never holds the whole tree's
|
|
725
|
+
* content in memory (issue #6). Same order, denylist, and byte budget
|
|
726
|
+
* as the compat collectFiles drain below.
|
|
727
|
+
*/
|
|
728
|
+
export async function* iterateCollectFiles(
|
|
654
729
|
root: string,
|
|
655
730
|
byteLimit: number,
|
|
656
731
|
workspaceId: string,
|
|
657
732
|
session?: Pick<SessionState, "id" | "segment" | "resumed">,
|
|
658
|
-
):
|
|
659
|
-
const files: CollectorFile[] = [];
|
|
733
|
+
): AsyncGenerator<CollectorFile> {
|
|
660
734
|
let used = 0;
|
|
661
735
|
const metadata = JSON.stringify({
|
|
662
736
|
workspace_id: workspaceId,
|
|
@@ -668,19 +742,32 @@ export async function collectFiles(
|
|
|
668
742
|
root_name: root.split(sep).filter(Boolean).at(-1) ?? "workspace",
|
|
669
743
|
git: await gitMetadata(root),
|
|
670
744
|
});
|
|
671
|
-
|
|
745
|
+
yield { path: `${AGENT_DIR}/workspace.json`, content: metadata };
|
|
672
746
|
used += Buffer.byteLength(metadata);
|
|
673
747
|
|
|
674
748
|
const paths = (await listWorkspaceFiles(root)).sort(comparePaths);
|
|
675
749
|
for (const path of paths) {
|
|
676
750
|
if (used >= byteLimit) break;
|
|
677
|
-
for (const file of
|
|
751
|
+
for await (const file of iterateWorkspaceEntries(root, path)) {
|
|
678
752
|
const size = Buffer.byteLength(file.content ?? "");
|
|
679
753
|
if (used + size > byteLimit) continue;
|
|
680
|
-
|
|
754
|
+
yield file;
|
|
681
755
|
used += size;
|
|
682
756
|
}
|
|
683
757
|
}
|
|
758
|
+
}
|
|
759
|
+
|
|
760
|
+
/** Compat drain of iterateCollectFiles (tests + tooling). */
|
|
761
|
+
export async function collectFiles(
|
|
762
|
+
root: string,
|
|
763
|
+
byteLimit: number,
|
|
764
|
+
workspaceId: string,
|
|
765
|
+
session?: Pick<SessionState, "id" | "segment" | "resumed">,
|
|
766
|
+
): Promise<CollectorFile[]> {
|
|
767
|
+
const files: CollectorFile[] = [];
|
|
768
|
+
for await (const file of iterateCollectFiles(root, byteLimit, workspaceId, session)) {
|
|
769
|
+
files.push(file);
|
|
770
|
+
}
|
|
684
771
|
return files;
|
|
685
772
|
}
|
|
686
773
|
|
|
@@ -724,14 +811,11 @@ export type EnvelopePart = {
|
|
|
724
811
|
};
|
|
725
812
|
|
|
726
813
|
/**
|
|
727
|
-
* Deterministic part split for a full files array (
|
|
728
|
-
*
|
|
729
|
-
* each part <= maxCompressed compressed.
|
|
730
|
-
* part gets its own part; file content is never split across parts. A
|
|
731
|
-
* lone file that cannot fit even the hard rail is skipped with a warning.
|
|
732
|
-
* Envelope sequence numbers are state.sequence + 1 + partIndex.
|
|
814
|
+
* Deterministic part split for a full files array (compat drain of
|
|
815
|
+
* iterateEnvelopeParts, used by the tests): sort files by path,
|
|
816
|
+
* accumulate in order, keep each part <= maxCompressed compressed.
|
|
733
817
|
*/
|
|
734
|
-
export function buildEnvelopeParts(
|
|
818
|
+
export async function buildEnvelopeParts(
|
|
735
819
|
state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
|
|
736
820
|
snapshotType: SnapshotType,
|
|
737
821
|
files: CollectorFile[],
|
|
@@ -744,28 +828,34 @@ export function buildEnvelopeParts(
|
|
|
744
828
|
const sorted = files
|
|
745
829
|
.flatMap((file) => buildFileEntries(file))
|
|
746
830
|
.sort((left, right) => comparePaths(left.path, right.path));
|
|
747
|
-
|
|
831
|
+
const parts: EnvelopePart[] = [];
|
|
832
|
+
for await (const part of iterateEnvelopeParts(state, snapshotType, sorted, maxCompressed, log)) {
|
|
833
|
+
parts.push(part);
|
|
834
|
+
}
|
|
835
|
+
return parts;
|
|
748
836
|
}
|
|
749
837
|
|
|
750
838
|
/**
|
|
751
|
-
*
|
|
752
|
-
*
|
|
753
|
-
*
|
|
754
|
-
*
|
|
839
|
+
* Streaming packer for the one-shot path: yields compressed parts in
|
|
840
|
+
* order so callers can upload-and-discard instead of holding the whole
|
|
841
|
+
* workspace's parts (payload + compressed buffers) in memory (issue #6).
|
|
842
|
+
* Same greedy algorithm as before: split at the compressed target, a
|
|
843
|
+
* lone file may use the hard rail, lone-over-rail files are skipped
|
|
844
|
+
* loudly. File content is never split across parts.
|
|
755
845
|
*/
|
|
756
|
-
export async function
|
|
846
|
+
export async function* iterateEnvelopeParts(
|
|
757
847
|
state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
|
|
758
848
|
snapshotType: SnapshotType,
|
|
759
|
-
sortedFiles: CollectorFile
|
|
849
|
+
sortedFiles: Iterable<CollectorFile> | AsyncIterable<CollectorFile>,
|
|
760
850
|
maxCompressed = MAX_PART_COMPRESSED_BYTES,
|
|
761
851
|
log?: (message: string, attributes?: Record<string, unknown>) => void,
|
|
762
|
-
):
|
|
852
|
+
): AsyncGenerator<EnvelopePart> {
|
|
763
853
|
const hard = Math.max(maxCompressed, MAX_PART_COMPRESSED_HARD);
|
|
764
854
|
const measure = async (batch: CollectorFile[], partIndex: number): Promise<EnvelopePart> => {
|
|
765
855
|
const payload = buildEnvelopePayload({ ...state, sequence: (state.sequence ?? 0) + partIndex }, snapshotType, batch);
|
|
766
856
|
return { files: batch, payload, compressed: await compressZstd(payload) };
|
|
767
857
|
};
|
|
768
|
-
|
|
858
|
+
let partsYielded = 0;
|
|
769
859
|
let batch: CollectorFile[] = [];
|
|
770
860
|
const fit = async (batchToFit: CollectorFile[], partIndex: number) => {
|
|
771
861
|
let measured = await measure(batchToFit, partIndex);
|
|
@@ -787,15 +877,16 @@ export async function packEnvelopeParts(
|
|
|
787
877
|
}
|
|
788
878
|
return { measured, carry };
|
|
789
879
|
};
|
|
790
|
-
for (const file of sortedFiles) {
|
|
880
|
+
for await (const file of sortedFiles) {
|
|
791
881
|
batch.push(file);
|
|
792
|
-
const partIndex =
|
|
882
|
+
const partIndex = partsYielded;
|
|
793
883
|
const estimate = batch.reduce((total, f) => total + Buffer.byteLength(f.content), 0);
|
|
794
884
|
if (estimate < PART_UNCOMPRESSED_STEP) continue;
|
|
795
885
|
const { measured, carry } = await fit(batch, partIndex);
|
|
796
886
|
if (measured
|
|
797
887
|
&& (measured.compressed.length >= PART_READY_COMPRESSED_BYTES || estimate >= PART_BATCH_UNCOMPRESSED_MAX)) {
|
|
798
|
-
|
|
888
|
+
partsYielded += 1;
|
|
889
|
+
yield measured;
|
|
799
890
|
batch = carry;
|
|
800
891
|
} else if (measured) {
|
|
801
892
|
batch = [...batch, ...carry];
|
|
@@ -804,12 +895,14 @@ export async function packEnvelopeParts(
|
|
|
804
895
|
}
|
|
805
896
|
}
|
|
806
897
|
while (batch.length > 0) {
|
|
807
|
-
const { measured, carry } = await fit(batch,
|
|
808
|
-
if (measured)
|
|
898
|
+
const { measured, carry } = await fit(batch, partsYielded);
|
|
899
|
+
if (measured) {
|
|
900
|
+
partsYielded += 1;
|
|
901
|
+
yield measured;
|
|
902
|
+
}
|
|
809
903
|
if (carry.length === 0) break;
|
|
810
904
|
batch = carry;
|
|
811
905
|
}
|
|
812
|
-
return parts;
|
|
813
906
|
}
|
|
814
907
|
|
|
815
908
|
export class WorkspaceCollector {
|
|
@@ -840,6 +933,7 @@ export class WorkspaceCollector {
|
|
|
840
933
|
private readonly uploader?: CollectorOptions["upload"];
|
|
841
934
|
private readonly log: NonNullable<CollectorOptions["log"]>;
|
|
842
935
|
private readonly ledgerPath: string | null;
|
|
936
|
+
private readonly stateDir: string | null;
|
|
843
937
|
private readonly ledgerReady: Promise<void>;
|
|
844
938
|
private ledger: SessionLedger = { version: 1, sessions: {} };
|
|
845
939
|
private ledgerWriteTail: Promise<void> = Promise.resolve();
|
|
@@ -857,6 +951,7 @@ export class WorkspaceCollector {
|
|
|
857
951
|
this.refresh = options.refresh;
|
|
858
952
|
this.uploader = options.upload;
|
|
859
953
|
this.log = options.log ?? (() => undefined);
|
|
954
|
+
this.stateDir = options.stateDir ? resolve(options.stateDir) : null;
|
|
860
955
|
this.ledgerPath = options.stateDir ? join(resolve(options.stateDir), SESSION_LEDGER_FILE) : null;
|
|
861
956
|
this.ledgerReady = this.loadLedger();
|
|
862
957
|
this.changeDebounceMs = options.changeDebounceMs ?? CHANGE_DEBOUNCE_MS;
|
|
@@ -966,7 +1061,13 @@ export class WorkspaceCollector {
|
|
|
966
1061
|
watcherStartedAtMs: 0,
|
|
967
1062
|
knownPaths: new Set(),
|
|
968
1063
|
trace: [],
|
|
969
|
-
|
|
1064
|
+
journalPath: this.stateDir ? join(resolve(this.stateDir), `journal-${sessionId}.ndjson`) : null,
|
|
1065
|
+
journalAckOffset: 0,
|
|
1066
|
+
journalAppendedBytes: 0,
|
|
1067
|
+
journalAppendTail: Promise.resolve(),
|
|
1068
|
+
journalUploadEof: null,
|
|
1069
|
+
journalCaptureStopped: false,
|
|
1070
|
+
journalLastByPath: new Map(),
|
|
970
1071
|
changeCaptureTail: Promise.resolve(),
|
|
971
1072
|
ready: Promise.resolve(),
|
|
972
1073
|
tail: Promise.resolve(),
|
|
@@ -1031,6 +1132,7 @@ export class WorkspaceCollector {
|
|
|
1031
1132
|
if (state.trace.length > 0) pendingTrace.push(...state.trace.splice(0));
|
|
1032
1133
|
await state.changeCaptureTail;
|
|
1033
1134
|
if (finalTrace !== undefined) pendingTrace.push({ at: new Date().toISOString(), type: "session.completed", data: finalTrace });
|
|
1135
|
+
await this.teardownJournal(state); // ship unacked change records, then unlink
|
|
1034
1136
|
await this.uploadTrace(state, pendingTrace);
|
|
1035
1137
|
await this.uploadWorkspace(state, "end");
|
|
1036
1138
|
this.sessions.delete(sessionId);
|
|
@@ -1079,18 +1181,28 @@ export class WorkspaceCollector {
|
|
|
1079
1181
|
});
|
|
1080
1182
|
}
|
|
1081
1183
|
|
|
1184
|
+
/** Run the debounced change-snapshot upload immediately. */
|
|
1185
|
+
private flushChangeSnapshot(state: SessionState): void {
|
|
1186
|
+
if (state.finished || state.budgetExhausted) return;
|
|
1187
|
+
if (state.changeTimer) {
|
|
1188
|
+
clearTimeout(state.changeTimer);
|
|
1189
|
+
state.changeTimer = null;
|
|
1190
|
+
}
|
|
1191
|
+
this.enqueue(state, async () => {
|
|
1192
|
+
if (state.budgetExhausted || state.finished) return;
|
|
1193
|
+
const signature = await workspaceSignature(state.root);
|
|
1194
|
+
if (!signature || signature === state.lastSignature) return;
|
|
1195
|
+
state.lastSignature = signature;
|
|
1196
|
+
await this.uploadWorkspace(state, "change");
|
|
1197
|
+
});
|
|
1198
|
+
}
|
|
1199
|
+
|
|
1082
1200
|
private scheduleChange(state: SessionState): void {
|
|
1083
1201
|
if (state.finished || state.budgetExhausted) return;
|
|
1084
1202
|
if (state.changeTimer) clearTimeout(state.changeTimer);
|
|
1085
1203
|
state.changeTimer = setTimeout(() => {
|
|
1086
1204
|
state.changeTimer = null;
|
|
1087
|
-
this.
|
|
1088
|
-
if (state.budgetExhausted || state.finished) return;
|
|
1089
|
-
const signature = await workspaceSignature(state.root);
|
|
1090
|
-
if (!signature || signature === state.lastSignature) return;
|
|
1091
|
-
state.lastSignature = signature;
|
|
1092
|
-
await this.uploadWorkspace(state, "change");
|
|
1093
|
-
});
|
|
1205
|
+
this.flushChangeSnapshot(state);
|
|
1094
1206
|
}, this.changeDebounceMs);
|
|
1095
1207
|
state.changeTimer.unref?.();
|
|
1096
1208
|
}
|
|
@@ -1149,38 +1261,38 @@ export class WorkspaceCollector {
|
|
|
1149
1261
|
const at = new Date().toISOString();
|
|
1150
1262
|
const mode = file.mode & 0o777;
|
|
1151
1263
|
const modeField = mode ? { mode: mode.toString(8) } : {};
|
|
1152
|
-
|
|
1153
|
-
|
|
1154
|
-
|
|
1155
|
-
|
|
1156
|
-
|
|
1157
|
-
|
|
1158
|
-
|
|
1159
|
-
|
|
1160
|
-
content: sliceContents[0],
|
|
1161
|
-
...(encoding ? { encoding } : {}),
|
|
1162
|
-
...modeField,
|
|
1163
|
-
};
|
|
1164
|
-
} else {
|
|
1165
|
-
// Big file: one ordered record per slice so the journal stays
|
|
1166
|
-
// replayable; reassembly rides the slice metadata.
|
|
1167
|
-
sliceContents.forEach((content, index) => {
|
|
1264
|
+
// STREAMED capture: each chunk becomes a journal record the
|
|
1265
|
+
// moment it exists — a big file is never held in memory whole
|
|
1266
|
+
// (issue #6). One record per <= MAX_SLICE_BYTES slice; slice
|
|
1267
|
+
// metadata rides the records only when reassembly is needed.
|
|
1268
|
+
let sliceCount = 0;
|
|
1269
|
+
for await (const chunk of iterateContentChunks(absolute, file.size)) {
|
|
1270
|
+
for (const content of sliceString(chunk.content)) {
|
|
1271
|
+
sliceCount += 1;
|
|
1168
1272
|
records.push({
|
|
1169
1273
|
path,
|
|
1170
1274
|
at,
|
|
1171
1275
|
status: "present",
|
|
1172
1276
|
content,
|
|
1173
|
-
...(encoding ? { encoding } : {}),
|
|
1277
|
+
...(chunk.encoding ? { encoding: chunk.encoding } : {}),
|
|
1174
1278
|
...modeField,
|
|
1175
|
-
slice: {
|
|
1176
|
-
index,
|
|
1177
|
-
total: sliceContents.length,
|
|
1178
|
-
bytes: Buffer.byteLength(content),
|
|
1179
|
-
sha256: sha256Hex(content),
|
|
1180
|
-
},
|
|
1181
1279
|
});
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
if (sliceCount > 1) {
|
|
1283
|
+
records.forEach((record, index) => {
|
|
1284
|
+
record.slice = {
|
|
1285
|
+
index,
|
|
1286
|
+
total: sliceCount,
|
|
1287
|
+
bytes: Buffer.byteLength(record.content ?? ""),
|
|
1288
|
+
sha256: sha256Hex(record.content ?? ""),
|
|
1289
|
+
};
|
|
1182
1290
|
});
|
|
1183
|
-
|
|
1291
|
+
}
|
|
1292
|
+
if (sliceCount === 1) {
|
|
1293
|
+
entry = records.pop()!;
|
|
1294
|
+
} else {
|
|
1295
|
+
await this.appendJournalRecords(state, records);
|
|
1184
1296
|
return;
|
|
1185
1297
|
}
|
|
1186
1298
|
}
|
|
@@ -1199,27 +1311,156 @@ export class WorkspaceCollector {
|
|
|
1199
1311
|
// Consecutive-duplicate suppression: macOS FSEvents can deliver a
|
|
1200
1312
|
// late event for a path that was also captured directly, and an
|
|
1201
1313
|
// identical re-capture adds upload fat without a state change. The
|
|
1202
|
-
// replay result is identical either way.
|
|
1314
|
+
// replay result is identical either way. The last signature per
|
|
1315
|
+
// path lives in a tiny in-memory map (path strings only — content
|
|
1316
|
+
// is already on disk in the journal).
|
|
1203
1317
|
const journalSignature = (record: ChangeJournalEntry) =>
|
|
1204
1318
|
JSON.stringify([record.status, record.content, record.encoding, record.mode]);
|
|
1205
1319
|
const pendingRecords = [entry, ...records];
|
|
1320
|
+
const accepted: ChangeJournalEntry[] = [];
|
|
1206
1321
|
for (const record of pendingRecords) {
|
|
1207
|
-
|
|
1208
|
-
|
|
1209
|
-
|
|
1210
|
-
|
|
1211
|
-
|
|
1212
|
-
|
|
1322
|
+
const signature = journalSignature(record);
|
|
1323
|
+
if (state.journalLastByPath.get(record.path) === signature) continue;
|
|
1324
|
+
state.journalLastByPath.set(record.path, signature);
|
|
1325
|
+
accepted.push(record);
|
|
1326
|
+
}
|
|
1327
|
+
await this.appendJournalRecords(state, accepted);
|
|
1328
|
+
}
|
|
1329
|
+
|
|
1330
|
+
/**
|
|
1331
|
+
* APPEND records to the on-disk journal (ndjson — one JSON record per
|
|
1332
|
+
* line, 0600). Returns immediately when the capture hit the soft disk
|
|
1333
|
+
* cap (a backend down for a very long time must stop CAPTURING loudly
|
|
1334
|
+
* rather than fill the disk or RAM — the end snapshot still ships the
|
|
1335
|
+
* final tree). Serialized through the per-session append chain.
|
|
1336
|
+
*/
|
|
1337
|
+
private appendJournalRecords(state: SessionState, records: ChangeJournalEntry[]): Promise<void> {
|
|
1338
|
+
if (!state.journalPath || records.length === 0) return Promise.resolve();
|
|
1339
|
+
if (state.journalCaptureStopped) return Promise.resolve();
|
|
1340
|
+
const unacked = state.journalAppendedBytes - state.journalAckOffset;
|
|
1341
|
+
const capMb = Number(process.env.OMNIRUSH_JOURNAL_DISK_CAP_MB);
|
|
1342
|
+
const capBytes = Number.isFinite(capMb) && capMb > 0 ? capMb * 1024 * 1024 : JOURNAL_DISK_CAP_BYTES;
|
|
1343
|
+
if (unacked > capBytes) {
|
|
1344
|
+
state.journalCaptureStopped = true;
|
|
1345
|
+
this.log("warn", "OmniRush change journal hit the disk cap — change capture stopped for this session", {
|
|
1346
|
+
sessionId: state.id,
|
|
1347
|
+
capBytes: capBytes,
|
|
1348
|
+
unackedBytes: unacked,
|
|
1349
|
+
});
|
|
1350
|
+
state.trace.push({
|
|
1351
|
+
at: new Date().toISOString(),
|
|
1352
|
+
type: "journal.capture_stopped",
|
|
1353
|
+
data: { cap_bytes: capBytes, unacked_bytes: unacked },
|
|
1354
|
+
});
|
|
1355
|
+
return Promise.resolve();
|
|
1356
|
+
}
|
|
1357
|
+
const lines = records.map((record) => JSON.stringify(record) + "\n").join("");
|
|
1358
|
+
const bytes = Buffer.byteLength(lines);
|
|
1359
|
+
state.journalAppendTail = state.journalAppendTail
|
|
1360
|
+
.catch(() => undefined)
|
|
1361
|
+
.then(async () => {
|
|
1362
|
+
const { appendFile, mkdir } = await import("node:fs/promises");
|
|
1363
|
+
await mkdir(dirname(state.journalPath!), { recursive: true, mode: 0o700 });
|
|
1364
|
+
await appendFile(state.journalPath!, lines, { encoding: "utf8", mode: 0o600 });
|
|
1365
|
+
state.journalAppendedBytes += bytes;
|
|
1366
|
+
});
|
|
1367
|
+
return state.journalAppendTail;
|
|
1368
|
+
}
|
|
1369
|
+
|
|
1370
|
+
/**
|
|
1371
|
+
* Stream the journal's unacked records (from journalAckOffset to the
|
|
1372
|
+
* upload-start EOF) as parsed entries, each tagged with its raw line
|
|
1373
|
+
* bytes via a NON-enumerable property so the changes.json sidecar
|
|
1374
|
+
* serialization never sees it. Records appended while the upload runs
|
|
1375
|
+
* land beyond the read window and simply ship with the next snapshot.
|
|
1376
|
+
*/
|
|
1377
|
+
private async *readJournalRecords(state: SessionState): AsyncGenerator<ChangeJournalEntry> {
|
|
1378
|
+
if (!state.journalPath || state.journalAckOffset >= state.journalUploadEof) return;
|
|
1379
|
+
const { createReadStream } = await import("node:fs");
|
|
1380
|
+
const readline = await import("node:readline");
|
|
1381
|
+
const stream = createReadStream(state.journalPath, {
|
|
1382
|
+
start: state.journalAckOffset,
|
|
1383
|
+
end: state.journalUploadEof! - 1,
|
|
1384
|
+
encoding: "utf8",
|
|
1385
|
+
});
|
|
1386
|
+
const rl = readline.createInterface({ input: stream, crlfDelay: Infinity });
|
|
1387
|
+
for await (const line of rl) {
|
|
1388
|
+
if (!line.trim()) continue;
|
|
1389
|
+
try {
|
|
1390
|
+
const record = JSON.parse(line) as ChangeJournalEntry;
|
|
1391
|
+
Object.defineProperty(record, RAW_BYTES_FIELD, {
|
|
1392
|
+
value: Buffer.byteLength(line) + 1,
|
|
1393
|
+
enumerable: false,
|
|
1394
|
+
configurable: true,
|
|
1395
|
+
});
|
|
1396
|
+
yield record;
|
|
1397
|
+
} catch {
|
|
1398
|
+
this.log("warn", "OmniRush journal line was unreadable — skipped", { sessionId: state.id });
|
|
1213
1399
|
}
|
|
1214
|
-
if (!duplicate) state.changeJournal.push(record);
|
|
1215
1400
|
}
|
|
1216
1401
|
}
|
|
1217
1402
|
|
|
1218
|
-
/**
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1403
|
+
/**
|
|
1404
|
+
* Upload the journal's unacked records as change snapshots, then
|
|
1405
|
+
* compact: everything acked is removed from the file; records appended
|
|
1406
|
+
* during the upload (past journalUploadEof) survive untouched.
|
|
1407
|
+
*/
|
|
1408
|
+
private async uploadChangeJournal(state: SessionState): Promise<void> {
|
|
1409
|
+
if (!state.journalPath) return;
|
|
1410
|
+
let fileSize = 0;
|
|
1411
|
+
try {
|
|
1412
|
+
fileSize = (await lstat(state.journalPath)).size;
|
|
1413
|
+
} catch {
|
|
1414
|
+
return; // no journal yet — nothing captured
|
|
1415
|
+
}
|
|
1416
|
+
if (fileSize <= state.journalAckOffset) return;
|
|
1417
|
+
const ackStart = state.journalAckOffset;
|
|
1418
|
+
state.journalUploadEof = fileSize;
|
|
1419
|
+
try {
|
|
1420
|
+
await this.uploadChangeParts(state, this.readJournalRecords(state), {
|
|
1421
|
+
onAck: (batch) => {
|
|
1422
|
+
for (const record of batch) {
|
|
1423
|
+
const raw = (record as Record<symbol, unknown>)[RAW_BYTES_FIELD];
|
|
1424
|
+
if (typeof raw === "number") state.journalAckOffset += raw;
|
|
1425
|
+
}
|
|
1426
|
+
},
|
|
1427
|
+
});
|
|
1428
|
+
} finally {
|
|
1429
|
+
state.journalUploadEof = null;
|
|
1430
|
+
}
|
|
1431
|
+
if (state.journalAckOffset <= ackStart) return; // nothing acked — keep everything
|
|
1432
|
+
await this.compactJournal(state, state.journalAckOffset);
|
|
1433
|
+
state.journalAckOffset = 0;
|
|
1434
|
+
}
|
|
1435
|
+
|
|
1436
|
+
/** Rewrite the journal without its first `dropBytes` bytes. */
|
|
1437
|
+
private async compactJournal(state: SessionState, dropBytes: number): Promise<void> {
|
|
1438
|
+
if (!state.journalPath) return;
|
|
1439
|
+
const { createReadStream, createWriteStream } = await import("node:fs");
|
|
1440
|
+
const { rename } = await import("node:fs/promises");
|
|
1441
|
+
const tmpPath = `${state.journalPath}.compact`;
|
|
1442
|
+
await new Promise<void>((resolvePromise, rejectPromise) => {
|
|
1443
|
+
const input = createReadStream(state.journalPath!, { start: dropBytes, encoding: "utf8" });
|
|
1444
|
+
const output = createWriteStream(tmpPath, { encoding: "utf8", mode: 0o600 });
|
|
1445
|
+
input.on("error", rejectPromise);
|
|
1446
|
+
output.on("error", rejectPromise);
|
|
1447
|
+
output.on("finish", () => resolvePromise());
|
|
1448
|
+
input.pipe(output);
|
|
1449
|
+
});
|
|
1450
|
+
await rename(tmpPath, state.journalPath);
|
|
1451
|
+
}
|
|
1452
|
+
|
|
1453
|
+
/** Upload unacked journal records (best effort) and remove the file. */
|
|
1454
|
+
private async teardownJournal(state: SessionState): Promise<void> {
|
|
1455
|
+
try {
|
|
1456
|
+
await this.uploadChangeJournal(state);
|
|
1457
|
+
} catch {
|
|
1458
|
+
/* best effort — the end snapshot still ships the final tree */
|
|
1459
|
+
}
|
|
1460
|
+
if (state.journalPath) {
|
|
1461
|
+
const { rm } = await import("node:fs/promises");
|
|
1462
|
+
await rm(state.journalPath, { force: true }).catch(() => undefined);
|
|
1463
|
+
}
|
|
1223
1464
|
}
|
|
1224
1465
|
|
|
1225
1466
|
private enqueue(state: SessionState, operation: () => Promise<void>): void {
|
|
@@ -1234,17 +1475,10 @@ export class WorkspaceCollector {
|
|
|
1234
1475
|
private async uploadWorkspace(state: SessionState, type: "start" | "change" | "end"): Promise<void> {
|
|
1235
1476
|
if (state.budgetExhausted) return;
|
|
1236
1477
|
if (type === "change") {
|
|
1237
|
-
// Change snapshots carry ONLY the touched files
|
|
1238
|
-
//
|
|
1239
|
-
|
|
1240
|
-
|
|
1241
|
-
state.lastSignature = await workspaceSignature(state.root);
|
|
1242
|
-
return;
|
|
1243
|
-
}
|
|
1244
|
-
const uploaded = await this.uploadChangeParts(state, snapshot);
|
|
1245
|
-
// Acknowledge only after every part of the snapshot succeeded; on a
|
|
1246
|
-
// mid-budget exhaustion, acknowledge only what actually shipped.
|
|
1247
|
-
this.acknowledgeJournal(state, state.budgetExhausted ? uploaded.flat() : snapshot);
|
|
1478
|
+
// Change snapshots carry ONLY the touched files — streamed from
|
|
1479
|
+
// the on-disk journal, acked per part, then compacted. Never a
|
|
1480
|
+
// full re-enumeration of the tree, never held in memory.
|
|
1481
|
+
await this.uploadChangeJournal(state);
|
|
1248
1482
|
} else {
|
|
1249
1483
|
await this.uploadTreeParts(state, type);
|
|
1250
1484
|
}
|
|
@@ -1302,7 +1536,7 @@ export class WorkspaceCollector {
|
|
|
1302
1536
|
private async *iterTreeFiles(state: SessionState, paths: string[]): AsyncGenerator<CollectorFile> {
|
|
1303
1537
|
for (const path of paths) {
|
|
1304
1538
|
if (state.budgetExhausted) return;
|
|
1305
|
-
for (const entry of
|
|
1539
|
+
for await (const entry of iterateWorkspaceEntries(state.root, path)) {
|
|
1306
1540
|
yield entry;
|
|
1307
1541
|
}
|
|
1308
1542
|
}
|
|
@@ -1315,7 +1549,11 @@ export class WorkspaceCollector {
|
|
|
1315
1549
|
* records — replaying start + sidecars in sequence order reproduces the
|
|
1316
1550
|
* end tree.
|
|
1317
1551
|
*/
|
|
1318
|
-
private async uploadChangeParts(
|
|
1552
|
+
private async uploadChangeParts(
|
|
1553
|
+
state: SessionState,
|
|
1554
|
+
units: Iterable<ChangeJournalEntry> | AsyncIterable<ChangeJournalEntry>,
|
|
1555
|
+
opts: { onAck?: (batch: ChangeJournalEntry[]) => void } = {},
|
|
1556
|
+
): Promise<ChangeJournalEntry[][]> {
|
|
1319
1557
|
const uploadedSlices: ChangeJournalEntry[][] = [];
|
|
1320
1558
|
const filesForSlice = (entries: ChangeJournalEntry[]): CollectorFile[] => {
|
|
1321
1559
|
const lastByPath = new Map<string, CollectorFile>();
|
|
@@ -1372,7 +1610,7 @@ export class WorkspaceCollector {
|
|
|
1372
1610
|
return files;
|
|
1373
1611
|
};
|
|
1374
1612
|
const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
|
|
1375
|
-
await this.packAndUpload(state, "change",
|
|
1613
|
+
await this.packAndUpload(state, "change", units, {
|
|
1376
1614
|
sizeOf: (entry) =>
|
|
1377
1615
|
Buffer.byteLength(entry.content ?? "") + entry.path.length + 96,
|
|
1378
1616
|
measure: async (batch) => {
|
|
@@ -1382,7 +1620,10 @@ export class WorkspaceCollector {
|
|
|
1382
1620
|
},
|
|
1383
1621
|
upload: async (measured, batch) => {
|
|
1384
1622
|
const ok = await this.uploadEnvelope(state, "change", measured.files, measured);
|
|
1385
|
-
if (ok)
|
|
1623
|
+
if (ok) {
|
|
1624
|
+
uploadedSlices.push(batch);
|
|
1625
|
+
opts.onAck?.(batch);
|
|
1626
|
+
}
|
|
1386
1627
|
return ok;
|
|
1387
1628
|
},
|
|
1388
1629
|
drop: (entry) => {
|
|
@@ -1579,14 +1820,46 @@ export class WorkspaceCollector {
|
|
|
1579
1820
|
body: compressed.buffer.slice(compressed.byteOffset, compressed.byteOffset + compressed.byteLength) as ArrayBuffer,
|
|
1580
1821
|
signal: AbortSignal.timeout(120_000),
|
|
1581
1822
|
});
|
|
1582
|
-
|
|
1583
|
-
|
|
1584
|
-
|
|
1585
|
-
|
|
1586
|
-
|
|
1587
|
-
|
|
1588
|
-
response
|
|
1589
|
-
|
|
1823
|
+
// Shared retry layer: 429/5xx/network ride out backend deploys with
|
|
1824
|
+
// exponential backoff + jitter. 401 stays a protocol answer handled
|
|
1825
|
+
// inline (single-flight refresh, retry exactly once).
|
|
1826
|
+
const outcome = await withRetries(
|
|
1827
|
+
async () => {
|
|
1828
|
+
let response = await send();
|
|
1829
|
+
if (response.status === 401 && this.refresh) {
|
|
1830
|
+
// Broker pattern: single-flight refresh, retry exactly once.
|
|
1831
|
+
const rotated = await this.refresh(this.token).catch(() => null);
|
|
1832
|
+
if (rotated) {
|
|
1833
|
+
this.token = rotated;
|
|
1834
|
+
response = await send();
|
|
1835
|
+
}
|
|
1836
|
+
}
|
|
1837
|
+
return {
|
|
1838
|
+
response,
|
|
1839
|
+
retryAfterSec: Number(response?.headers?.get?.("retry-after")) || undefined,
|
|
1840
|
+
};
|
|
1841
|
+
},
|
|
1842
|
+
{
|
|
1843
|
+
attempts: retryAttempts(),
|
|
1844
|
+
isRetryable: ({ response }) => isRetryableStatus(response?.status),
|
|
1845
|
+
onRetry: ({ attempt, attempts, delayMs }) => {
|
|
1846
|
+
this.log("warn", "OmniRush upload unavailable — retrying", {
|
|
1847
|
+
sessionId: state.id,
|
|
1848
|
+
snapshotType,
|
|
1849
|
+
attempt: `${attempt + 1}/${attempts}`,
|
|
1850
|
+
delayMs,
|
|
1851
|
+
});
|
|
1852
|
+
},
|
|
1853
|
+
},
|
|
1854
|
+
);
|
|
1855
|
+
const response = outcome.result?.response;
|
|
1856
|
+
if (!response) {
|
|
1857
|
+
// Every attempt failed at the network layer.
|
|
1858
|
+
throw new Error(
|
|
1859
|
+
`collector upload failed: network error after ${outcome.attempts} attempt${outcome.attempts === 1 ? "" : "s"}` +
|
|
1860
|
+
` — ${outcome.error?.message ?? "unknown"} [ref ${outcome.ref}]` +
|
|
1861
|
+
` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
|
|
1862
|
+
);
|
|
1590
1863
|
}
|
|
1591
1864
|
if (!response.ok) {
|
|
1592
1865
|
// Diagnosable from a tester's screenshot alone: status, context,
|
|
@@ -1596,6 +1869,7 @@ export class WorkspaceCollector {
|
|
|
1596
1869
|
throw new Error(
|
|
1597
1870
|
`collector upload failed: HTTP ${response.status}` +
|
|
1598
1871
|
`${snippet ? ` — ${snippet}` : ""}` +
|
|
1872
|
+
` [ref ${outcome.ref}]` +
|
|
1599
1873
|
` (session ${state.id}, ${snapshotType}, sequence ${state.sequence + 1})`,
|
|
1600
1874
|
);
|
|
1601
1875
|
}
|