omnirush 0.2.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,14 +1,42 @@
1
1
  // Omnirush workspace collector — CLI port of the GUI reference
2
- // implementation (gui-reference/apps/server/src/workspace-collector.ts,
3
- // 772 lines). Ported faithfully: watch + debounce + change journal +
4
- // session ledger + caps + redaction. Deviations, both forced by the CLI
5
- // runtime:
6
- // 1. minimatch is replaced by an internal gitignore-style matcher
7
- // (`ignoreMatch`) — pi extensions are loaded through jiti with a
8
- // fixed virtual-module set and cannot import npm packages.
9
- // 2. Uploads go straight to the manager /collect endpoint with the
10
- // device access token; an optional `refresh` callback gives the
11
- // broker's on-401 single-flight refresh + retry-once behavior.
2
+ // implementation (gui-reference/apps/server/src/workspace-collector.ts),
3
+ // rebuilt 2026-09-23 around the "reproducible environment" spec:
4
+ //
5
+ // 1. Whole project BEFORE — full-tree "start" snapshot at session start.
6
+ // 2. Replayable change log — ordered per-file journal records
7
+ // (at, path, status, content) uploaded as change snapshots; replaying
8
+ // start + the ordered records reproduces the end state.
9
+ // 3. Whole project AFTER — full-tree "end" snapshot at session close.
10
+ // 4. Fat traces — no truncation of trace events (trace_truncated only at
11
+ // the session budget wall).
12
+ // 5. No file-count / journal caps. The only limit is the 100 GiB
13
+ // per-session budget (collection stops, loudly, never silent mid-file
14
+ // truncation). The backend's original 4 MiB per-file limit was
15
+ // removed server-side, so big files SHIP: content is sliced into
16
+ // <= 8 MiB entries (`<path>.__agent_part<i>`) with a
17
+ // `<path>.__agent_manifest.json` record (slice count, per-slice
18
+ // sha256) for deterministic reassembly. Parts target <= 15 MiB
19
+ // compressed (body rail live-verified >= 31.5 MiB), are ordered by
20
+ // sequence, and file content is never split across parts.
21
+ // 6. FULL-WALK enumeration: every file in the workspace tree ships —
22
+ // node_modules, build outputs, .git (full reproducibility) — with
23
+ // ONLY the secrets denylist (.env*, keys, credentials, private
24
+ // keys) excluded. Binaries are captured (base64), symlinks are
25
+ // stored as records (path + target) so replay recreates them, and
26
+ // the POSIX mode/executable bit rides each file entry for replay
27
+ // (stored-and-ignored on Windows). Redaction also scrubs
28
+ // credentials embedded in URLs (e.g. .git/config remotes).
29
+ // In-workspace journal files and stored trace/schema ids carry
30
+ // generic "agent" branding: __agent__/, agent.trace.v1.
31
+ //
32
+ // Deviations from the original port, forced by the CLI runtime:
33
+ // - The internal gitignore-style matcher was REMOVED together with
34
+ // .gitignore honoring: build outputs and node_modules are typically
35
+ // gitignored, so the walker no longer applies ignore rules at all
36
+ // (the secrets denylist still applies).
37
+ // - Uploads go straight to the manager /collect endpoint with the
38
+ // device access token; an optional `refresh` callback gives the
39
+ // broker's on-401 single-flight refresh + retry-once behavior.
12
40
  //
13
41
  // Runs under the bundled bun (which provides node:zlib zstdCompress;
14
42
  // plain Node needs >= 22.15). All helpers used by node:test are pure or
@@ -17,23 +45,59 @@
17
45
  import { createHash } from "node:crypto";
18
46
  import { execFile } from "node:child_process";
19
47
  import { watch, type FSWatcher } from "node:fs";
20
- import { lstat, mkdir, readFile, readdir, writeFile } from "node:fs/promises";
48
+ import { lstat, mkdir, readFile, readdir, readlink, writeFile } from "node:fs/promises";
21
49
  import { dirname, join, relative, resolve, sep } from "node:path";
22
50
  import { promisify } from "node:util";
23
51
  import { zstdCompress as zstdCompressCb } from "node:zlib";
24
52
 
25
53
  const execFileAsync = promisify(execFile);
26
54
 
27
- export const MAX_COLLECTOR_FILE_BYTES = 1024 * 1024;
28
- export const MAX_COLLECTOR_SESSION_BYTES = 64 * 1024 * 1024;
29
- export const MAX_SNAPSHOT_BYTES = 20 * 1024 * 1024;
30
- export const MAX_TRACE_BYTES = 4 * 1024 * 1024;
31
- export const MAX_FILES = 20_000;
32
- export const MAX_COMPRESSED_BYTES = 16 * 1024 * 1024;
55
+ // --- limits ---------------------------------------------------------------
56
+ // Session budget: collection stops (with a warning) once this much
57
+ // uncompressed payload has been sent for a session. There are no other
58
+ // policy caps — no file-count cap, no journal cap.
59
+ export const MAX_SESSION_BYTES = 100 * 1024 * 1024 * 1024;
60
+ // Per-content-entry slice size. The backend's original 4 MiB per-file
61
+ // limit was REMOVED server-side (2026-09-23), so big files no longer get
62
+ // skipped: their content is sliced into <= MAX_SLICE_BYTES entries with
63
+ // sequential virtual paths (`<path>.__agent_part<i>`) plus a
64
+ // `<path>.__agent_manifest.json` record (path, slice count, per-slice
65
+ // byte size + sha256) so replay reassembles the original deterministically.
66
+ export const MAX_SLICE_BYTES = 4 * 1024 * 1024;
67
+ // Whole-read ceiling. Files up to this size are read and processed as
68
+ // one string (worst case: binary -> base64 inflates 4/3 to ~427 MiB
69
+ // chars, comfortably under the runtime's ~512 MiB MAX_STRING_LENGTH).
70
+ // Bigger files are streamed in fixed raw chunks and sliced.
71
+ const WHOLE_READ_BYTES = 320 * 1024 * 1024;
72
+ // Multi-file parts target this compressed size (the backend body rail
73
+ // was live-verified at >= 31.5 MiB compressed, so 15 MiB leaves ample
74
+ // headroom for any ingress in front of the manager).
75
+ export const MAX_PART_COMPRESSED_BYTES = 15 * 1024 * 1024;
76
+ // Absolute rail for any single request body (client-side backstop well
77
+ // above the validated 31.5 MiB; single-file parts may use it).
78
+ export const MAX_PART_COMPRESSED_HARD = 32 * 1024 * 1024;
79
+ // Accumulate this many uncompressed bytes before a compress-and-check.
80
+ export const PART_UNCOMPRESSED_STEP = 8 * 1024 * 1024;
81
+ // Flush a part when a check shows at least this much compressed payload.
82
+ export const PART_READY_COMPRESSED_BYTES = 12 * 1024 * 1024;
83
+ // Absolute uncompressed ceiling for one part batch — bounds resident
84
+ // memory even for highly-compressible content.
85
+ export const PART_BATCH_UNCOMPRESSED_MAX = 48 * 1024 * 1024;
86
+ // Change/trace parts flush by PLAIN size so their __agent__/changes.json
87
+ // / trace.json entries (single file entries whose content grows with the
88
+ // batch) stay small and parts upload promptly.
89
+ export const PART_SIDECAR_PLAIN_MAX = 1024 * 1024;
90
+ // Soft in-memory relief for trace events (they are uploaded, never
91
+ // dropped — this only bounds resident memory between flushes).
92
+ export const TRACE_RELIEF_EVENTS = 2_000;
93
+
94
+ // Generic in-workspace journal directory + stored schema id (branding
95
+ // neutral — the traces are the product).
96
+ export const AGENT_DIR = "__agent__";
97
+ export const AGENT_TRACE_SCHEMA = "agent.trace.v1";
98
+
33
99
  export const CHANGE_DEBOUNCE_MS = 2_000;
34
100
  export const FALLBACK_SCAN_MS = 10_000;
35
- export const MAX_CHANGE_JOURNAL_ENTRIES = 512;
36
- export const MAX_CHANGE_JOURNAL_BYTES = 768 * 1024;
37
101
  export const MAX_SESSION_LEDGER_ENTRIES = 512;
38
102
  export const SESSION_LEDGER_FILE = "omnirush-collector-sessions.json";
39
103
 
@@ -41,7 +105,15 @@ type SnapshotType = "start" | "change" | "trace" | "end";
41
105
 
42
106
  export type CollectorFile = {
43
107
  path: string;
44
- content: string;
108
+ /** utf8 text (default) or base64 when encoding is "base64". Absent for
109
+ * symlink records, which carry `target` instead. */
110
+ content?: string;
111
+ /** Absent = utf8 text. "base64" for binary content. */
112
+ encoding?: "base64";
113
+ /** Symlink record: the link target as stored on disk. */
114
+ target?: string;
115
+ /** POSIX mode bits (octal string, e.g. "755"). Absent for symlinks. */
116
+ mode?: string;
45
117
  };
46
118
 
47
119
  type TraceEvent = {
@@ -51,10 +123,15 @@ type TraceEvent = {
51
123
  };
52
124
 
53
125
  export type ChangeJournalEntry = {
54
- path: string;
55
126
  at: string;
56
- status: "present" | "deleted" | "skipped";
127
+ path: string;
128
+ status: "present" | "deleted";
57
129
  content?: string;
130
+ encoding?: "base64";
131
+ target?: string;
132
+ mode?: string;
133
+ /** Present when the record carries one slice of a bigger file. */
134
+ slice?: { index: number; total: number; bytes: number; sha256: string };
58
135
  };
59
136
 
60
137
  export type SessionLedgerRecord = {
@@ -82,12 +159,24 @@ type SessionState = {
82
159
  lastSignature: string;
83
160
  started: boolean;
84
161
  finished: boolean;
162
+ budgetExhausted: boolean;
85
163
  changeTimer: ReturnType<typeof setTimeout> | null;
86
164
  scanTimer: ReturnType<typeof setInterval> | null;
87
165
  watcher: FSWatcher | null;
166
+ /** Wall-clock instant the fs.watch baseline begins. Present-records
167
+ * whose mtime is not after this are pre-baseline activity already
168
+ * reflected in the start snapshot (macOS FSEvents delivers events for
169
+ * files created just before watch() began — inotify does not). */
170
+ watcherStartedAtMs: number;
171
+ /** Paths the session knows exist(ed): the baseline walk plus every
172
+ * captured present record. Deletion records require membership —
173
+ * macOS FSEvents leaks parent-directory events (e.g. the previous
174
+ * tmpdir's removal) into a freshly created watch, which must not
175
+ * fabricate deletions inside the workspace. */
176
+ knownPaths: Set<string>;
88
177
  trace: TraceEvent[];
89
- changeJournal: Map<string, ChangeJournalEntry>;
90
- changeJournalBytes: number;
178
+ /** Ordered change records — the replayable change log. */
179
+ changeJournal: ChangeJournalEntry[];
91
180
  changeCaptureTail: Promise<void>;
92
181
  ready: Promise<void>;
93
182
  tail: Promise<void>;
@@ -104,12 +193,20 @@ export type CollectorOptions = {
104
193
  log?: (level: "info" | "warn", message: string, attributes?: Record<string, unknown>) => void;
105
194
  changeDebounceMs?: number;
106
195
  fallbackScanMs?: number;
196
+ /** Per-session collection budget. Default MAX_SESSION_BYTES (100 GiB). */
197
+ sessionBudgetBytes?: number;
198
+ /** Multi-file part compressed target. Default MAX_PART_COMPRESSED_BYTES. */
199
+ partCompressedBytes?: number;
200
+ /** Uncompressed accumulation step between compress checks. */
201
+ partCheckBytes?: number;
107
202
  };
108
203
 
204
+ // .git is deliberately NOT denied: full reproducibility includes the
205
+ // repository metadata. Credentials inside it (e.g. remote URLs with
206
+ // embedded passwords in .git/config) are covered by the URL redaction
207
+ // pattern below and the credential-name denylist.
109
208
  const DENIED_EXACT_NAMES = new Set([
110
- ".git",
111
209
  ".ssh",
112
- "node_modules",
113
210
  "keys",
114
211
  "secrets",
115
212
  ".npmrc",
@@ -127,6 +224,9 @@ const SECRET_PATTERNS: Array<[RegExp, string]> = [
127
224
  [/\bAKIA[0-9A-Z]{16}\b/g, "[REDACTED]"],
128
225
  [/\b(?:sk|rk|pk)-(?:proj-)?[A-Za-z0-9_-]{16,}\b/g, "[REDACTED]"],
129
226
  [/^([A-Z][A-Z0-9_]*(?:TOKEN|SECRET|PASSWORD|API_KEY)\s*=\s*)([^\s#]{6,})$/gim, "$1[REDACTED]"],
227
+ // Credentials embedded in URLs (git remotes in .git/config, pip
228
+ // indexes, ...): https://user:password@host -> https://user:[REDACTED]@host
229
+ [/\b(https?:\/\/[^\s\/@:#]+:)([^\s\/@]+)(@)/gi, "$1[REDACTED]$3"],
130
230
  ];
131
231
  const PII_PATTERNS: Array<[RegExp, string]> = [
132
232
  [/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi, "[REDACTED_PII]"],
@@ -197,89 +297,8 @@ function isBinary(buffer: Buffer): boolean {
197
297
  return sample.length > 0 && suspicious / sample.length > 0.1;
198
298
  }
199
299
 
200
- /**
201
- * Minimal gitignore-style glob matcher (the minimatch replacement).
202
- * Supports `*` (within a segment), `**` (across segments, including a
203
- * leading globstar-slash that matches zero directories), `?`, and
204
- * matchBase semantics: a pattern without `/` matches the file name in
205
- * any directory. Always dot-matches (gitignore semantics for `*`).
206
- */
207
- export function ignoreMatch(path: string, rawPattern: string): boolean {
208
- const pattern = rawPattern.replace(/^\//, "").replace(/\/$/, "/**").replace(/\/{2,}/g, "/");
209
- if (!pattern) return false;
210
- const hasSlash = pattern.includes("/");
211
- let re = "^";
212
- let i = 0;
213
- while (i < pattern.length) {
214
- const char = pattern[i];
215
- if (char === "*") {
216
- if (pattern[i + 1] === "*") {
217
- let j = i;
218
- while (pattern[j] === "*") j += 1;
219
- if (pattern[j] === "/" && (i === 0 || pattern[i - 1] === "/")) {
220
- // Leading or interior `**/` — zero or more whole directories.
221
- re += "(?:[^/]+/)*";
222
- i = j + 1;
223
- continue;
224
- }
225
- re += ".*";
226
- i = j;
227
- continue;
228
- }
229
- re += "[^/]*";
230
- i += 1;
231
- continue;
232
- }
233
- if (char === "?") {
234
- re += "[^/]";
235
- i += 1;
236
- continue;
237
- }
238
- re += char.replace(/[.+^${}()|[\]\\]/g, "\\$&");
239
- i += 1;
240
- }
241
- const anchored = `${re}$`;
242
- if (regexMatch(path, anchored)) return true;
243
- // matchBase: pattern without a slash also matches the file name alone.
244
- if (!hasSlash) {
245
- const base = path.split("/").pop() ?? "";
246
- return regexMatch(base, anchored);
247
- }
248
- return false;
249
- }
250
-
251
- function regexMatch(value: string, source: string): boolean {
252
- try {
253
- return new RegExp(source).test(value);
254
- } catch {
255
- return false;
256
- }
257
- }
258
-
259
- function ignoredByRules(path: string, rules: string[]): boolean {
260
- let ignored = false;
261
- for (const raw of rules) {
262
- const negated = raw.startsWith("!");
263
- const body = negated ? raw.slice(1) : raw;
264
- const pattern = body.replace(/^\//, "").replace(/\/$/, "/**");
265
- if (!pattern) continue;
266
- if (ignoreMatch(path, pattern)) ignored = !negated;
267
- }
268
- return ignored;
269
- }
270
-
271
- async function gitIgnoresPath(root: string, path: string): Promise<boolean> {
272
- try {
273
- await execFileAsync("git", ["-C", root, "check-ignore", "--no-index", "-q", "--", path], {
274
- timeout: 5_000,
275
- maxBuffer: 64 * 1024,
276
- });
277
- return true;
278
- } catch {
279
- // Exit status 1 means the path is not ignored. A non-git workspace also
280
- // falls through here and is handled by the normal snapshot walker.
281
- return false;
282
- }
300
+ function comparePaths(left: string, right: string): number {
301
+ return left < right ? -1 : left > right ? 1 : 0;
283
302
  }
284
303
 
285
304
  export async function readSessionLedger(path: string | null): Promise<SessionLedger> {
@@ -309,10 +328,11 @@ export async function readSessionLedger(path: string | null): Promise<SessionLed
309
328
  export function boundedTracePayload(
310
329
  state: Pick<SessionState, "id" | "workspaceId" | "segment" | "resumed">,
311
330
  events: TraceEvent[],
312
- maxBytes = MAX_TRACE_BYTES,
331
+ maxBytes = MAX_SESSION_BYTES,
313
332
  ): Buffer {
314
333
  const encode = (selected: TraceEvent[], truncated: boolean, includedCount = selected.length) => Buffer.from(redactCollectorText(JSON.stringify({
315
334
  schema_version: 1,
335
+ schema: AGENT_TRACE_SCHEMA,
316
336
  session_id: state.id,
317
337
  workspace_id: state.workspaceId,
318
338
  session_segment: state.segment,
@@ -345,26 +365,39 @@ export function boundedTracePayload(
345
365
  return payload;
346
366
  }
347
367
 
348
- async function readGitignoreFile(directory: string): Promise<string[]> {
349
- try {
350
- return (await readFile(resolve(directory, ".gitignore"), "utf8"))
351
- .split(/\r?\n/)
352
- .map((line) => line.trim())
353
- .filter((line) => line && !line.startsWith("#"));
354
- } catch {
355
- return [];
368
+ /**
369
+ * Select the longest prefix of events whose JSON fits maxBytes — the
370
+ * budget-wall behavior for traces. With the default 100 GiB budget this
371
+ * only fires at the wall; traces are otherwise never truncated.
372
+ */
373
+ export function selectTraceEvents(
374
+ events: TraceEvent[],
375
+ maxBytes: number,
376
+ ): { events: TraceEvent[]; truncated: boolean; droppedCount: number } {
377
+ if (events.length === 0) return { events: [], truncated: false, droppedCount: 0 };
378
+ const fits = (candidate: TraceEvent[]) => Buffer.byteLength(JSON.stringify(candidate)) <= maxBytes;
379
+ if (fits(events)) return { events, truncated: false, droppedCount: 0 };
380
+ const selected: TraceEvent[] = [];
381
+ for (const event of events) {
382
+ if (!fits([...selected, event])) break;
383
+ selected.push(event);
356
384
  }
385
+ return { events: selected, truncated: true, droppedCount: events.length - selected.length };
357
386
  }
358
387
 
359
- async function walkFallback(root: string, directory = root, rules: string[] = [], output: string[] = []): Promise<string[]> {
360
- const localRules = await readGitignoreFile(directory);
361
- const prefix = portablePath(root, directory);
362
- const ignoreRules = [...rules, ...localRules.map((rule) => {
363
- const negated = rule.startsWith("!");
364
- const body = negated ? rule.slice(1) : rule;
365
- const scoped = prefix ? `${prefix}/${body}` : body;
366
- return negated ? `!${scoped}` : scoped;
367
- })];
388
+ /**
389
+ * FULL-WALK enumeration: every path in the workspace tree — regular
390
+ * files, symlinks (including links to directories; links are recorded,
391
+ * never followed, so cycles are impossible), and .git internals. The
392
+ * ONLY exclusions are the secrets-denylist paths. Git-aware listing was
393
+ * removed on purpose: `git ls-files` drops anything .gitignored (build
394
+ * outputs, node_modules), which broke the whole-project guarantee.
395
+ */
396
+ async function listWorkspaceFiles(root: string): Promise<string[]> {
397
+ return walkWorkspace(root);
398
+ }
399
+
400
+ async function walkWorkspace(root: string, directory = root, output: string[] = []): Promise<string[]> {
368
401
  let entries;
369
402
  try {
370
403
  entries = await readdir(directory, { withFileTypes: true });
@@ -372,33 +405,15 @@ async function walkFallback(root: string, directory = root, rules: string[] = []
372
405
  return output;
373
406
  }
374
407
  for (const entry of entries) {
375
- if (output.length >= MAX_FILES) break;
376
408
  const fullPath = resolve(directory, entry.name);
377
409
  const path = portablePath(root, fullPath);
378
- if (!path || isCollectorPathDenied(path) || ignoredByRules(path, ignoreRules)) continue;
379
- if (entry.isDirectory()) await walkFallback(root, fullPath, ignoreRules, output);
380
- else if (entry.isFile()) output.push(path);
410
+ if (!path || isCollectorPathDenied(path)) continue;
411
+ if (entry.isDirectory()) await walkWorkspace(root, fullPath, output);
412
+ else output.push(path); // regular files AND symlinks (files or dirs)
381
413
  }
382
414
  return output;
383
415
  }
384
416
 
385
- async function listWorkspaceFiles(root: string): Promise<string[]> {
386
- try {
387
- const { stdout } = await execFileAsync("git", ["-C", root, "ls-files", "-co", "--exclude-standard", "-z"], {
388
- encoding: "buffer",
389
- maxBuffer: 16 * 1024 * 1024,
390
- timeout: 15_000,
391
- });
392
- return Buffer.from(stdout)
393
- .toString("utf8")
394
- .split("\0")
395
- .filter((path) => path && !isCollectorPathDenied(path))
396
- .slice(0, MAX_FILES);
397
- } catch {
398
- return walkFallback(root);
399
- }
400
- }
401
-
402
417
  async function gitMetadata(root: string): Promise<Record<string, string | null>> {
403
418
  const git = async (...args: string[]) => {
404
419
  try {
@@ -421,8 +436,11 @@ async function workspaceSignature(root: string): Promise<string> {
421
436
  for (const path of await listWorkspaceFiles(root)) {
422
437
  try {
423
438
  const file = await lstat(resolve(root, path));
424
- if (!file.isFile() || file.isSymbolicLink() || file.size > MAX_COLLECTOR_FILE_BYTES) continue;
425
- hash.update(path).update("\0").update(String(file.size)).update("\0").update(String(file.mtimeMs)).update("\0");
439
+ if (file.isFile()) {
440
+ hash.update(path).update("\0").update(String(file.size)).update("\0").update(String(file.mtimeMs)).update("\0");
441
+ } else if (file.isSymbolicLink()) {
442
+ hash.update(path).update("\0").update("link").update("\0").update(String(file.mtimeMs)).update("\0");
443
+ }
426
444
  } catch {
427
445
  // A file can disappear while the workspace is being scanned.
428
446
  }
@@ -430,6 +448,205 @@ async function workspaceSignature(root: string): Promise<string> {
430
448
  return hash.digest("hex");
431
449
  }
432
450
 
451
+ function sha256Hex(content: string): string {
452
+ return createHash("sha256").update(content).digest("hex");
453
+ }
454
+
455
+ /** Split a string into <= maxUnits slices without splitting surrogates. */
456
+ export function sliceString(content: string, maxUnits = MAX_SLICE_BYTES): string[] {
457
+ if (content.length <= maxUnits) return [content];
458
+ const slices: string[] = [];
459
+ let start = 0;
460
+ while (start < content.length) {
461
+ let end = Math.min(start + maxUnits, content.length);
462
+ if (end < content.length) {
463
+ const code = content.charCodeAt(end - 1);
464
+ if (code >= 0xd800 && code <= 0xdbff) end -= 1;
465
+ }
466
+ slices.push(content.slice(start, end));
467
+ start = end;
468
+ }
469
+ return slices;
470
+ }
471
+
472
+ /** Slice entry virtual path + manifest path for a logical file path. */
473
+ export function slicePartPath(path: string, index: number): string {
474
+ return `${path}.__agent_part${index}`;
475
+ }
476
+ export function sliceManifestPath(path: string): string {
477
+ return `${path}.__agent_manifest.json`;
478
+ }
479
+
480
+ /**
481
+ * Slice one logical file's content into <= MAX_SLICE_BYTES entries plus a
482
+ * manifest record (path, slice count, per-slice byte size + sha256) so
483
+ * replay reassembles the original deterministically. Small files pass
484
+ * through unchanged.
485
+ */
486
+ export function buildFileEntries(file: CollectorFile): CollectorFile[] {
487
+ const content = file.content ?? "";
488
+ if (Buffer.byteLength(content) <= MAX_SLICE_BYTES) return [file];
489
+ const slices = sliceString(content);
490
+ const descriptors = slices.map((slice, index) => ({
491
+ index,
492
+ bytes: Buffer.byteLength(slice),
493
+ sha256: sha256Hex(slice),
494
+ }));
495
+ return [
496
+ { path: sliceManifestPath(file.path), content: JSON.stringify(sliceManifest(file, descriptors)) },
497
+ ...slices.map((slice, index) => ({
498
+ path: slicePartPath(file.path, index),
499
+ content: slice,
500
+ ...(file.encoding ? { encoding: file.encoding } : {}),
501
+ })),
502
+ ];
503
+ }
504
+
505
+ /**
506
+ * Manifest for a sliced file. `slice_count` is the GLOBAL count and each
507
+ * entry carries its global `index`, so manifests arriving across
508
+ * different parts union deterministically: a replay collects the part
509
+ * entries by index (verifying per-slice sha256) until all
510
+ * `slice_count` indexes are present, then reassembles.
511
+ */
512
+ function sliceManifest(
513
+ file: { path: string; encoding?: "base64"; mode?: string },
514
+ sliceDescriptors: Array<{ index: number; bytes: number; sha256: string }>,
515
+ totalCount?: number,
516
+ ): Record<string, unknown> {
517
+ return {
518
+ schema_version: 1,
519
+ path: file.path,
520
+ encoding: file.encoding ?? null,
521
+ mode: file.mode ?? null,
522
+ slice_count: totalCount ?? sliceDescriptors.length,
523
+ slices: sliceDescriptors,
524
+ };
525
+ }
526
+
527
+ type ContentChunk = { content: string; encoding?: "base64" };
528
+
529
+ /**
530
+ * Read a regular file into <= ~8 MiB content chunks: small files in one
531
+ * whole read (fully redacted), big files streamed in fixed raw chunks —
532
+ * strings stay far below the runtime string limit regardless of file
533
+ * size. Binary detection happens on the first chunk and applies to the
534
+ * whole file.
535
+ */
536
+ async function readContentChunks(absolute: string, size: number): Promise<ContentChunk[]> {
537
+ if (size <= WHOLE_READ_BYTES) {
538
+ const buffer = await readFile(absolute);
539
+ if (isBinary(buffer)) {
540
+ return [{ content: buffer.toString("base64"), encoding: "base64" }];
541
+ }
542
+ return [{ content: redactCollectorText(buffer.toString("utf8")).text }];
543
+ }
544
+ const { open } = await import("node:fs/promises");
545
+ const handle = await open(absolute, "r");
546
+ try {
547
+ const chunkBytes = 8 * 1024 * 1024;
548
+ const buffer = Buffer.alloc(chunkBytes);
549
+ const first = await handle.read(buffer, 0, chunkBytes, null);
550
+ const binary = isBinary(buffer.subarray(0, first.bytesRead));
551
+ const chunks: ContentChunk[] = [];
552
+ if (binary) {
553
+ // Binaries carry no redactable text: base64 each raw chunk.
554
+ chunks.push({ content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" });
555
+ let read = 0;
556
+ while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
557
+ chunks.push({ content: buffer.subarray(0, read).toString("base64"), encoding: "base64" });
558
+ }
559
+ } else {
560
+ // Text: decode incrementally (StringDecoder absorbs multibyte
561
+ // sequences split across chunk boundaries), emit only up to the
562
+ // last complete line, carry the remainder, and redact each
563
+ // line-aligned segment.
564
+ const { StringDecoder } = await import("node:string_decoder");
565
+ const decoder = new StringDecoder("utf8");
566
+ let pending = decoder.write(buffer.subarray(0, first.bytesRead));
567
+ let read = 0;
568
+ while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
569
+ pending += decoder.write(buffer.subarray(0, read));
570
+ const lastNl = pending.lastIndexOf("\n");
571
+ if (lastNl >= 0) {
572
+ chunks.push({ content: redactCollectorText(pending.slice(0, lastNl + 1)).text });
573
+ pending = pending.slice(lastNl + 1);
574
+ }
575
+ }
576
+ pending += decoder.end();
577
+ if (pending) chunks.push({ content: redactCollectorText(pending).text });
578
+ }
579
+ return chunks;
580
+ } finally {
581
+ await handle.close();
582
+ }
583
+ }
584
+
585
+ /**
586
+ * Read one workspace entry into CollectorFile entries. Regular files:
587
+ * text is redacted, binaries base64-encoded ("encoding": "base64"), and
588
+ * the POSIX mode rides along (octal string) so replay can restore the
589
+ * executable bit (stored-and-ignored on Windows). Content bigger than
590
+ * MAX_SLICE_BYTES is sliced (see buildFileEntries). Symlinks become
591
+ * records { path, target } (target verbatim; replay recreates the link).
592
+ * Returns [] when the entry vanished or is a special file.
593
+ */
594
+ async function readWorkspaceEntries(root: string, path: string): Promise<CollectorFile[]> {
595
+ try {
596
+ const absolute = resolve(root, path);
597
+ if (portablePath(root, absolute).startsWith("../")) return [];
598
+ const file = await lstat(absolute);
599
+ if (file.isSymbolicLink()) {
600
+ const target = await readlink(absolute);
601
+ return [{ path, target }];
602
+ }
603
+ if (!file.isFile()) return [];
604
+ const mode = file.mode & 0o777;
605
+ const chunks: ContentChunk[] = await readContentChunks(absolute, file.size);
606
+ if (chunks.length === 1) {
607
+ const chunk = chunks[0];
608
+ return buildFileEntries({
609
+ path,
610
+ content: chunk.content,
611
+ ...(chunk.encoding ? { encoding: chunk.encoding } : {}),
612
+ ...(mode ? { mode: mode.toString(8) } : {}),
613
+ });
614
+ }
615
+ // Big file: streamed chunks -> one logical entry per chunk (already
616
+ // <= ~8 MiB for raw chunks; base64 inflates 4/3, so sub-slice those),
617
+ // plus the manifest for deterministic reassembly.
618
+ const encoding = chunks[0].encoding;
619
+ const entries: CollectorFile[] = [];
620
+ const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
621
+ entries.push({
622
+ path: sliceManifestPath(path),
623
+ content: JSON.stringify(sliceManifest(
624
+ {
625
+ path,
626
+ ...(encoding ? { encoding } : {}),
627
+ ...(mode ? { mode: mode.toString(8) } : {}),
628
+ },
629
+ sliceContents.map((slice, index) => ({
630
+ index,
631
+ bytes: Buffer.byteLength(slice),
632
+ sha256: sha256Hex(slice),
633
+ })),
634
+ )),
635
+ });
636
+ sliceContents.forEach((slice, index) => {
637
+ entries.push({
638
+ path: slicePartPath(path, index),
639
+ content: slice,
640
+ ...(encoding ? { encoding } : {}),
641
+ });
642
+ });
643
+ return entries;
644
+ } catch {
645
+ // Workspaces are live; races are expected and retried by the next snapshot.
646
+ return [];
647
+ }
648
+ }
649
+
433
650
  export async function collectFiles(
434
651
  root: string,
435
652
  byteLimit: number,
@@ -448,25 +665,17 @@ export async function collectFiles(
448
665
  root_name: root.split(sep).filter(Boolean).at(-1) ?? "workspace",
449
666
  git: await gitMetadata(root),
450
667
  });
451
- files.push({ path: "__omnirush__/workspace.json", content: metadata });
668
+ files.push({ path: `${AGENT_DIR}/workspace.json`, content: metadata });
452
669
  used += Buffer.byteLength(metadata);
453
670
 
454
- for (const path of await listWorkspaceFiles(root)) {
455
- if (files.length >= MAX_FILES || used >= byteLimit) break;
456
- try {
457
- const absolute = resolve(root, path);
458
- if (portablePath(root, absolute).startsWith("../")) continue;
459
- const file = await lstat(absolute);
460
- if (!file.isFile() || file.isSymbolicLink() || file.size > MAX_COLLECTOR_FILE_BYTES) continue;
461
- const buffer = await readFile(absolute);
462
- if (buffer.length > MAX_COLLECTOR_FILE_BYTES || isBinary(buffer)) continue;
463
- const redacted = redactCollectorText(buffer.toString("utf8")).text;
464
- const size = Buffer.byteLength(redacted);
671
+ const paths = (await listWorkspaceFiles(root)).sort(comparePaths);
672
+ for (const path of paths) {
673
+ if (used >= byteLimit) break;
674
+ for (const file of await readWorkspaceEntries(root, path)) {
675
+ const size = Buffer.byteLength(file.content ?? file.target ?? "");
465
676
  if (used + size > byteLimit) continue;
466
- files.push({ path, content: redacted });
677
+ files.push(file);
467
678
  used += size;
468
- } catch {
469
- // Workspaces are live; races are expected and retried by the next snapshot.
470
679
  }
471
680
  }
472
681
  return files;
@@ -505,6 +714,101 @@ export function buildEnvelopePayload(
505
714
  }));
506
715
  }
507
716
 
717
+ export type EnvelopePart = {
718
+ files: CollectorFile[];
719
+ payload: Buffer;
720
+ compressed: Buffer;
721
+ };
722
+
723
+ /**
724
+ * Deterministic part split for a full files array (used by `omnirush
725
+ * collect` and the tests): sort files by path, accumulate in order, keep
726
+ * each part <= maxCompressed compressed. A single file bigger than one
727
+ * part gets its own part; file content is never split across parts. A
728
+ * lone file that cannot fit even the hard rail is skipped with a warning.
729
+ * Envelope sequence numbers are state.sequence + 1 + partIndex.
730
+ */
731
+ export function buildEnvelopeParts(
732
+ state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
733
+ snapshotType: SnapshotType,
734
+ files: CollectorFile[],
735
+ maxCompressed = MAX_PART_COMPRESSED_BYTES,
736
+ log?: (message: string, attributes?: Record<string, unknown>) => void,
737
+ ): Promise<EnvelopePart[]> {
738
+ // Slice any entry whose content exceeds MAX_SLICE_BYTES (manifest +
739
+ // ordered parts) BEFORE packing, so callers feeding raw file lists get
740
+ // the same deterministic slice layout as the tree walker.
741
+ const sorted = files
742
+ .flatMap((file) => buildFileEntries(file))
743
+ .sort((left, right) => comparePaths(left.path, right.path));
744
+ return packEnvelopeParts(state, snapshotType, sorted, maxCompressed, log);
745
+ }
746
+
747
+ /**
748
+ * Shared greedy packer: split an ordered file list into compressed parts,
749
+ * preserving path order across parts. Used by buildEnvelopeParts (pure,
750
+ * whole list in memory) and mirrored by the collector's streaming packer
751
+ * (packAndUpload, which never holds more than one part in memory).
752
+ */
753
+ export async function packEnvelopeParts(
754
+ state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
755
+ snapshotType: SnapshotType,
756
+ sortedFiles: CollectorFile[],
757
+ maxCompressed = MAX_PART_COMPRESSED_BYTES,
758
+ log?: (message: string, attributes?: Record<string, unknown>) => void,
759
+ ): Promise<EnvelopePart[]> {
760
+ const hard = Math.max(maxCompressed, MAX_PART_COMPRESSED_HARD);
761
+ const measure = async (batch: CollectorFile[], partIndex: number): Promise<EnvelopePart> => {
762
+ const payload = buildEnvelopePayload({ ...state, sequence: (state.sequence ?? 0) + partIndex }, snapshotType, batch);
763
+ return { files: batch, payload, compressed: await compressZstd(payload) };
764
+ };
765
+ const parts: EnvelopePart[] = [];
766
+ let batch: CollectorFile[] = [];
767
+ const fit = async (batchToFit: CollectorFile[], partIndex: number) => {
768
+ let measured = await measure(batchToFit, partIndex);
769
+ const carry: CollectorFile[] = [];
770
+ // Split at the compressed target; a lone file may use the body
771
+ // headroom up to the hard rail, beyond which it is skipped loudly
772
+ // rather than split mid-content.
773
+ while (measured.compressed.length > maxCompressed && batchToFit.length > 1) {
774
+ carry.unshift(batchToFit.pop()!);
775
+ measured = await measure(batchToFit, partIndex);
776
+ }
777
+ if (measured.compressed.length > hard && batchToFit.length === 1) {
778
+ const lone = batchToFit.pop()!;
779
+ log?.("OmniRush file exceeds the upload limits and cannot be shipped as a single part; skipped", {
780
+ path: lone.path,
781
+ bytes: Buffer.byteLength(lone.content ?? lone.target ?? ""),
782
+ });
783
+ return { measured: null, carry };
784
+ }
785
+ return { measured, carry };
786
+ };
787
+ for (const file of sortedFiles) {
788
+ batch.push(file);
789
+ const partIndex = parts.length;
790
+ const estimate = batch.reduce((total, f) => total + Buffer.byteLength(f.content), 0);
791
+ if (estimate < PART_UNCOMPRESSED_STEP) continue;
792
+ const { measured, carry } = await fit(batch, partIndex);
793
+ if (measured
794
+ && (measured.compressed.length >= PART_READY_COMPRESSED_BYTES || estimate >= PART_BATCH_UNCOMPRESSED_MAX)) {
795
+ parts.push(measured);
796
+ batch = carry;
797
+ } else if (measured) {
798
+ batch = [...batch, ...carry];
799
+ } else {
800
+ batch = carry;
801
+ }
802
+ }
803
+ while (batch.length > 0) {
804
+ const { measured, carry } = await fit(batch, parts.length);
805
+ if (measured) parts.push(measured);
806
+ if (carry.length === 0) break;
807
+ batch = carry;
808
+ }
809
+ return parts;
810
+ }
811
+
508
812
  export class WorkspaceCollector {
509
813
  private readonly collectUrl: string | null;
510
814
  private token: string;
@@ -519,6 +823,9 @@ export class WorkspaceCollector {
519
823
  private readonly sessions = new Map<string, SessionState>();
520
824
  private readonly changeDebounceMs: number;
521
825
  private readonly fallbackScanMs: number;
826
+ private readonly sessionBudgetBytes: number;
827
+ private readonly partCompressedBytes: number;
828
+ private readonly partCheckBytes: number;
522
829
 
523
830
  constructor(options: CollectorOptions = {}) {
524
831
  this.collectUrl = resolveCollectUrl(options.gatewayUrl ?? process.env.OMNIRUSH_GATEWAY_URL);
@@ -531,6 +838,12 @@ export class WorkspaceCollector {
531
838
  this.ledgerReady = this.loadLedger();
532
839
  this.changeDebounceMs = options.changeDebounceMs ?? CHANGE_DEBOUNCE_MS;
533
840
  this.fallbackScanMs = options.fallbackScanMs ?? FALLBACK_SCAN_MS;
841
+ this.sessionBudgetBytes = Math.max(1024, options.sessionBudgetBytes ?? MAX_SESSION_BYTES);
842
+ this.partCompressedBytes = Math.min(
843
+ Math.max(1024, options.partCompressedBytes ?? MAX_PART_COMPRESSED_BYTES),
844
+ MAX_PART_COMPRESSED_HARD,
845
+ );
846
+ this.partCheckBytes = Math.max(1024, options.partCheckBytes ?? PART_UNCOMPRESSED_STEP);
534
847
  }
535
848
 
536
849
  private async loadLedger(): Promise<void> {
@@ -623,12 +936,14 @@ export class WorkspaceCollector {
623
936
  lastSignature: "",
624
937
  started: false,
625
938
  finished: false,
939
+ budgetExhausted: false,
626
940
  changeTimer: null,
627
941
  scanTimer: null,
628
942
  watcher: null,
943
+ watcherStartedAtMs: 0,
944
+ knownPaths: new Set(),
629
945
  trace: [],
630
- changeJournal: new Map(),
631
- changeJournalBytes: 0,
946
+ changeJournal: [],
632
947
  changeCaptureTail: Promise.resolve(),
633
948
  ready: Promise.resolve(),
634
949
  tail: Promise.resolve(),
@@ -637,11 +952,14 @@ export class WorkspaceCollector {
637
952
  this.sessions.set(sessionId, state);
638
953
  this.enqueue(state, async () => {
639
954
  await state.ready;
955
+ const baselinePaths = await listWorkspaceFiles(root);
956
+ state.knownPaths = new Set(baselinePaths);
640
957
  state.lastSignature = await workspaceSignature(root);
641
958
  await this.uploadWorkspace(state, "start");
642
959
  state.started = true;
643
960
  });
644
961
  try {
962
+ state.watcherStartedAtMs = Date.now();
645
963
  state.watcher = watch(root, { recursive: true }, (_event, filename) => {
646
964
  if (filename && isCollectorPathDenied(String(filename))) return;
647
965
  if (filename) this.queueChangedPath(state, String(filename));
@@ -665,9 +983,16 @@ export class WorkspaceCollector {
665
983
 
666
984
  recordTrace(sessionId: string, type: string, data?: unknown): void {
667
985
  const state = this.sessions.get(sessionId);
668
- if (!state || state.finished) return;
986
+ if (!state || state.finished || state.budgetExhausted) return;
669
987
  state.trace.push({ at: new Date().toISOString(), type, ...(data === undefined ? {} : { data }) });
670
- if (state.trace.length > 5_000) state.trace.splice(0, state.trace.length - 5_000);
988
+ if (state.trace.length >= TRACE_RELIEF_EVENTS) {
989
+ // Memory relief, NOT truncation: the whole batch is uploaded.
990
+ const batch = state.trace.splice(0);
991
+ this.enqueue(state, async () => {
992
+ await state.ready;
993
+ await this.uploadTrace(state, batch);
994
+ });
995
+ }
671
996
  }
672
997
 
673
998
  finishSession(sessionId: string, finalTrace?: unknown): void {
@@ -707,12 +1032,37 @@ export class WorkspaceCollector {
707
1032
  await Promise.allSettled([...this.sessions.values()].map((state) => state.tail));
708
1033
  }
709
1034
 
1035
+ private stopWatching(state: SessionState): void {
1036
+ if (state.changeTimer) {
1037
+ clearTimeout(state.changeTimer);
1038
+ state.changeTimer = null;
1039
+ }
1040
+ if (state.scanTimer) {
1041
+ clearInterval(state.scanTimer);
1042
+ state.scanTimer = null;
1043
+ }
1044
+ state.watcher?.close();
1045
+ state.watcher = null;
1046
+ }
1047
+
1048
+ private exhaustSessionBudget(state: SessionState): void {
1049
+ if (state.budgetExhausted) return;
1050
+ state.budgetExhausted = true;
1051
+ this.stopWatching(state);
1052
+ this.log("warn", "OmniRush session budget exhausted — collection stopped for this session", {
1053
+ sessionId: state.id,
1054
+ budgetBytes: this.sessionBudgetBytes,
1055
+ sentBytes: state.sentBytes,
1056
+ });
1057
+ }
1058
+
710
1059
  private scheduleChange(state: SessionState): void {
711
- if (state.finished) return;
1060
+ if (state.finished || state.budgetExhausted) return;
712
1061
  if (state.changeTimer) clearTimeout(state.changeTimer);
713
1062
  state.changeTimer = setTimeout(() => {
714
1063
  state.changeTimer = null;
715
1064
  this.enqueue(state, async () => {
1065
+ if (state.budgetExhausted || state.finished) return;
716
1066
  const signature = await workspaceSignature(state.root);
717
1067
  if (!signature || signature === state.lastSignature) return;
718
1068
  state.lastSignature = signature;
@@ -723,7 +1073,7 @@ export class WorkspaceCollector {
723
1073
  }
724
1074
 
725
1075
  private queueChangedPath(state: SessionState, filename: string): void {
726
- if (state.finished) return;
1076
+ if (state.finished || state.budgetExhausted) return;
727
1077
  state.changeCaptureTail = state.changeCaptureTail
728
1078
  .catch(() => undefined)
729
1079
  .then(() => this.captureChangedPath(state, filename))
@@ -736,51 +1086,110 @@ export class WorkspaceCollector {
736
1086
  });
737
1087
  }
738
1088
 
739
- private async captureChangedPath(state: SessionState, filename: string): Promise<void> {
1089
+ private async captureChangedPath(
1090
+ state: SessionState,
1091
+ filename: string,
1092
+ options?: { force?: boolean },
1093
+ ): Promise<void> {
740
1094
  const absolute = resolve(state.root, filename);
741
1095
  const path = portablePath(state.root, absolute);
742
- if (!path || path.startsWith("../") || isCollectorPathDenied(path) || await gitIgnoresPath(state.root, path)) return;
1096
+ if (!path || path.startsWith("../") || isCollectorPathDenied(path)) return;
1097
+ const records: ChangeJournalEntry[] = [];
743
1098
  let entry: ChangeJournalEntry;
744
1099
  try {
745
1100
  const file = await lstat(absolute);
746
- if (!file.isFile() || file.isSymbolicLink() || file.size > MAX_COLLECTOR_FILE_BYTES) {
747
- entry = { path, at: new Date().toISOString(), status: "skipped" };
1101
+ // Pre-baseline activity: the start snapshot already carries this
1102
+ // state, and macOS FSEvents (unlike inotify) delivers events for
1103
+ // files created just before watch() began — often after the start
1104
+ // upload finished. The mtime is compared at millisecond
1105
+ // resolution: sub-ms APFS timestamps make a pre-watch write's
1106
+ // mtime land a few microseconds after Date.now()'s truncated
1107
+ // stamp, which must still count as pre-baseline. Deletions stay
1108
+ // unguarded (replaying a delete of a baseline file the user
1109
+ // removed mid-session is required).
1110
+ if (!options?.force && Math.floor(file.mtimeMs) <= state.watcherStartedAtMs) return;
1111
+ state.knownPaths.add(path);
1112
+ if (file.isSymbolicLink()) {
1113
+ // Symlink record: path + target so replay recreates the link.
1114
+ const target = await readlink(absolute);
1115
+ entry = { path, at: new Date().toISOString(), status: "present", target };
1116
+ } else if (!file.isFile()) {
1117
+ return;
748
1118
  } else {
749
- const buffer = await readFile(absolute);
750
- if (buffer.length > MAX_COLLECTOR_FILE_BYTES || isBinary(buffer)) {
751
- entry = { path, at: new Date().toISOString(), status: "skipped" };
752
- } else {
1119
+ const at = new Date().toISOString();
1120
+ const mode = file.mode & 0o777;
1121
+ const modeField = mode ? { mode: mode.toString(8) } : {};
1122
+ const chunks = await readContentChunks(absolute, file.size);
1123
+ const encoding = chunks[0].encoding;
1124
+ const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
1125
+ if (sliceContents.length === 1) {
753
1126
  entry = {
754
1127
  path,
755
- at: new Date().toISOString(),
1128
+ at,
756
1129
  status: "present",
757
- content: redactCollectorText(buffer.toString("utf8")).text,
1130
+ content: sliceContents[0],
1131
+ ...(encoding ? { encoding } : {}),
1132
+ ...modeField,
758
1133
  };
1134
+ } else {
1135
+ // Big file: one ordered record per slice so the journal stays
1136
+ // replayable; reassembly rides the slice metadata.
1137
+ sliceContents.forEach((content, index) => {
1138
+ records.push({
1139
+ path,
1140
+ at,
1141
+ status: "present",
1142
+ content,
1143
+ ...(encoding ? { encoding } : {}),
1144
+ ...modeField,
1145
+ slice: {
1146
+ index,
1147
+ total: sliceContents.length,
1148
+ bytes: Buffer.byteLength(content),
1149
+ sha256: sha256Hex(content),
1150
+ },
1151
+ });
1152
+ });
1153
+ state.changeJournal.push(...records);
1154
+ return;
759
1155
  }
760
1156
  }
761
- } catch {
1157
+ } catch (error) {
1158
+ // Only a real disappearance is a deletion. Any other lstat failure
1159
+ // (EPERM/EBUSY/... under heavy watcher load) must not fabricate a
1160
+ // deleted record for a file that still exists — the fallback scan
1161
+ // and the end snapshot re-check those paths.
1162
+ if ((error as any)?.code !== "ENOENT") return;
1163
+ // Only paths the baseline (or a captured change) saw can be
1164
+ // deleted inside the workspace; FSEvents sibling/parent leaks
1165
+ // reference foreign paths that must not fabricate deletions.
1166
+ if (!state.knownPaths.has(path)) return;
762
1167
  entry = { path, at: new Date().toISOString(), status: "deleted" };
763
1168
  }
764
- const previous = state.changeJournal.get(path);
765
- if (previous) state.changeJournalBytes -= Buffer.byteLength(JSON.stringify(previous));
766
- state.changeJournal.set(path, entry);
767
- state.changeJournalBytes += Buffer.byteLength(JSON.stringify(entry));
768
- while (state.changeJournal.size > MAX_CHANGE_JOURNAL_ENTRIES || state.changeJournalBytes > MAX_CHANGE_JOURNAL_BYTES) {
769
- const oldest = state.changeJournal.keys().next().value as string | undefined;
770
- if (!oldest) break;
771
- const removed = state.changeJournal.get(oldest);
772
- if (removed) state.changeJournalBytes -= Buffer.byteLength(JSON.stringify(removed));
773
- state.changeJournal.delete(oldest);
1169
+ // Consecutive-duplicate suppression: macOS FSEvents can deliver a
1170
+ // late event for a path that was also captured directly, and an
1171
+ // identical re-capture adds upload fat without a state change. The
1172
+ // replay result is identical either way.
1173
+ const journalSignature = (record: ChangeJournalEntry) =>
1174
+ JSON.stringify([record.status, record.content, record.encoding, record.target, record.mode]);
1175
+ const pendingRecords = [entry, ...records];
1176
+ for (const record of pendingRecords) {
1177
+ let duplicate = false;
1178
+ for (let i = state.changeJournal.length - 1, scanned = 0; i >= 0 && scanned < 100; i--, scanned++) {
1179
+ const prior = state.changeJournal[i];
1180
+ if (prior.path !== record.path) continue;
1181
+ duplicate = journalSignature(prior) === journalSignature(record);
1182
+ break;
1183
+ }
1184
+ if (!duplicate) state.changeJournal.push(record);
774
1185
  }
775
1186
  }
776
1187
 
777
- private acknowledgeJournal(state: SessionState, entries: ChangeJournalEntry[]): void {
778
- for (const entry of entries) {
779
- if (state.changeJournal.get(entry.path)?.at === entry.at) {
780
- state.changeJournal.delete(entry.path);
781
- state.changeJournalBytes -= Buffer.byteLength(JSON.stringify(entry));
782
- }
783
- }
1188
+ /** Remove exactly the uploaded entries from the ordered journal. */
1189
+ private acknowledgeJournal(state: SessionState, uploaded: ChangeJournalEntry[]): void {
1190
+ if (uploaded.length === 0) return;
1191
+ const sent = uploaded.length > 32 ? new Set(uploaded) : null;
1192
+ state.changeJournal = state.changeJournal.filter((entry) => sent ? !sent.has(entry) : !uploaded.includes(entry));
784
1193
  }
785
1194
 
786
1195
  private enqueue(state: SessionState, operation: () => Promise<void>): void {
@@ -792,41 +1201,342 @@ export class WorkspaceCollector {
792
1201
  });
793
1202
  }
794
1203
 
795
- private async uploadWorkspace(state: SessionState, type: Exclude<SnapshotType, "trace">): Promise<void> {
796
- const remaining = MAX_COLLECTOR_SESSION_BYTES - state.sentBytes;
797
- if (remaining <= 1_024) return;
798
- const files = await collectFiles(
799
- state.root,
800
- Math.min(MAX_SNAPSHOT_BYTES, remaining - 1_024),
801
- state.workspaceId,
802
- state,
803
- );
804
- const journal = type === "change" || type === "end" ? [...state.changeJournal.values()] : [];
805
- if (journal.length > 0) {
1204
+ private async uploadWorkspace(state: SessionState, type: "start" | "change" | "end"): Promise<void> {
1205
+ if (state.budgetExhausted) return;
1206
+ if (type === "change") {
1207
+ // Change snapshots carry ONLY the touched files from the journal —
1208
+ // never a full re-enumeration of the tree.
1209
+ const snapshot = [...state.changeJournal];
1210
+ if (snapshot.length === 0) {
1211
+ state.lastSignature = await workspaceSignature(state.root);
1212
+ return;
1213
+ }
1214
+ const uploaded = await this.uploadChangeParts(state, snapshot);
1215
+ // Acknowledge only after every part of the snapshot succeeded; on a
1216
+ // mid-budget exhaustion, acknowledge only what actually shipped.
1217
+ this.acknowledgeJournal(state, state.budgetExhausted ? uploaded.flat() : snapshot);
1218
+ } else {
1219
+ await this.uploadTreeParts(state, type);
1220
+ }
1221
+ state.lastSignature = await workspaceSignature(state.root);
1222
+ }
1223
+
1224
+ private async workspaceMetadataFile(state: SessionState): Promise<CollectorFile> {
1225
+ return {
1226
+ path: `${AGENT_DIR}/workspace.json`,
1227
+ content: JSON.stringify({
1228
+ workspace_id: state.workspaceId,
1229
+ session_id: state.id,
1230
+ session_segment: state.segment,
1231
+ session_resumed: state.resumed,
1232
+ root_name: state.root.split(sep).filter(Boolean).at(-1) ?? "workspace",
1233
+ git: await gitMetadata(state.root),
1234
+ }),
1235
+ };
1236
+ }
1237
+
1238
+ /** Full-tree snapshot ("start" | "end"), streamed into ordered parts. */
1239
+ private async uploadTreeParts(state: SessionState, snapshotType: "start" | "end"): Promise<void> {
1240
+ const paths = (await listWorkspaceFiles(state.root)).sort(comparePaths);
1241
+ const metadata = await this.workspaceMetadataFile(state);
1242
+ let metadataSent = false;
1243
+ const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
1244
+ const measure = async (batch: CollectorFile[]) => {
1245
+ const files = metadataSent ? batch : [metadata, ...batch];
1246
+ const payload = buildEnvelopePayload(bySequence(), snapshotType, files);
1247
+ return { files, payload, compressed: await compressZstd(payload) };
1248
+ };
1249
+ await this.packAndUpload(state, snapshotType, this.iterTreeFiles(state, paths), {
1250
+ sizeOf: (file) => Buffer.byteLength(file.content ?? file.target ?? "") + file.path.length + 64,
1251
+ measure,
1252
+ upload: async (measured) => {
1253
+ const ok = await this.uploadEnvelope(state, snapshotType, measured.files, measured);
1254
+ if (ok) metadataSent = true;
1255
+ return ok;
1256
+ },
1257
+ drop: (file) => {
1258
+ this.log("warn", "OmniRush file exceeds the upload rail and cannot be shipped as a single part; skipped", {
1259
+ sessionId: state.id,
1260
+ path: file.path,
1261
+ bytes: Buffer.byteLength(file.content ?? file.target ?? ""),
1262
+ });
1263
+ state.trace.push({
1264
+ at: new Date().toISOString(),
1265
+ type: "file.unshippable",
1266
+ data: { path: file.path, bytes: Buffer.byteLength(file.content ?? file.target ?? "") },
1267
+ });
1268
+ },
1269
+ });
1270
+ }
1271
+
1272
+ private async *iterTreeFiles(state: SessionState, paths: string[]): AsyncGenerator<CollectorFile> {
1273
+ for (const path of paths) {
1274
+ if (state.budgetExhausted) return;
1275
+ for (const entry of await readWorkspaceEntries(state.root, path)) {
1276
+ yield entry;
1277
+ }
1278
+ }
1279
+ }
1280
+
1281
+ /**
1282
+ * Change snapshot: ordered journal slices. Each part carries the
1283
+ * touched files for its slice (deduped last-wins, sorted by path) plus
1284
+ * the __agent__/changes.json sidecar holding the slice's ordered
1285
+ * records — replaying start + sidecars in sequence order reproduces the
1286
+ * end tree.
1287
+ */
1288
+ private async uploadChangeParts(state: SessionState, snapshot: ChangeJournalEntry[]): Promise<ChangeJournalEntry[][]> {
1289
+ const uploadedSlices: ChangeJournalEntry[][] = [];
1290
+ const filesForSlice = (entries: ChangeJournalEntry[]): CollectorFile[] => {
1291
+ const lastByPath = new Map<string, CollectorFile>();
1292
+ const slicedByPath = new Map<string, ChangeJournalEntry[]>();
1293
+ for (const entry of entries) {
1294
+ if (entry.status !== "present") continue;
1295
+ if (entry.target !== undefined) {
1296
+ lastByPath.set(entry.path, { path: entry.path, target: entry.target });
1297
+ } else if (entry.slice) {
1298
+ const group = slicedByPath.get(entry.path) ?? [];
1299
+ group.push(entry);
1300
+ slicedByPath.set(entry.path, group);
1301
+ } else if (entry.content !== undefined) {
1302
+ lastByPath.set(entry.path, {
1303
+ path: entry.path,
1304
+ content: entry.content,
1305
+ ...(entry.encoding ? { encoding: entry.encoding } : {}),
1306
+ ...(entry.mode ? { mode: entry.mode } : {}),
1307
+ });
1308
+ }
1309
+ }
1310
+ const files = [...lastByPath.values()];
1311
+ for (const [logicalPath, records] of slicedByPath) {
1312
+ records.sort((left, right) => (left.slice?.index ?? 0) - (right.slice?.index ?? 0));
1313
+ const encoding = records.find((record) => record.encoding)?.encoding;
1314
+ const mode = records.find((record) => record.mode)?.mode;
1315
+ for (const record of records) {
1316
+ files.push({
1317
+ path: slicePartPath(logicalPath, record.slice!.index),
1318
+ content: record.content,
1319
+ ...(encoding ? { encoding } : {}),
1320
+ });
1321
+ }
1322
+ files.push({
1323
+ path: sliceManifestPath(logicalPath),
1324
+ content: JSON.stringify(sliceManifest(
1325
+ {
1326
+ path: logicalPath,
1327
+ ...(encoding ? { encoding } : {}),
1328
+ ...(mode ? { mode } : {}),
1329
+ },
1330
+ records.map((record) => ({
1331
+ index: record.slice!.index,
1332
+ bytes: record.slice!.bytes,
1333
+ sha256: record.slice!.sha256,
1334
+ })),
1335
+ records[0].slice!.total,
1336
+ )),
1337
+ });
1338
+ }
1339
+ files.sort((left, right) => comparePaths(left.path, right.path));
806
1340
  files.push({
807
- path: "__omnirush__/changes.json",
808
- content: JSON.stringify({ schema_version: 1, session_id: state.id, entries: journal }),
1341
+ path: `${AGENT_DIR}/changes.json`,
1342
+ content: JSON.stringify({ schema_version: 1, session_id: state.id, entries }),
809
1343
  });
1344
+ return files;
1345
+ };
1346
+ const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
1347
+ await this.packAndUpload(state, "change", snapshot, {
1348
+ sizeOf: (entry) =>
1349
+ Buffer.byteLength(entry.content ?? entry.target ?? "") + entry.path.length + 96,
1350
+ measure: async (batch) => {
1351
+ const files = filesForSlice(batch);
1352
+ const payload = buildEnvelopePayload(bySequence(), "change", files);
1353
+ return { files, payload, compressed: await compressZstd(payload) };
1354
+ },
1355
+ upload: async (measured, batch) => {
1356
+ const ok = await this.uploadEnvelope(state, "change", measured.files, measured);
1357
+ if (ok) uploadedSlices.push(batch);
1358
+ return ok;
1359
+ },
1360
+ drop: (entry) => {
1361
+ this.log("warn", "OmniRush change record exceeds the upload limits and cannot be shipped; skipped", {
1362
+ sessionId: state.id,
1363
+ path: entry.path,
1364
+ bytes: Buffer.byteLength(entry.content ?? entry.target ?? ""),
1365
+ });
1366
+ state.trace.push({
1367
+ at: new Date().toISOString(),
1368
+ type: "file.unshippable",
1369
+ data: { path: entry.path, reason: "change_record_over_limit" },
1370
+ });
1371
+ },
1372
+ }, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
1373
+ return uploadedSlices;
1374
+ }
1375
+
1376
+ /**
1377
+ * Streaming packer shared by tree and change uploads: accumulate units
1378
+ * in order, compress-check every partCheckBytes of uncompressed growth,
1379
+ * flush parts targeting partCompressedBytes, and respect the hard rail
1380
+ * (a lone unit over the rail is skipped via hooks.drop — never split).
1381
+ * fit() mutates the batch it is given: the returned `measured` always
1382
+ * corresponds to the (mutated) batch passed to hooks.upload, and popped
1383
+ * units come back as `carry` to lead the next part. hooks.upload must
1384
+ * persist sequence bookkeeping via uploadEnvelope.
1385
+ */
1386
+ private async packAndUpload<U>(
1387
+ state: SessionState,
1388
+ snapshotType: SnapshotType,
1389
+ units: Iterable<U> | AsyncIterable<U>,
1390
+ hooks: {
1391
+ sizeOf: (unit: U) => number;
1392
+ measure: (batch: U[]) => Promise<{ files: CollectorFile[]; payload: Buffer; compressed: Buffer }>;
1393
+ upload: (measured: { files: CollectorFile[]; payload: Buffer; compressed: Buffer }, batch: U[]) => Promise<boolean>;
1394
+ drop?: (unit: U) => void;
1395
+ },
1396
+ opts: { checkBytes?: number; flushPlainBytes?: number } = {},
1397
+ ): Promise<void> {
1398
+ const checkBytes = Math.max(1024, opts.checkBytes ?? this.partCheckBytes);
1399
+ const flushPlainBytes = Math.max(checkBytes, opts.flushPlainBytes ?? PART_BATCH_UNCOMPRESSED_MAX);
1400
+ const hard = MAX_PART_COMPRESSED_HARD;
1401
+ const fit = async (batch: U[]) => {
1402
+ let measured = await hooks.measure(batch);
1403
+ const carry: U[] = [];
1404
+ // Multi-file parts split at the compressed target. A lone unit
1405
+ // (e.g. one slice entry) may use the body headroom up to the hard
1406
+ // rail; beyond that it is dropped loudly rather than split
1407
+ // mid-content.
1408
+ while (measured.compressed.length > this.partCompressedBytes && batch.length > 1) {
1409
+ carry.unshift(batch.pop()!);
1410
+ measured = await hooks.measure(batch);
1411
+ }
1412
+ if (measured.compressed.length > hard && batch.length === 1) {
1413
+ const lone = batch.pop()!;
1414
+ hooks.drop?.(lone);
1415
+ return { measured: batch.length > 0 ? await hooks.measure(batch) : null, carry };
1416
+ }
1417
+ return { measured, carry };
1418
+ };
1419
+
1420
+ let batch: U[] = [];
1421
+ let batchBytes = 0;
1422
+ let nextCheck = checkBytes;
1423
+ const resetBatch = (unitsToKeep: U[]) => {
1424
+ batch = unitsToKeep;
1425
+ batchBytes = batch.reduce((total, unit) => total + hooks.sizeOf(unit), 0);
1426
+ nextCheck = batchBytes + checkBytes;
1427
+ };
1428
+ const shouldFlush = (measured: { compressed: Buffer }) =>
1429
+ measured.compressed.length >= PART_READY_COMPRESSED_BYTES || batchBytes >= flushPlainBytes;
1430
+
1431
+ for await (const unit of units as AsyncIterable<U>) {
1432
+ if (state.budgetExhausted) return;
1433
+ batch.push(unit);
1434
+ batchBytes += hooks.sizeOf(unit);
1435
+ if (batchBytes < nextCheck) continue;
1436
+ const { measured, carry } = await fit(batch);
1437
+ if (measured && shouldFlush(measured)) {
1438
+ if (!(await hooks.upload(measured, batch))) return;
1439
+ resetBatch(carry);
1440
+ } else if (measured) {
1441
+ // Under the flush thresholds — keep accumulating (carry is empty
1442
+ // here: popping only happens above the flush levels).
1443
+ batch = [...batch, ...carry];
1444
+ nextCheck = batchBytes + checkBytes;
1445
+ } else {
1446
+ resetBatch(carry);
1447
+ }
1448
+ }
1449
+ while (batch.length > 0 && !state.budgetExhausted) {
1450
+ const { measured, carry } = await fit(batch);
1451
+ if (measured) {
1452
+ if (!(await hooks.upload(measured, batch))) return;
1453
+ }
1454
+ if (carry.length === 0) break;
1455
+ batch = carry;
1456
+ batchBytes = batch.reduce((total, unit) => total + hooks.sizeOf(unit), 0);
810
1457
  }
811
- const uploaded = await this.uploadEnvelope(state, type, files);
812
- if (uploaded && journal.length > 0) this.acknowledgeJournal(state, journal);
813
- state.lastSignature = await workspaceSignature(state.root);
814
1458
  }
815
1459
 
1460
+ /** Chunked trace upload: event subsets per part, ordered by sequence. */
816
1461
  private async uploadTrace(state: SessionState, traceEvents: TraceEvent[]): Promise<void> {
817
- const remaining = MAX_COLLECTOR_SESSION_BYTES - state.sentBytes;
818
- if (remaining <= 1_024 || traceEvents.length === 0) return;
819
- const bounded = boundedTracePayload(state, traceEvents, Math.min(MAX_TRACE_BYTES, remaining - 1_024));
820
- await this.uploadEnvelope(state, "trace", [{ path: "__omnirush__/trace.json", content: bounded.toString("utf8") }]);
1462
+ if (state.budgetExhausted || traceEvents.length === 0) return;
1463
+ const remaining = this.sessionBudgetBytes - state.sentBytes;
1464
+ if (remaining <= 1024) {
1465
+ this.exhaustSessionBudget(state);
1466
+ return;
1467
+ }
1468
+ // Budget wall: truncate (loudly, trace_truncated=true) only here.
1469
+ const selected = selectTraceEvents(traceEvents, remaining - 1024);
1470
+ const meta = {
1471
+ truncated: selected.truncated,
1472
+ dropped: selected.droppedCount,
1473
+ };
1474
+ if (selected.truncated) {
1475
+ this.log("warn", "OmniRush trace hit the session budget wall; older events dropped", {
1476
+ sessionId: state.id,
1477
+ droppedEventCount: selected.droppedCount,
1478
+ });
1479
+ }
1480
+ let partIndex = 0;
1481
+ const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
1482
+ await this.packAndUpload(state, "trace", selected.events, {
1483
+ sizeOf: (event) => Buffer.byteLength(JSON.stringify(event)) + 32,
1484
+ measure: async (batch) => {
1485
+ const content = JSON.stringify({
1486
+ schema_version: 1,
1487
+ schema: AGENT_TRACE_SCHEMA,
1488
+ session_id: state.id,
1489
+ session_segment: state.segment,
1490
+ session_resumed: state.resumed,
1491
+ trace_truncated: meta.truncated,
1492
+ dropped_event_count: meta.dropped,
1493
+ trace_part: { index: partIndex },
1494
+ events: batch,
1495
+ });
1496
+ const files = [{ path: `${AGENT_DIR}/trace.json`, content }];
1497
+ const payload = buildEnvelopePayload(bySequence(), "trace", files);
1498
+ return { files, payload, compressed: await compressZstd(payload) };
1499
+ },
1500
+ upload: async (measured) => {
1501
+ const ok = await this.uploadEnvelope(state, "trace", measured.files, measured);
1502
+ if (ok) partIndex += 1;
1503
+ return ok;
1504
+ },
1505
+ drop: (event) => {
1506
+ this.log("warn", "OmniRush trace event exceeds the upload limits; skipped", {
1507
+ sessionId: state.id,
1508
+ type: event.type,
1509
+ bytes: Buffer.byteLength(JSON.stringify(event)),
1510
+ });
1511
+ },
1512
+ }, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
821
1513
  }
822
1514
 
823
- private async uploadEnvelope(state: SessionState, snapshotType: SnapshotType, files: CollectorFile[]): Promise<boolean> {
1515
+ /**
1516
+ * Send one envelope (one part). `precompressed` skips the rebuild +
1517
+ * recompress when the caller already measured this exact batch.
1518
+ */
1519
+ private async uploadEnvelope(
1520
+ state: SessionState,
1521
+ snapshotType: SnapshotType,
1522
+ files: CollectorFile[],
1523
+ precompressed?: { payload: Buffer; compressed: Buffer },
1524
+ ): Promise<boolean> {
824
1525
  if (!this.collectUrl && !this.uploader) return false;
825
- const payload = buildEnvelopePayload(state, snapshotType, files);
826
- if (state.sentBytes + payload.length > MAX_COLLECTOR_SESSION_BYTES) return false;
827
- const compressed = await compressZstd(payload);
828
- if (compressed.length > MAX_COMPRESSED_BYTES) {
829
- this.log("warn", "OmniRush collection payload exceeded compressed limit", { sessionId: state.id, snapshotType });
1526
+ if (state.budgetExhausted) return false;
1527
+ const payload = precompressed?.payload ?? buildEnvelopePayload(state, snapshotType, files);
1528
+ if (state.sentBytes + payload.length > this.sessionBudgetBytes) {
1529
+ this.exhaustSessionBudget(state);
1530
+ return false;
1531
+ }
1532
+ const compressed = precompressed?.compressed ?? await compressZstd(payload);
1533
+ if (compressed.length > MAX_PART_COMPRESSED_HARD) {
1534
+ this.log("warn", "OmniRush part exceeded the backend body rail; not sent", {
1535
+ sessionId: state.id,
1536
+ snapshotType,
1537
+ compressedBytes: compressed.length,
1538
+ railBytes: MAX_PART_COMPRESSED_HARD,
1539
+ });
830
1540
  return false;
831
1541
  }
832
1542
  const send = () => this.uploader
@@ -839,7 +1549,7 @@ export class WorkspaceCollector {
839
1549
  "X-Omnirush-Session-ID": state.id,
840
1550
  },
841
1551
  body: compressed.buffer.slice(compressed.byteOffset, compressed.byteOffset + compressed.byteLength) as ArrayBuffer,
842
- signal: AbortSignal.timeout(30_000),
1552
+ signal: AbortSignal.timeout(120_000),
843
1553
  });
844
1554
  let response = await send();
845
1555
  if (response.status === 401 && this.refresh) {
@@ -867,6 +1577,7 @@ export class WorkspaceCollector {
867
1577
  snapshotType,
868
1578
  fileCount: files.length,
869
1579
  compressedBytes: compressed.length,
1580
+ sequence: state.sequence,
870
1581
  });
871
1582
  return true;
872
1583
  }