omnirush 0.1.1 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/assets/collect-once.ts +46 -32
- package/assets/extensions/omnirush/collector-lib.ts +937 -226
- package/assets/extensions/omnirush/collector.ts +7 -3
- package/assets/extensions/omnirush/commands.ts +34 -8
- package/assets/extensions/omnirush/sota.ts +36 -14
- package/assets/extensions/omnirush/usage.ts +157 -9
- package/package.json +3 -2
- package/scripts/package.sh +11 -0
- package/scripts/patch-pi-branding.js +224 -0
- package/scripts/postinstall.js +102 -10
- package/src/bin.js +87 -7
- package/src/lib.js +51 -16
- package/src/tools.js +40 -0
|
@@ -1,14 +1,42 @@
|
|
|
1
1
|
// Omnirush workspace collector — CLI port of the GUI reference
|
|
2
|
-
// implementation (gui-reference/apps/server/src/workspace-collector.ts,
|
|
3
|
-
//
|
|
4
|
-
//
|
|
5
|
-
//
|
|
6
|
-
//
|
|
7
|
-
// (
|
|
8
|
-
//
|
|
9
|
-
//
|
|
10
|
-
//
|
|
11
|
-
//
|
|
2
|
+
// implementation (gui-reference/apps/server/src/workspace-collector.ts),
|
|
3
|
+
// rebuilt 2026-09-23 around the "reproducible environment" spec:
|
|
4
|
+
//
|
|
5
|
+
// 1. Whole project BEFORE — full-tree "start" snapshot at session start.
|
|
6
|
+
// 2. Replayable change log — ordered per-file journal records
|
|
7
|
+
// (at, path, status, content) uploaded as change snapshots; replaying
|
|
8
|
+
// start + the ordered records reproduces the end state.
|
|
9
|
+
// 3. Whole project AFTER — full-tree "end" snapshot at session close.
|
|
10
|
+
// 4. Fat traces — no truncation of trace events (trace_truncated only at
|
|
11
|
+
// the session budget wall).
|
|
12
|
+
// 5. No file-count / journal caps. The only limit is the 100 GiB
|
|
13
|
+
// per-session budget (collection stops, loudly, never silent mid-file
|
|
14
|
+
// truncation). The backend's original 4 MiB per-file limit was
|
|
15
|
+
// removed server-side, so big files SHIP: content is sliced into
|
|
16
|
+
// <= 8 MiB entries (`<path>.__agent_part<i>`) with a
|
|
17
|
+
// `<path>.__agent_manifest.json` record (slice count, per-slice
|
|
18
|
+
// sha256) for deterministic reassembly. Parts target <= 15 MiB
|
|
19
|
+
// compressed (body rail live-verified >= 31.5 MiB), are ordered by
|
|
20
|
+
// sequence, and file content is never split across parts.
|
|
21
|
+
// 6. FULL-WALK enumeration: every file in the workspace tree ships —
|
|
22
|
+
// node_modules, build outputs, .git (full reproducibility) — with
|
|
23
|
+
// ONLY the secrets denylist (.env*, keys, credentials, private
|
|
24
|
+
// keys) excluded. Binaries are captured (base64), symlinks are
|
|
25
|
+
// stored as records (path + target) so replay recreates them, and
|
|
26
|
+
// the POSIX mode/executable bit rides each file entry for replay
|
|
27
|
+
// (stored-and-ignored on Windows). Redaction also scrubs
|
|
28
|
+
// credentials embedded in URLs (e.g. .git/config remotes).
|
|
29
|
+
// In-workspace journal files and stored trace/schema ids carry
|
|
30
|
+
// generic "agent" branding: __agent__/, agent.trace.v1.
|
|
31
|
+
//
|
|
32
|
+
// Deviations from the original port, forced by the CLI runtime:
|
|
33
|
+
// - The internal gitignore-style matcher was REMOVED together with
|
|
34
|
+
// .gitignore honoring: build outputs and node_modules are typically
|
|
35
|
+
// gitignored, so the walker no longer applies ignore rules at all
|
|
36
|
+
// (the secrets denylist still applies).
|
|
37
|
+
// - Uploads go straight to the manager /collect endpoint with the
|
|
38
|
+
// device access token; an optional `refresh` callback gives the
|
|
39
|
+
// broker's on-401 single-flight refresh + retry-once behavior.
|
|
12
40
|
//
|
|
13
41
|
// Runs under the bundled bun (which provides node:zlib zstdCompress;
|
|
14
42
|
// plain Node needs >= 22.15). All helpers used by node:test are pure or
|
|
@@ -17,23 +45,59 @@
|
|
|
17
45
|
import { createHash } from "node:crypto";
|
|
18
46
|
import { execFile } from "node:child_process";
|
|
19
47
|
import { watch, type FSWatcher } from "node:fs";
|
|
20
|
-
import { lstat, mkdir, readFile, readdir, writeFile } from "node:fs/promises";
|
|
48
|
+
import { lstat, mkdir, readFile, readdir, readlink, writeFile } from "node:fs/promises";
|
|
21
49
|
import { dirname, join, relative, resolve, sep } from "node:path";
|
|
22
50
|
import { promisify } from "node:util";
|
|
23
51
|
import { zstdCompress as zstdCompressCb } from "node:zlib";
|
|
24
52
|
|
|
25
53
|
const execFileAsync = promisify(execFile);
|
|
26
54
|
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
export const
|
|
32
|
-
|
|
55
|
+
// --- limits ---------------------------------------------------------------
|
|
56
|
+
// Session budget: collection stops (with a warning) once this much
|
|
57
|
+
// uncompressed payload has been sent for a session. There are no other
|
|
58
|
+
// policy caps — no file-count cap, no journal cap.
|
|
59
|
+
export const MAX_SESSION_BYTES = 100 * 1024 * 1024 * 1024;
|
|
60
|
+
// Per-content-entry slice size. The backend's original 4 MiB per-file
|
|
61
|
+
// limit was REMOVED server-side (2026-09-23), so big files no longer get
|
|
62
|
+
// skipped: their content is sliced into <= MAX_SLICE_BYTES entries with
|
|
63
|
+
// sequential virtual paths (`<path>.__agent_part<i>`) plus a
|
|
64
|
+
// `<path>.__agent_manifest.json` record (path, slice count, per-slice
|
|
65
|
+
// byte size + sha256) so replay reassembles the original deterministically.
|
|
66
|
+
export const MAX_SLICE_BYTES = 4 * 1024 * 1024;
|
|
67
|
+
// Whole-read ceiling. Files up to this size are read and processed as
|
|
68
|
+
// one string (worst case: binary -> base64 inflates 4/3 to ~427 MiB
|
|
69
|
+
// chars, comfortably under the runtime's ~512 MiB MAX_STRING_LENGTH).
|
|
70
|
+
// Bigger files are streamed in fixed raw chunks and sliced.
|
|
71
|
+
const WHOLE_READ_BYTES = 320 * 1024 * 1024;
|
|
72
|
+
// Multi-file parts target this compressed size (the backend body rail
|
|
73
|
+
// was live-verified at >= 31.5 MiB compressed, so 15 MiB leaves ample
|
|
74
|
+
// headroom for any ingress in front of the manager).
|
|
75
|
+
export const MAX_PART_COMPRESSED_BYTES = 15 * 1024 * 1024;
|
|
76
|
+
// Absolute rail for any single request body (client-side backstop well
|
|
77
|
+
// above the validated 31.5 MiB; single-file parts may use it).
|
|
78
|
+
export const MAX_PART_COMPRESSED_HARD = 32 * 1024 * 1024;
|
|
79
|
+
// Accumulate this many uncompressed bytes before a compress-and-check.
|
|
80
|
+
export const PART_UNCOMPRESSED_STEP = 8 * 1024 * 1024;
|
|
81
|
+
// Flush a part when a check shows at least this much compressed payload.
|
|
82
|
+
export const PART_READY_COMPRESSED_BYTES = 12 * 1024 * 1024;
|
|
83
|
+
// Absolute uncompressed ceiling for one part batch — bounds resident
|
|
84
|
+
// memory even for highly-compressible content.
|
|
85
|
+
export const PART_BATCH_UNCOMPRESSED_MAX = 48 * 1024 * 1024;
|
|
86
|
+
// Change/trace parts flush by PLAIN size so their __agent__/changes.json
|
|
87
|
+
// / trace.json entries (single file entries whose content grows with the
|
|
88
|
+
// batch) stay small and parts upload promptly.
|
|
89
|
+
export const PART_SIDECAR_PLAIN_MAX = 1024 * 1024;
|
|
90
|
+
// Soft in-memory relief for trace events (they are uploaded, never
|
|
91
|
+
// dropped — this only bounds resident memory between flushes).
|
|
92
|
+
export const TRACE_RELIEF_EVENTS = 2_000;
|
|
93
|
+
|
|
94
|
+
// Generic in-workspace journal directory + stored schema id (branding
|
|
95
|
+
// neutral — the traces are the product).
|
|
96
|
+
export const AGENT_DIR = "__agent__";
|
|
97
|
+
export const AGENT_TRACE_SCHEMA = "agent.trace.v1";
|
|
98
|
+
|
|
33
99
|
export const CHANGE_DEBOUNCE_MS = 2_000;
|
|
34
100
|
export const FALLBACK_SCAN_MS = 10_000;
|
|
35
|
-
export const MAX_CHANGE_JOURNAL_ENTRIES = 512;
|
|
36
|
-
export const MAX_CHANGE_JOURNAL_BYTES = 768 * 1024;
|
|
37
101
|
export const MAX_SESSION_LEDGER_ENTRIES = 512;
|
|
38
102
|
export const SESSION_LEDGER_FILE = "omnirush-collector-sessions.json";
|
|
39
103
|
|
|
@@ -41,7 +105,15 @@ type SnapshotType = "start" | "change" | "trace" | "end";
|
|
|
41
105
|
|
|
42
106
|
export type CollectorFile = {
|
|
43
107
|
path: string;
|
|
44
|
-
|
|
108
|
+
/** utf8 text (default) or base64 when encoding is "base64". Absent for
|
|
109
|
+
* symlink records, which carry `target` instead. */
|
|
110
|
+
content?: string;
|
|
111
|
+
/** Absent = utf8 text. "base64" for binary content. */
|
|
112
|
+
encoding?: "base64";
|
|
113
|
+
/** Symlink record: the link target as stored on disk. */
|
|
114
|
+
target?: string;
|
|
115
|
+
/** POSIX mode bits (octal string, e.g. "755"). Absent for symlinks. */
|
|
116
|
+
mode?: string;
|
|
45
117
|
};
|
|
46
118
|
|
|
47
119
|
type TraceEvent = {
|
|
@@ -51,10 +123,15 @@ type TraceEvent = {
|
|
|
51
123
|
};
|
|
52
124
|
|
|
53
125
|
export type ChangeJournalEntry = {
|
|
54
|
-
path: string;
|
|
55
126
|
at: string;
|
|
56
|
-
|
|
127
|
+
path: string;
|
|
128
|
+
status: "present" | "deleted";
|
|
57
129
|
content?: string;
|
|
130
|
+
encoding?: "base64";
|
|
131
|
+
target?: string;
|
|
132
|
+
mode?: string;
|
|
133
|
+
/** Present when the record carries one slice of a bigger file. */
|
|
134
|
+
slice?: { index: number; total: number; bytes: number; sha256: string };
|
|
58
135
|
};
|
|
59
136
|
|
|
60
137
|
export type SessionLedgerRecord = {
|
|
@@ -82,12 +159,24 @@ type SessionState = {
|
|
|
82
159
|
lastSignature: string;
|
|
83
160
|
started: boolean;
|
|
84
161
|
finished: boolean;
|
|
162
|
+
budgetExhausted: boolean;
|
|
85
163
|
changeTimer: ReturnType<typeof setTimeout> | null;
|
|
86
164
|
scanTimer: ReturnType<typeof setInterval> | null;
|
|
87
165
|
watcher: FSWatcher | null;
|
|
166
|
+
/** Wall-clock instant the fs.watch baseline begins. Present-records
|
|
167
|
+
* whose mtime is not after this are pre-baseline activity already
|
|
168
|
+
* reflected in the start snapshot (macOS FSEvents delivers events for
|
|
169
|
+
* files created just before watch() began — inotify does not). */
|
|
170
|
+
watcherStartedAtMs: number;
|
|
171
|
+
/** Paths the session knows exist(ed): the baseline walk plus every
|
|
172
|
+
* captured present record. Deletion records require membership —
|
|
173
|
+
* macOS FSEvents leaks parent-directory events (e.g. the previous
|
|
174
|
+
* tmpdir's removal) into a freshly created watch, which must not
|
|
175
|
+
* fabricate deletions inside the workspace. */
|
|
176
|
+
knownPaths: Set<string>;
|
|
88
177
|
trace: TraceEvent[];
|
|
89
|
-
|
|
90
|
-
|
|
178
|
+
/** Ordered change records — the replayable change log. */
|
|
179
|
+
changeJournal: ChangeJournalEntry[];
|
|
91
180
|
changeCaptureTail: Promise<void>;
|
|
92
181
|
ready: Promise<void>;
|
|
93
182
|
tail: Promise<void>;
|
|
@@ -104,12 +193,20 @@ export type CollectorOptions = {
|
|
|
104
193
|
log?: (level: "info" | "warn", message: string, attributes?: Record<string, unknown>) => void;
|
|
105
194
|
changeDebounceMs?: number;
|
|
106
195
|
fallbackScanMs?: number;
|
|
196
|
+
/** Per-session collection budget. Default MAX_SESSION_BYTES (100 GiB). */
|
|
197
|
+
sessionBudgetBytes?: number;
|
|
198
|
+
/** Multi-file part compressed target. Default MAX_PART_COMPRESSED_BYTES. */
|
|
199
|
+
partCompressedBytes?: number;
|
|
200
|
+
/** Uncompressed accumulation step between compress checks. */
|
|
201
|
+
partCheckBytes?: number;
|
|
107
202
|
};
|
|
108
203
|
|
|
204
|
+
// .git is deliberately NOT denied: full reproducibility includes the
|
|
205
|
+
// repository metadata. Credentials inside it (e.g. remote URLs with
|
|
206
|
+
// embedded passwords in .git/config) are covered by the URL redaction
|
|
207
|
+
// pattern below and the credential-name denylist.
|
|
109
208
|
const DENIED_EXACT_NAMES = new Set([
|
|
110
|
-
".git",
|
|
111
209
|
".ssh",
|
|
112
|
-
"node_modules",
|
|
113
210
|
"keys",
|
|
114
211
|
"secrets",
|
|
115
212
|
".npmrc",
|
|
@@ -127,6 +224,9 @@ const SECRET_PATTERNS: Array<[RegExp, string]> = [
|
|
|
127
224
|
[/\bAKIA[0-9A-Z]{16}\b/g, "[REDACTED]"],
|
|
128
225
|
[/\b(?:sk|rk|pk)-(?:proj-)?[A-Za-z0-9_-]{16,}\b/g, "[REDACTED]"],
|
|
129
226
|
[/^([A-Z][A-Z0-9_]*(?:TOKEN|SECRET|PASSWORD|API_KEY)\s*=\s*)([^\s#]{6,})$/gim, "$1[REDACTED]"],
|
|
227
|
+
// Credentials embedded in URLs (git remotes in .git/config, pip
|
|
228
|
+
// indexes, ...): https://user:password@host -> https://user:[REDACTED]@host
|
|
229
|
+
[/\b(https?:\/\/[^\s\/@:#]+:)([^\s\/@]+)(@)/gi, "$1[REDACTED]$3"],
|
|
130
230
|
];
|
|
131
231
|
const PII_PATTERNS: Array<[RegExp, string]> = [
|
|
132
232
|
[/\b[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}\b/gi, "[REDACTED_PII]"],
|
|
@@ -197,89 +297,8 @@ function isBinary(buffer: Buffer): boolean {
|
|
|
197
297
|
return sample.length > 0 && suspicious / sample.length > 0.1;
|
|
198
298
|
}
|
|
199
299
|
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
* Supports `*` (within a segment), `**` (across segments, including a
|
|
203
|
-
* leading globstar-slash that matches zero directories), `?`, and
|
|
204
|
-
* matchBase semantics: a pattern without `/` matches the file name in
|
|
205
|
-
* any directory. Always dot-matches (gitignore semantics for `*`).
|
|
206
|
-
*/
|
|
207
|
-
export function ignoreMatch(path: string, rawPattern: string): boolean {
|
|
208
|
-
const pattern = rawPattern.replace(/^\//, "").replace(/\/$/, "/**").replace(/\/{2,}/g, "/");
|
|
209
|
-
if (!pattern) return false;
|
|
210
|
-
const hasSlash = pattern.includes("/");
|
|
211
|
-
let re = "^";
|
|
212
|
-
let i = 0;
|
|
213
|
-
while (i < pattern.length) {
|
|
214
|
-
const char = pattern[i];
|
|
215
|
-
if (char === "*") {
|
|
216
|
-
if (pattern[i + 1] === "*") {
|
|
217
|
-
let j = i;
|
|
218
|
-
while (pattern[j] === "*") j += 1;
|
|
219
|
-
if (pattern[j] === "/" && (i === 0 || pattern[i - 1] === "/")) {
|
|
220
|
-
// Leading or interior `**/` — zero or more whole directories.
|
|
221
|
-
re += "(?:[^/]+/)*";
|
|
222
|
-
i = j + 1;
|
|
223
|
-
continue;
|
|
224
|
-
}
|
|
225
|
-
re += ".*";
|
|
226
|
-
i = j;
|
|
227
|
-
continue;
|
|
228
|
-
}
|
|
229
|
-
re += "[^/]*";
|
|
230
|
-
i += 1;
|
|
231
|
-
continue;
|
|
232
|
-
}
|
|
233
|
-
if (char === "?") {
|
|
234
|
-
re += "[^/]";
|
|
235
|
-
i += 1;
|
|
236
|
-
continue;
|
|
237
|
-
}
|
|
238
|
-
re += char.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
239
|
-
i += 1;
|
|
240
|
-
}
|
|
241
|
-
const anchored = `${re}$`;
|
|
242
|
-
if (regexMatch(path, anchored)) return true;
|
|
243
|
-
// matchBase: pattern without a slash also matches the file name alone.
|
|
244
|
-
if (!hasSlash) {
|
|
245
|
-
const base = path.split("/").pop() ?? "";
|
|
246
|
-
return regexMatch(base, anchored);
|
|
247
|
-
}
|
|
248
|
-
return false;
|
|
249
|
-
}
|
|
250
|
-
|
|
251
|
-
function regexMatch(value: string, source: string): boolean {
|
|
252
|
-
try {
|
|
253
|
-
return new RegExp(source).test(value);
|
|
254
|
-
} catch {
|
|
255
|
-
return false;
|
|
256
|
-
}
|
|
257
|
-
}
|
|
258
|
-
|
|
259
|
-
function ignoredByRules(path: string, rules: string[]): boolean {
|
|
260
|
-
let ignored = false;
|
|
261
|
-
for (const raw of rules) {
|
|
262
|
-
const negated = raw.startsWith("!");
|
|
263
|
-
const body = negated ? raw.slice(1) : raw;
|
|
264
|
-
const pattern = body.replace(/^\//, "").replace(/\/$/, "/**");
|
|
265
|
-
if (!pattern) continue;
|
|
266
|
-
if (ignoreMatch(path, pattern)) ignored = !negated;
|
|
267
|
-
}
|
|
268
|
-
return ignored;
|
|
269
|
-
}
|
|
270
|
-
|
|
271
|
-
async function gitIgnoresPath(root: string, path: string): Promise<boolean> {
|
|
272
|
-
try {
|
|
273
|
-
await execFileAsync("git", ["-C", root, "check-ignore", "--no-index", "-q", "--", path], {
|
|
274
|
-
timeout: 5_000,
|
|
275
|
-
maxBuffer: 64 * 1024,
|
|
276
|
-
});
|
|
277
|
-
return true;
|
|
278
|
-
} catch {
|
|
279
|
-
// Exit status 1 means the path is not ignored. A non-git workspace also
|
|
280
|
-
// falls through here and is handled by the normal snapshot walker.
|
|
281
|
-
return false;
|
|
282
|
-
}
|
|
300
|
+
function comparePaths(left: string, right: string): number {
|
|
301
|
+
return left < right ? -1 : left > right ? 1 : 0;
|
|
283
302
|
}
|
|
284
303
|
|
|
285
304
|
export async function readSessionLedger(path: string | null): Promise<SessionLedger> {
|
|
@@ -309,10 +328,11 @@ export async function readSessionLedger(path: string | null): Promise<SessionLed
|
|
|
309
328
|
export function boundedTracePayload(
|
|
310
329
|
state: Pick<SessionState, "id" | "workspaceId" | "segment" | "resumed">,
|
|
311
330
|
events: TraceEvent[],
|
|
312
|
-
maxBytes =
|
|
331
|
+
maxBytes = MAX_SESSION_BYTES,
|
|
313
332
|
): Buffer {
|
|
314
333
|
const encode = (selected: TraceEvent[], truncated: boolean, includedCount = selected.length) => Buffer.from(redactCollectorText(JSON.stringify({
|
|
315
334
|
schema_version: 1,
|
|
335
|
+
schema: AGENT_TRACE_SCHEMA,
|
|
316
336
|
session_id: state.id,
|
|
317
337
|
workspace_id: state.workspaceId,
|
|
318
338
|
session_segment: state.segment,
|
|
@@ -345,26 +365,39 @@ export function boundedTracePayload(
|
|
|
345
365
|
return payload;
|
|
346
366
|
}
|
|
347
367
|
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
368
|
+
/**
|
|
369
|
+
* Select the longest prefix of events whose JSON fits maxBytes — the
|
|
370
|
+
* budget-wall behavior for traces. With the default 100 GiB budget this
|
|
371
|
+
* only fires at the wall; traces are otherwise never truncated.
|
|
372
|
+
*/
|
|
373
|
+
export function selectTraceEvents(
|
|
374
|
+
events: TraceEvent[],
|
|
375
|
+
maxBytes: number,
|
|
376
|
+
): { events: TraceEvent[]; truncated: boolean; droppedCount: number } {
|
|
377
|
+
if (events.length === 0) return { events: [], truncated: false, droppedCount: 0 };
|
|
378
|
+
const fits = (candidate: TraceEvent[]) => Buffer.byteLength(JSON.stringify(candidate)) <= maxBytes;
|
|
379
|
+
if (fits(events)) return { events, truncated: false, droppedCount: 0 };
|
|
380
|
+
const selected: TraceEvent[] = [];
|
|
381
|
+
for (const event of events) {
|
|
382
|
+
if (!fits([...selected, event])) break;
|
|
383
|
+
selected.push(event);
|
|
356
384
|
}
|
|
385
|
+
return { events: selected, truncated: true, droppedCount: events.length - selected.length };
|
|
357
386
|
}
|
|
358
387
|
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
388
|
+
/**
|
|
389
|
+
* FULL-WALK enumeration: every path in the workspace tree — regular
|
|
390
|
+
* files, symlinks (including links to directories; links are recorded,
|
|
391
|
+
* never followed, so cycles are impossible), and .git internals. The
|
|
392
|
+
* ONLY exclusions are the secrets-denylist paths. Git-aware listing was
|
|
393
|
+
* removed on purpose: `git ls-files` drops anything .gitignored (build
|
|
394
|
+
* outputs, node_modules), which broke the whole-project guarantee.
|
|
395
|
+
*/
|
|
396
|
+
async function listWorkspaceFiles(root: string): Promise<string[]> {
|
|
397
|
+
return walkWorkspace(root);
|
|
398
|
+
}
|
|
399
|
+
|
|
400
|
+
async function walkWorkspace(root: string, directory = root, output: string[] = []): Promise<string[]> {
|
|
368
401
|
let entries;
|
|
369
402
|
try {
|
|
370
403
|
entries = await readdir(directory, { withFileTypes: true });
|
|
@@ -372,33 +405,15 @@ async function walkFallback(root: string, directory = root, rules: string[] = []
|
|
|
372
405
|
return output;
|
|
373
406
|
}
|
|
374
407
|
for (const entry of entries) {
|
|
375
|
-
if (output.length >= MAX_FILES) break;
|
|
376
408
|
const fullPath = resolve(directory, entry.name);
|
|
377
409
|
const path = portablePath(root, fullPath);
|
|
378
|
-
if (!path || isCollectorPathDenied(path)
|
|
379
|
-
if (entry.isDirectory()) await
|
|
380
|
-
else
|
|
410
|
+
if (!path || isCollectorPathDenied(path)) continue;
|
|
411
|
+
if (entry.isDirectory()) await walkWorkspace(root, fullPath, output);
|
|
412
|
+
else output.push(path); // regular files AND symlinks (files or dirs)
|
|
381
413
|
}
|
|
382
414
|
return output;
|
|
383
415
|
}
|
|
384
416
|
|
|
385
|
-
async function listWorkspaceFiles(root: string): Promise<string[]> {
|
|
386
|
-
try {
|
|
387
|
-
const { stdout } = await execFileAsync("git", ["-C", root, "ls-files", "-co", "--exclude-standard", "-z"], {
|
|
388
|
-
encoding: "buffer",
|
|
389
|
-
maxBuffer: 16 * 1024 * 1024,
|
|
390
|
-
timeout: 15_000,
|
|
391
|
-
});
|
|
392
|
-
return Buffer.from(stdout)
|
|
393
|
-
.toString("utf8")
|
|
394
|
-
.split("\0")
|
|
395
|
-
.filter((path) => path && !isCollectorPathDenied(path))
|
|
396
|
-
.slice(0, MAX_FILES);
|
|
397
|
-
} catch {
|
|
398
|
-
return walkFallback(root);
|
|
399
|
-
}
|
|
400
|
-
}
|
|
401
|
-
|
|
402
417
|
async function gitMetadata(root: string): Promise<Record<string, string | null>> {
|
|
403
418
|
const git = async (...args: string[]) => {
|
|
404
419
|
try {
|
|
@@ -421,8 +436,11 @@ async function workspaceSignature(root: string): Promise<string> {
|
|
|
421
436
|
for (const path of await listWorkspaceFiles(root)) {
|
|
422
437
|
try {
|
|
423
438
|
const file = await lstat(resolve(root, path));
|
|
424
|
-
if (
|
|
425
|
-
|
|
439
|
+
if (file.isFile()) {
|
|
440
|
+
hash.update(path).update("\0").update(String(file.size)).update("\0").update(String(file.mtimeMs)).update("\0");
|
|
441
|
+
} else if (file.isSymbolicLink()) {
|
|
442
|
+
hash.update(path).update("\0").update("link").update("\0").update(String(file.mtimeMs)).update("\0");
|
|
443
|
+
}
|
|
426
444
|
} catch {
|
|
427
445
|
// A file can disappear while the workspace is being scanned.
|
|
428
446
|
}
|
|
@@ -430,6 +448,205 @@ async function workspaceSignature(root: string): Promise<string> {
|
|
|
430
448
|
return hash.digest("hex");
|
|
431
449
|
}
|
|
432
450
|
|
|
451
|
+
function sha256Hex(content: string): string {
|
|
452
|
+
return createHash("sha256").update(content).digest("hex");
|
|
453
|
+
}
|
|
454
|
+
|
|
455
|
+
/** Split a string into <= maxUnits slices without splitting surrogates. */
|
|
456
|
+
export function sliceString(content: string, maxUnits = MAX_SLICE_BYTES): string[] {
|
|
457
|
+
if (content.length <= maxUnits) return [content];
|
|
458
|
+
const slices: string[] = [];
|
|
459
|
+
let start = 0;
|
|
460
|
+
while (start < content.length) {
|
|
461
|
+
let end = Math.min(start + maxUnits, content.length);
|
|
462
|
+
if (end < content.length) {
|
|
463
|
+
const code = content.charCodeAt(end - 1);
|
|
464
|
+
if (code >= 0xd800 && code <= 0xdbff) end -= 1;
|
|
465
|
+
}
|
|
466
|
+
slices.push(content.slice(start, end));
|
|
467
|
+
start = end;
|
|
468
|
+
}
|
|
469
|
+
return slices;
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
/** Slice entry virtual path + manifest path for a logical file path. */
|
|
473
|
+
export function slicePartPath(path: string, index: number): string {
|
|
474
|
+
return `${path}.__agent_part${index}`;
|
|
475
|
+
}
|
|
476
|
+
export function sliceManifestPath(path: string): string {
|
|
477
|
+
return `${path}.__agent_manifest.json`;
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
/**
|
|
481
|
+
* Slice one logical file's content into <= MAX_SLICE_BYTES entries plus a
|
|
482
|
+
* manifest record (path, slice count, per-slice byte size + sha256) so
|
|
483
|
+
* replay reassembles the original deterministically. Small files pass
|
|
484
|
+
* through unchanged.
|
|
485
|
+
*/
|
|
486
|
+
export function buildFileEntries(file: CollectorFile): CollectorFile[] {
|
|
487
|
+
const content = file.content ?? "";
|
|
488
|
+
if (Buffer.byteLength(content) <= MAX_SLICE_BYTES) return [file];
|
|
489
|
+
const slices = sliceString(content);
|
|
490
|
+
const descriptors = slices.map((slice, index) => ({
|
|
491
|
+
index,
|
|
492
|
+
bytes: Buffer.byteLength(slice),
|
|
493
|
+
sha256: sha256Hex(slice),
|
|
494
|
+
}));
|
|
495
|
+
return [
|
|
496
|
+
{ path: sliceManifestPath(file.path), content: JSON.stringify(sliceManifest(file, descriptors)) },
|
|
497
|
+
...slices.map((slice, index) => ({
|
|
498
|
+
path: slicePartPath(file.path, index),
|
|
499
|
+
content: slice,
|
|
500
|
+
...(file.encoding ? { encoding: file.encoding } : {}),
|
|
501
|
+
})),
|
|
502
|
+
];
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
/**
|
|
506
|
+
* Manifest for a sliced file. `slice_count` is the GLOBAL count and each
|
|
507
|
+
* entry carries its global `index`, so manifests arriving across
|
|
508
|
+
* different parts union deterministically: a replay collects the part
|
|
509
|
+
* entries by index (verifying per-slice sha256) until all
|
|
510
|
+
* `slice_count` indexes are present, then reassembles.
|
|
511
|
+
*/
|
|
512
|
+
function sliceManifest(
|
|
513
|
+
file: { path: string; encoding?: "base64"; mode?: string },
|
|
514
|
+
sliceDescriptors: Array<{ index: number; bytes: number; sha256: string }>,
|
|
515
|
+
totalCount?: number,
|
|
516
|
+
): Record<string, unknown> {
|
|
517
|
+
return {
|
|
518
|
+
schema_version: 1,
|
|
519
|
+
path: file.path,
|
|
520
|
+
encoding: file.encoding ?? null,
|
|
521
|
+
mode: file.mode ?? null,
|
|
522
|
+
slice_count: totalCount ?? sliceDescriptors.length,
|
|
523
|
+
slices: sliceDescriptors,
|
|
524
|
+
};
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
type ContentChunk = { content: string; encoding?: "base64" };
|
|
528
|
+
|
|
529
|
+
/**
|
|
530
|
+
* Read a regular file into <= ~8 MiB content chunks: small files in one
|
|
531
|
+
* whole read (fully redacted), big files streamed in fixed raw chunks —
|
|
532
|
+
* strings stay far below the runtime string limit regardless of file
|
|
533
|
+
* size. Binary detection happens on the first chunk and applies to the
|
|
534
|
+
* whole file.
|
|
535
|
+
*/
|
|
536
|
+
async function readContentChunks(absolute: string, size: number): Promise<ContentChunk[]> {
|
|
537
|
+
if (size <= WHOLE_READ_BYTES) {
|
|
538
|
+
const buffer = await readFile(absolute);
|
|
539
|
+
if (isBinary(buffer)) {
|
|
540
|
+
return [{ content: buffer.toString("base64"), encoding: "base64" }];
|
|
541
|
+
}
|
|
542
|
+
return [{ content: redactCollectorText(buffer.toString("utf8")).text }];
|
|
543
|
+
}
|
|
544
|
+
const { open } = await import("node:fs/promises");
|
|
545
|
+
const handle = await open(absolute, "r");
|
|
546
|
+
try {
|
|
547
|
+
const chunkBytes = 8 * 1024 * 1024;
|
|
548
|
+
const buffer = Buffer.alloc(chunkBytes);
|
|
549
|
+
const first = await handle.read(buffer, 0, chunkBytes, null);
|
|
550
|
+
const binary = isBinary(buffer.subarray(0, first.bytesRead));
|
|
551
|
+
const chunks: ContentChunk[] = [];
|
|
552
|
+
if (binary) {
|
|
553
|
+
// Binaries carry no redactable text: base64 each raw chunk.
|
|
554
|
+
chunks.push({ content: buffer.subarray(0, first.bytesRead).toString("base64"), encoding: "base64" });
|
|
555
|
+
let read = 0;
|
|
556
|
+
while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
|
|
557
|
+
chunks.push({ content: buffer.subarray(0, read).toString("base64"), encoding: "base64" });
|
|
558
|
+
}
|
|
559
|
+
} else {
|
|
560
|
+
// Text: decode incrementally (StringDecoder absorbs multibyte
|
|
561
|
+
// sequences split across chunk boundaries), emit only up to the
|
|
562
|
+
// last complete line, carry the remainder, and redact each
|
|
563
|
+
// line-aligned segment.
|
|
564
|
+
const { StringDecoder } = await import("node:string_decoder");
|
|
565
|
+
const decoder = new StringDecoder("utf8");
|
|
566
|
+
let pending = decoder.write(buffer.subarray(0, first.bytesRead));
|
|
567
|
+
let read = 0;
|
|
568
|
+
while ((read = (await handle.read(buffer, 0, chunkBytes, null)).bytesRead) > 0) {
|
|
569
|
+
pending += decoder.write(buffer.subarray(0, read));
|
|
570
|
+
const lastNl = pending.lastIndexOf("\n");
|
|
571
|
+
if (lastNl >= 0) {
|
|
572
|
+
chunks.push({ content: redactCollectorText(pending.slice(0, lastNl + 1)).text });
|
|
573
|
+
pending = pending.slice(lastNl + 1);
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
pending += decoder.end();
|
|
577
|
+
if (pending) chunks.push({ content: redactCollectorText(pending).text });
|
|
578
|
+
}
|
|
579
|
+
return chunks;
|
|
580
|
+
} finally {
|
|
581
|
+
await handle.close();
|
|
582
|
+
}
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
/**
|
|
586
|
+
* Read one workspace entry into CollectorFile entries. Regular files:
|
|
587
|
+
* text is redacted, binaries base64-encoded ("encoding": "base64"), and
|
|
588
|
+
* the POSIX mode rides along (octal string) so replay can restore the
|
|
589
|
+
* executable bit (stored-and-ignored on Windows). Content bigger than
|
|
590
|
+
* MAX_SLICE_BYTES is sliced (see buildFileEntries). Symlinks become
|
|
591
|
+
* records { path, target } (target verbatim; replay recreates the link).
|
|
592
|
+
* Returns [] when the entry vanished or is a special file.
|
|
593
|
+
*/
|
|
594
|
+
async function readWorkspaceEntries(root: string, path: string): Promise<CollectorFile[]> {
|
|
595
|
+
try {
|
|
596
|
+
const absolute = resolve(root, path);
|
|
597
|
+
if (portablePath(root, absolute).startsWith("../")) return [];
|
|
598
|
+
const file = await lstat(absolute);
|
|
599
|
+
if (file.isSymbolicLink()) {
|
|
600
|
+
const target = await readlink(absolute);
|
|
601
|
+
return [{ path, target }];
|
|
602
|
+
}
|
|
603
|
+
if (!file.isFile()) return [];
|
|
604
|
+
const mode = file.mode & 0o777;
|
|
605
|
+
const chunks: ContentChunk[] = await readContentChunks(absolute, file.size);
|
|
606
|
+
if (chunks.length === 1) {
|
|
607
|
+
const chunk = chunks[0];
|
|
608
|
+
return buildFileEntries({
|
|
609
|
+
path,
|
|
610
|
+
content: chunk.content,
|
|
611
|
+
...(chunk.encoding ? { encoding: chunk.encoding } : {}),
|
|
612
|
+
...(mode ? { mode: mode.toString(8) } : {}),
|
|
613
|
+
});
|
|
614
|
+
}
|
|
615
|
+
// Big file: streamed chunks -> one logical entry per chunk (already
|
|
616
|
+
// <= ~8 MiB for raw chunks; base64 inflates 4/3, so sub-slice those),
|
|
617
|
+
// plus the manifest for deterministic reassembly.
|
|
618
|
+
const encoding = chunks[0].encoding;
|
|
619
|
+
const entries: CollectorFile[] = [];
|
|
620
|
+
const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
|
|
621
|
+
entries.push({
|
|
622
|
+
path: sliceManifestPath(path),
|
|
623
|
+
content: JSON.stringify(sliceManifest(
|
|
624
|
+
{
|
|
625
|
+
path,
|
|
626
|
+
...(encoding ? { encoding } : {}),
|
|
627
|
+
...(mode ? { mode: mode.toString(8) } : {}),
|
|
628
|
+
},
|
|
629
|
+
sliceContents.map((slice, index) => ({
|
|
630
|
+
index,
|
|
631
|
+
bytes: Buffer.byteLength(slice),
|
|
632
|
+
sha256: sha256Hex(slice),
|
|
633
|
+
})),
|
|
634
|
+
)),
|
|
635
|
+
});
|
|
636
|
+
sliceContents.forEach((slice, index) => {
|
|
637
|
+
entries.push({
|
|
638
|
+
path: slicePartPath(path, index),
|
|
639
|
+
content: slice,
|
|
640
|
+
...(encoding ? { encoding } : {}),
|
|
641
|
+
});
|
|
642
|
+
});
|
|
643
|
+
return entries;
|
|
644
|
+
} catch {
|
|
645
|
+
// Workspaces are live; races are expected and retried by the next snapshot.
|
|
646
|
+
return [];
|
|
647
|
+
}
|
|
648
|
+
}
|
|
649
|
+
|
|
433
650
|
export async function collectFiles(
|
|
434
651
|
root: string,
|
|
435
652
|
byteLimit: number,
|
|
@@ -448,25 +665,17 @@ export async function collectFiles(
|
|
|
448
665
|
root_name: root.split(sep).filter(Boolean).at(-1) ?? "workspace",
|
|
449
666
|
git: await gitMetadata(root),
|
|
450
667
|
});
|
|
451
|
-
files.push({ path:
|
|
668
|
+
files.push({ path: `${AGENT_DIR}/workspace.json`, content: metadata });
|
|
452
669
|
used += Buffer.byteLength(metadata);
|
|
453
670
|
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
const file = await lstat(absolute);
|
|
460
|
-
if (!file.isFile() || file.isSymbolicLink() || file.size > MAX_COLLECTOR_FILE_BYTES) continue;
|
|
461
|
-
const buffer = await readFile(absolute);
|
|
462
|
-
if (buffer.length > MAX_COLLECTOR_FILE_BYTES || isBinary(buffer)) continue;
|
|
463
|
-
const redacted = redactCollectorText(buffer.toString("utf8")).text;
|
|
464
|
-
const size = Buffer.byteLength(redacted);
|
|
671
|
+
const paths = (await listWorkspaceFiles(root)).sort(comparePaths);
|
|
672
|
+
for (const path of paths) {
|
|
673
|
+
if (used >= byteLimit) break;
|
|
674
|
+
for (const file of await readWorkspaceEntries(root, path)) {
|
|
675
|
+
const size = Buffer.byteLength(file.content ?? file.target ?? "");
|
|
465
676
|
if (used + size > byteLimit) continue;
|
|
466
|
-
files.push(
|
|
677
|
+
files.push(file);
|
|
467
678
|
used += size;
|
|
468
|
-
} catch {
|
|
469
|
-
// Workspaces are live; races are expected and retried by the next snapshot.
|
|
470
679
|
}
|
|
471
680
|
}
|
|
472
681
|
return files;
|
|
@@ -505,6 +714,101 @@ export function buildEnvelopePayload(
|
|
|
505
714
|
}));
|
|
506
715
|
}
|
|
507
716
|
|
|
717
|
+
export type EnvelopePart = {
|
|
718
|
+
files: CollectorFile[];
|
|
719
|
+
payload: Buffer;
|
|
720
|
+
compressed: Buffer;
|
|
721
|
+
};
|
|
722
|
+
|
|
723
|
+
/**
|
|
724
|
+
* Deterministic part split for a full files array (used by `omnirush
|
|
725
|
+
* collect` and the tests): sort files by path, accumulate in order, keep
|
|
726
|
+
* each part <= maxCompressed compressed. A single file bigger than one
|
|
727
|
+
* part gets its own part; file content is never split across parts. A
|
|
728
|
+
* lone file that cannot fit even the hard rail is skipped with a warning.
|
|
729
|
+
* Envelope sequence numbers are state.sequence + 1 + partIndex.
|
|
730
|
+
*/
|
|
731
|
+
export function buildEnvelopeParts(
|
|
732
|
+
state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
|
|
733
|
+
snapshotType: SnapshotType,
|
|
734
|
+
files: CollectorFile[],
|
|
735
|
+
maxCompressed = MAX_PART_COMPRESSED_BYTES,
|
|
736
|
+
log?: (message: string, attributes?: Record<string, unknown>) => void,
|
|
737
|
+
): Promise<EnvelopePart[]> {
|
|
738
|
+
// Slice any entry whose content exceeds MAX_SLICE_BYTES (manifest +
|
|
739
|
+
// ordered parts) BEFORE packing, so callers feeding raw file lists get
|
|
740
|
+
// the same deterministic slice layout as the tree walker.
|
|
741
|
+
const sorted = files
|
|
742
|
+
.flatMap((file) => buildFileEntries(file))
|
|
743
|
+
.sort((left, right) => comparePaths(left.path, right.path));
|
|
744
|
+
return packEnvelopeParts(state, snapshotType, sorted, maxCompressed, log);
|
|
745
|
+
}
|
|
746
|
+
|
|
747
|
+
/**
|
|
748
|
+
* Shared greedy packer: split an ordered file list into compressed parts,
|
|
749
|
+
* preserving path order across parts. Used by buildEnvelopeParts (pure,
|
|
750
|
+
* whole list in memory) and mirrored by the collector's streaming packer
|
|
751
|
+
* (packAndUpload, which never holds more than one part in memory).
|
|
752
|
+
*/
|
|
753
|
+
export async function packEnvelopeParts(
|
|
754
|
+
state: Pick<SessionState, "id" | "segment" | "resumed"> & { sequence?: number },
|
|
755
|
+
snapshotType: SnapshotType,
|
|
756
|
+
sortedFiles: CollectorFile[],
|
|
757
|
+
maxCompressed = MAX_PART_COMPRESSED_BYTES,
|
|
758
|
+
log?: (message: string, attributes?: Record<string, unknown>) => void,
|
|
759
|
+
): Promise<EnvelopePart[]> {
|
|
760
|
+
const hard = Math.max(maxCompressed, MAX_PART_COMPRESSED_HARD);
|
|
761
|
+
const measure = async (batch: CollectorFile[], partIndex: number): Promise<EnvelopePart> => {
|
|
762
|
+
const payload = buildEnvelopePayload({ ...state, sequence: (state.sequence ?? 0) + partIndex }, snapshotType, batch);
|
|
763
|
+
return { files: batch, payload, compressed: await compressZstd(payload) };
|
|
764
|
+
};
|
|
765
|
+
const parts: EnvelopePart[] = [];
|
|
766
|
+
let batch: CollectorFile[] = [];
|
|
767
|
+
const fit = async (batchToFit: CollectorFile[], partIndex: number) => {
|
|
768
|
+
let measured = await measure(batchToFit, partIndex);
|
|
769
|
+
const carry: CollectorFile[] = [];
|
|
770
|
+
// Split at the compressed target; a lone file may use the body
|
|
771
|
+
// headroom up to the hard rail, beyond which it is skipped loudly
|
|
772
|
+
// rather than split mid-content.
|
|
773
|
+
while (measured.compressed.length > maxCompressed && batchToFit.length > 1) {
|
|
774
|
+
carry.unshift(batchToFit.pop()!);
|
|
775
|
+
measured = await measure(batchToFit, partIndex);
|
|
776
|
+
}
|
|
777
|
+
if (measured.compressed.length > hard && batchToFit.length === 1) {
|
|
778
|
+
const lone = batchToFit.pop()!;
|
|
779
|
+
log?.("OmniRush file exceeds the upload limits and cannot be shipped as a single part; skipped", {
|
|
780
|
+
path: lone.path,
|
|
781
|
+
bytes: Buffer.byteLength(lone.content ?? lone.target ?? ""),
|
|
782
|
+
});
|
|
783
|
+
return { measured: null, carry };
|
|
784
|
+
}
|
|
785
|
+
return { measured, carry };
|
|
786
|
+
};
|
|
787
|
+
for (const file of sortedFiles) {
|
|
788
|
+
batch.push(file);
|
|
789
|
+
const partIndex = parts.length;
|
|
790
|
+
const estimate = batch.reduce((total, f) => total + Buffer.byteLength(f.content), 0);
|
|
791
|
+
if (estimate < PART_UNCOMPRESSED_STEP) continue;
|
|
792
|
+
const { measured, carry } = await fit(batch, partIndex);
|
|
793
|
+
if (measured
|
|
794
|
+
&& (measured.compressed.length >= PART_READY_COMPRESSED_BYTES || estimate >= PART_BATCH_UNCOMPRESSED_MAX)) {
|
|
795
|
+
parts.push(measured);
|
|
796
|
+
batch = carry;
|
|
797
|
+
} else if (measured) {
|
|
798
|
+
batch = [...batch, ...carry];
|
|
799
|
+
} else {
|
|
800
|
+
batch = carry;
|
|
801
|
+
}
|
|
802
|
+
}
|
|
803
|
+
while (batch.length > 0) {
|
|
804
|
+
const { measured, carry } = await fit(batch, parts.length);
|
|
805
|
+
if (measured) parts.push(measured);
|
|
806
|
+
if (carry.length === 0) break;
|
|
807
|
+
batch = carry;
|
|
808
|
+
}
|
|
809
|
+
return parts;
|
|
810
|
+
}
|
|
811
|
+
|
|
508
812
|
export class WorkspaceCollector {
|
|
509
813
|
private readonly collectUrl: string | null;
|
|
510
814
|
private token: string;
|
|
@@ -519,6 +823,9 @@ export class WorkspaceCollector {
|
|
|
519
823
|
private readonly sessions = new Map<string, SessionState>();
|
|
520
824
|
private readonly changeDebounceMs: number;
|
|
521
825
|
private readonly fallbackScanMs: number;
|
|
826
|
+
private readonly sessionBudgetBytes: number;
|
|
827
|
+
private readonly partCompressedBytes: number;
|
|
828
|
+
private readonly partCheckBytes: number;
|
|
522
829
|
|
|
523
830
|
constructor(options: CollectorOptions = {}) {
|
|
524
831
|
this.collectUrl = resolveCollectUrl(options.gatewayUrl ?? process.env.OMNIRUSH_GATEWAY_URL);
|
|
@@ -531,6 +838,12 @@ export class WorkspaceCollector {
|
|
|
531
838
|
this.ledgerReady = this.loadLedger();
|
|
532
839
|
this.changeDebounceMs = options.changeDebounceMs ?? CHANGE_DEBOUNCE_MS;
|
|
533
840
|
this.fallbackScanMs = options.fallbackScanMs ?? FALLBACK_SCAN_MS;
|
|
841
|
+
this.sessionBudgetBytes = Math.max(1024, options.sessionBudgetBytes ?? MAX_SESSION_BYTES);
|
|
842
|
+
this.partCompressedBytes = Math.min(
|
|
843
|
+
Math.max(1024, options.partCompressedBytes ?? MAX_PART_COMPRESSED_BYTES),
|
|
844
|
+
MAX_PART_COMPRESSED_HARD,
|
|
845
|
+
);
|
|
846
|
+
this.partCheckBytes = Math.max(1024, options.partCheckBytes ?? PART_UNCOMPRESSED_STEP);
|
|
534
847
|
}
|
|
535
848
|
|
|
536
849
|
private async loadLedger(): Promise<void> {
|
|
@@ -623,12 +936,14 @@ export class WorkspaceCollector {
|
|
|
623
936
|
lastSignature: "",
|
|
624
937
|
started: false,
|
|
625
938
|
finished: false,
|
|
939
|
+
budgetExhausted: false,
|
|
626
940
|
changeTimer: null,
|
|
627
941
|
scanTimer: null,
|
|
628
942
|
watcher: null,
|
|
943
|
+
watcherStartedAtMs: 0,
|
|
944
|
+
knownPaths: new Set(),
|
|
629
945
|
trace: [],
|
|
630
|
-
changeJournal:
|
|
631
|
-
changeJournalBytes: 0,
|
|
946
|
+
changeJournal: [],
|
|
632
947
|
changeCaptureTail: Promise.resolve(),
|
|
633
948
|
ready: Promise.resolve(),
|
|
634
949
|
tail: Promise.resolve(),
|
|
@@ -637,11 +952,14 @@ export class WorkspaceCollector {
|
|
|
637
952
|
this.sessions.set(sessionId, state);
|
|
638
953
|
this.enqueue(state, async () => {
|
|
639
954
|
await state.ready;
|
|
955
|
+
const baselinePaths = await listWorkspaceFiles(root);
|
|
956
|
+
state.knownPaths = new Set(baselinePaths);
|
|
640
957
|
state.lastSignature = await workspaceSignature(root);
|
|
641
958
|
await this.uploadWorkspace(state, "start");
|
|
642
959
|
state.started = true;
|
|
643
960
|
});
|
|
644
961
|
try {
|
|
962
|
+
state.watcherStartedAtMs = Date.now();
|
|
645
963
|
state.watcher = watch(root, { recursive: true }, (_event, filename) => {
|
|
646
964
|
if (filename && isCollectorPathDenied(String(filename))) return;
|
|
647
965
|
if (filename) this.queueChangedPath(state, String(filename));
|
|
@@ -665,9 +983,16 @@ export class WorkspaceCollector {
|
|
|
665
983
|
|
|
666
984
|
recordTrace(sessionId: string, type: string, data?: unknown): void {
|
|
667
985
|
const state = this.sessions.get(sessionId);
|
|
668
|
-
if (!state || state.finished) return;
|
|
986
|
+
if (!state || state.finished || state.budgetExhausted) return;
|
|
669
987
|
state.trace.push({ at: new Date().toISOString(), type, ...(data === undefined ? {} : { data }) });
|
|
670
|
-
if (state.trace.length
|
|
988
|
+
if (state.trace.length >= TRACE_RELIEF_EVENTS) {
|
|
989
|
+
// Memory relief, NOT truncation: the whole batch is uploaded.
|
|
990
|
+
const batch = state.trace.splice(0);
|
|
991
|
+
this.enqueue(state, async () => {
|
|
992
|
+
await state.ready;
|
|
993
|
+
await this.uploadTrace(state, batch);
|
|
994
|
+
});
|
|
995
|
+
}
|
|
671
996
|
}
|
|
672
997
|
|
|
673
998
|
finishSession(sessionId: string, finalTrace?: unknown): void {
|
|
@@ -707,12 +1032,37 @@ export class WorkspaceCollector {
|
|
|
707
1032
|
await Promise.allSettled([...this.sessions.values()].map((state) => state.tail));
|
|
708
1033
|
}
|
|
709
1034
|
|
|
1035
|
+
private stopWatching(state: SessionState): void {
|
|
1036
|
+
if (state.changeTimer) {
|
|
1037
|
+
clearTimeout(state.changeTimer);
|
|
1038
|
+
state.changeTimer = null;
|
|
1039
|
+
}
|
|
1040
|
+
if (state.scanTimer) {
|
|
1041
|
+
clearInterval(state.scanTimer);
|
|
1042
|
+
state.scanTimer = null;
|
|
1043
|
+
}
|
|
1044
|
+
state.watcher?.close();
|
|
1045
|
+
state.watcher = null;
|
|
1046
|
+
}
|
|
1047
|
+
|
|
1048
|
+
private exhaustSessionBudget(state: SessionState): void {
|
|
1049
|
+
if (state.budgetExhausted) return;
|
|
1050
|
+
state.budgetExhausted = true;
|
|
1051
|
+
this.stopWatching(state);
|
|
1052
|
+
this.log("warn", "OmniRush session budget exhausted — collection stopped for this session", {
|
|
1053
|
+
sessionId: state.id,
|
|
1054
|
+
budgetBytes: this.sessionBudgetBytes,
|
|
1055
|
+
sentBytes: state.sentBytes,
|
|
1056
|
+
});
|
|
1057
|
+
}
|
|
1058
|
+
|
|
710
1059
|
private scheduleChange(state: SessionState): void {
|
|
711
|
-
if (state.finished) return;
|
|
1060
|
+
if (state.finished || state.budgetExhausted) return;
|
|
712
1061
|
if (state.changeTimer) clearTimeout(state.changeTimer);
|
|
713
1062
|
state.changeTimer = setTimeout(() => {
|
|
714
1063
|
state.changeTimer = null;
|
|
715
1064
|
this.enqueue(state, async () => {
|
|
1065
|
+
if (state.budgetExhausted || state.finished) return;
|
|
716
1066
|
const signature = await workspaceSignature(state.root);
|
|
717
1067
|
if (!signature || signature === state.lastSignature) return;
|
|
718
1068
|
state.lastSignature = signature;
|
|
@@ -723,7 +1073,7 @@ export class WorkspaceCollector {
|
|
|
723
1073
|
}
|
|
724
1074
|
|
|
725
1075
|
private queueChangedPath(state: SessionState, filename: string): void {
|
|
726
|
-
if (state.finished) return;
|
|
1076
|
+
if (state.finished || state.budgetExhausted) return;
|
|
727
1077
|
state.changeCaptureTail = state.changeCaptureTail
|
|
728
1078
|
.catch(() => undefined)
|
|
729
1079
|
.then(() => this.captureChangedPath(state, filename))
|
|
@@ -736,51 +1086,110 @@ export class WorkspaceCollector {
|
|
|
736
1086
|
});
|
|
737
1087
|
}
|
|
738
1088
|
|
|
739
|
-
private async captureChangedPath(
|
|
1089
|
+
private async captureChangedPath(
|
|
1090
|
+
state: SessionState,
|
|
1091
|
+
filename: string,
|
|
1092
|
+
options?: { force?: boolean },
|
|
1093
|
+
): Promise<void> {
|
|
740
1094
|
const absolute = resolve(state.root, filename);
|
|
741
1095
|
const path = portablePath(state.root, absolute);
|
|
742
|
-
if (!path || path.startsWith("../") || isCollectorPathDenied(path)
|
|
1096
|
+
if (!path || path.startsWith("../") || isCollectorPathDenied(path)) return;
|
|
1097
|
+
const records: ChangeJournalEntry[] = [];
|
|
743
1098
|
let entry: ChangeJournalEntry;
|
|
744
1099
|
try {
|
|
745
1100
|
const file = await lstat(absolute);
|
|
746
|
-
|
|
747
|
-
|
|
1101
|
+
// Pre-baseline activity: the start snapshot already carries this
|
|
1102
|
+
// state, and macOS FSEvents (unlike inotify) delivers events for
|
|
1103
|
+
// files created just before watch() began — often after the start
|
|
1104
|
+
// upload finished. The mtime is compared at millisecond
|
|
1105
|
+
// resolution: sub-ms APFS timestamps make a pre-watch write's
|
|
1106
|
+
// mtime land a few microseconds after Date.now()'s truncated
|
|
1107
|
+
// stamp, which must still count as pre-baseline. Deletions stay
|
|
1108
|
+
// unguarded (replaying a delete of a baseline file the user
|
|
1109
|
+
// removed mid-session is required).
|
|
1110
|
+
if (!options?.force && Math.floor(file.mtimeMs) <= state.watcherStartedAtMs) return;
|
|
1111
|
+
state.knownPaths.add(path);
|
|
1112
|
+
if (file.isSymbolicLink()) {
|
|
1113
|
+
// Symlink record: path + target so replay recreates the link.
|
|
1114
|
+
const target = await readlink(absolute);
|
|
1115
|
+
entry = { path, at: new Date().toISOString(), status: "present", target };
|
|
1116
|
+
} else if (!file.isFile()) {
|
|
1117
|
+
return;
|
|
748
1118
|
} else {
|
|
749
|
-
const
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
1119
|
+
const at = new Date().toISOString();
|
|
1120
|
+
const mode = file.mode & 0o777;
|
|
1121
|
+
const modeField = mode ? { mode: mode.toString(8) } : {};
|
|
1122
|
+
const chunks = await readContentChunks(absolute, file.size);
|
|
1123
|
+
const encoding = chunks[0].encoding;
|
|
1124
|
+
const sliceContents = chunks.flatMap((chunk) => sliceString(chunk.content));
|
|
1125
|
+
if (sliceContents.length === 1) {
|
|
753
1126
|
entry = {
|
|
754
1127
|
path,
|
|
755
|
-
at
|
|
1128
|
+
at,
|
|
756
1129
|
status: "present",
|
|
757
|
-
content:
|
|
1130
|
+
content: sliceContents[0],
|
|
1131
|
+
...(encoding ? { encoding } : {}),
|
|
1132
|
+
...modeField,
|
|
758
1133
|
};
|
|
1134
|
+
} else {
|
|
1135
|
+
// Big file: one ordered record per slice so the journal stays
|
|
1136
|
+
// replayable; reassembly rides the slice metadata.
|
|
1137
|
+
sliceContents.forEach((content, index) => {
|
|
1138
|
+
records.push({
|
|
1139
|
+
path,
|
|
1140
|
+
at,
|
|
1141
|
+
status: "present",
|
|
1142
|
+
content,
|
|
1143
|
+
...(encoding ? { encoding } : {}),
|
|
1144
|
+
...modeField,
|
|
1145
|
+
slice: {
|
|
1146
|
+
index,
|
|
1147
|
+
total: sliceContents.length,
|
|
1148
|
+
bytes: Buffer.byteLength(content),
|
|
1149
|
+
sha256: sha256Hex(content),
|
|
1150
|
+
},
|
|
1151
|
+
});
|
|
1152
|
+
});
|
|
1153
|
+
state.changeJournal.push(...records);
|
|
1154
|
+
return;
|
|
759
1155
|
}
|
|
760
1156
|
}
|
|
761
|
-
} catch {
|
|
1157
|
+
} catch (error) {
|
|
1158
|
+
// Only a real disappearance is a deletion. Any other lstat failure
|
|
1159
|
+
// (EPERM/EBUSY/... under heavy watcher load) must not fabricate a
|
|
1160
|
+
// deleted record for a file that still exists — the fallback scan
|
|
1161
|
+
// and the end snapshot re-check those paths.
|
|
1162
|
+
if ((error as any)?.code !== "ENOENT") return;
|
|
1163
|
+
// Only paths the baseline (or a captured change) saw can be
|
|
1164
|
+
// deleted inside the workspace; FSEvents sibling/parent leaks
|
|
1165
|
+
// reference foreign paths that must not fabricate deletions.
|
|
1166
|
+
if (!state.knownPaths.has(path)) return;
|
|
762
1167
|
entry = { path, at: new Date().toISOString(), status: "deleted" };
|
|
763
1168
|
}
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
state.
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
state.changeJournal.
|
|
1169
|
+
// Consecutive-duplicate suppression: macOS FSEvents can deliver a
|
|
1170
|
+
// late event for a path that was also captured directly, and an
|
|
1171
|
+
// identical re-capture adds upload fat without a state change. The
|
|
1172
|
+
// replay result is identical either way.
|
|
1173
|
+
const journalSignature = (record: ChangeJournalEntry) =>
|
|
1174
|
+
JSON.stringify([record.status, record.content, record.encoding, record.target, record.mode]);
|
|
1175
|
+
const pendingRecords = [entry, ...records];
|
|
1176
|
+
for (const record of pendingRecords) {
|
|
1177
|
+
let duplicate = false;
|
|
1178
|
+
for (let i = state.changeJournal.length - 1, scanned = 0; i >= 0 && scanned < 100; i--, scanned++) {
|
|
1179
|
+
const prior = state.changeJournal[i];
|
|
1180
|
+
if (prior.path !== record.path) continue;
|
|
1181
|
+
duplicate = journalSignature(prior) === journalSignature(record);
|
|
1182
|
+
break;
|
|
1183
|
+
}
|
|
1184
|
+
if (!duplicate) state.changeJournal.push(record);
|
|
774
1185
|
}
|
|
775
1186
|
}
|
|
776
1187
|
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
}
|
|
783
|
-
}
|
|
1188
|
+
/** Remove exactly the uploaded entries from the ordered journal. */
|
|
1189
|
+
private acknowledgeJournal(state: SessionState, uploaded: ChangeJournalEntry[]): void {
|
|
1190
|
+
if (uploaded.length === 0) return;
|
|
1191
|
+
const sent = uploaded.length > 32 ? new Set(uploaded) : null;
|
|
1192
|
+
state.changeJournal = state.changeJournal.filter((entry) => sent ? !sent.has(entry) : !uploaded.includes(entry));
|
|
784
1193
|
}
|
|
785
1194
|
|
|
786
1195
|
private enqueue(state: SessionState, operation: () => Promise<void>): void {
|
|
@@ -792,41 +1201,342 @@ export class WorkspaceCollector {
|
|
|
792
1201
|
});
|
|
793
1202
|
}
|
|
794
1203
|
|
|
795
|
-
private async uploadWorkspace(state: SessionState, type:
|
|
796
|
-
|
|
797
|
-
if (
|
|
798
|
-
|
|
799
|
-
|
|
800
|
-
|
|
801
|
-
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
|
|
805
|
-
|
|
1204
|
+
private async uploadWorkspace(state: SessionState, type: "start" | "change" | "end"): Promise<void> {
|
|
1205
|
+
if (state.budgetExhausted) return;
|
|
1206
|
+
if (type === "change") {
|
|
1207
|
+
// Change snapshots carry ONLY the touched files from the journal —
|
|
1208
|
+
// never a full re-enumeration of the tree.
|
|
1209
|
+
const snapshot = [...state.changeJournal];
|
|
1210
|
+
if (snapshot.length === 0) {
|
|
1211
|
+
state.lastSignature = await workspaceSignature(state.root);
|
|
1212
|
+
return;
|
|
1213
|
+
}
|
|
1214
|
+
const uploaded = await this.uploadChangeParts(state, snapshot);
|
|
1215
|
+
// Acknowledge only after every part of the snapshot succeeded; on a
|
|
1216
|
+
// mid-budget exhaustion, acknowledge only what actually shipped.
|
|
1217
|
+
this.acknowledgeJournal(state, state.budgetExhausted ? uploaded.flat() : snapshot);
|
|
1218
|
+
} else {
|
|
1219
|
+
await this.uploadTreeParts(state, type);
|
|
1220
|
+
}
|
|
1221
|
+
state.lastSignature = await workspaceSignature(state.root);
|
|
1222
|
+
}
|
|
1223
|
+
|
|
1224
|
+
private async workspaceMetadataFile(state: SessionState): Promise<CollectorFile> {
|
|
1225
|
+
return {
|
|
1226
|
+
path: `${AGENT_DIR}/workspace.json`,
|
|
1227
|
+
content: JSON.stringify({
|
|
1228
|
+
workspace_id: state.workspaceId,
|
|
1229
|
+
session_id: state.id,
|
|
1230
|
+
session_segment: state.segment,
|
|
1231
|
+
session_resumed: state.resumed,
|
|
1232
|
+
root_name: state.root.split(sep).filter(Boolean).at(-1) ?? "workspace",
|
|
1233
|
+
git: await gitMetadata(state.root),
|
|
1234
|
+
}),
|
|
1235
|
+
};
|
|
1236
|
+
}
|
|
1237
|
+
|
|
1238
|
+
/** Full-tree snapshot ("start" | "end"), streamed into ordered parts. */
|
|
1239
|
+
private async uploadTreeParts(state: SessionState, snapshotType: "start" | "end"): Promise<void> {
|
|
1240
|
+
const paths = (await listWorkspaceFiles(state.root)).sort(comparePaths);
|
|
1241
|
+
const metadata = await this.workspaceMetadataFile(state);
|
|
1242
|
+
let metadataSent = false;
|
|
1243
|
+
const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
|
|
1244
|
+
const measure = async (batch: CollectorFile[]) => {
|
|
1245
|
+
const files = metadataSent ? batch : [metadata, ...batch];
|
|
1246
|
+
const payload = buildEnvelopePayload(bySequence(), snapshotType, files);
|
|
1247
|
+
return { files, payload, compressed: await compressZstd(payload) };
|
|
1248
|
+
};
|
|
1249
|
+
await this.packAndUpload(state, snapshotType, this.iterTreeFiles(state, paths), {
|
|
1250
|
+
sizeOf: (file) => Buffer.byteLength(file.content ?? file.target ?? "") + file.path.length + 64,
|
|
1251
|
+
measure,
|
|
1252
|
+
upload: async (measured) => {
|
|
1253
|
+
const ok = await this.uploadEnvelope(state, snapshotType, measured.files, measured);
|
|
1254
|
+
if (ok) metadataSent = true;
|
|
1255
|
+
return ok;
|
|
1256
|
+
},
|
|
1257
|
+
drop: (file) => {
|
|
1258
|
+
this.log("warn", "OmniRush file exceeds the upload rail and cannot be shipped as a single part; skipped", {
|
|
1259
|
+
sessionId: state.id,
|
|
1260
|
+
path: file.path,
|
|
1261
|
+
bytes: Buffer.byteLength(file.content ?? file.target ?? ""),
|
|
1262
|
+
});
|
|
1263
|
+
state.trace.push({
|
|
1264
|
+
at: new Date().toISOString(),
|
|
1265
|
+
type: "file.unshippable",
|
|
1266
|
+
data: { path: file.path, bytes: Buffer.byteLength(file.content ?? file.target ?? "") },
|
|
1267
|
+
});
|
|
1268
|
+
},
|
|
1269
|
+
});
|
|
1270
|
+
}
|
|
1271
|
+
|
|
1272
|
+
private async *iterTreeFiles(state: SessionState, paths: string[]): AsyncGenerator<CollectorFile> {
|
|
1273
|
+
for (const path of paths) {
|
|
1274
|
+
if (state.budgetExhausted) return;
|
|
1275
|
+
for (const entry of await readWorkspaceEntries(state.root, path)) {
|
|
1276
|
+
yield entry;
|
|
1277
|
+
}
|
|
1278
|
+
}
|
|
1279
|
+
}
|
|
1280
|
+
|
|
1281
|
+
/**
|
|
1282
|
+
* Change snapshot: ordered journal slices. Each part carries the
|
|
1283
|
+
* touched files for its slice (deduped last-wins, sorted by path) plus
|
|
1284
|
+
* the __agent__/changes.json sidecar holding the slice's ordered
|
|
1285
|
+
* records — replaying start + sidecars in sequence order reproduces the
|
|
1286
|
+
* end tree.
|
|
1287
|
+
*/
|
|
1288
|
+
private async uploadChangeParts(state: SessionState, snapshot: ChangeJournalEntry[]): Promise<ChangeJournalEntry[][]> {
|
|
1289
|
+
const uploadedSlices: ChangeJournalEntry[][] = [];
|
|
1290
|
+
const filesForSlice = (entries: ChangeJournalEntry[]): CollectorFile[] => {
|
|
1291
|
+
const lastByPath = new Map<string, CollectorFile>();
|
|
1292
|
+
const slicedByPath = new Map<string, ChangeJournalEntry[]>();
|
|
1293
|
+
for (const entry of entries) {
|
|
1294
|
+
if (entry.status !== "present") continue;
|
|
1295
|
+
if (entry.target !== undefined) {
|
|
1296
|
+
lastByPath.set(entry.path, { path: entry.path, target: entry.target });
|
|
1297
|
+
} else if (entry.slice) {
|
|
1298
|
+
const group = slicedByPath.get(entry.path) ?? [];
|
|
1299
|
+
group.push(entry);
|
|
1300
|
+
slicedByPath.set(entry.path, group);
|
|
1301
|
+
} else if (entry.content !== undefined) {
|
|
1302
|
+
lastByPath.set(entry.path, {
|
|
1303
|
+
path: entry.path,
|
|
1304
|
+
content: entry.content,
|
|
1305
|
+
...(entry.encoding ? { encoding: entry.encoding } : {}),
|
|
1306
|
+
...(entry.mode ? { mode: entry.mode } : {}),
|
|
1307
|
+
});
|
|
1308
|
+
}
|
|
1309
|
+
}
|
|
1310
|
+
const files = [...lastByPath.values()];
|
|
1311
|
+
for (const [logicalPath, records] of slicedByPath) {
|
|
1312
|
+
records.sort((left, right) => (left.slice?.index ?? 0) - (right.slice?.index ?? 0));
|
|
1313
|
+
const encoding = records.find((record) => record.encoding)?.encoding;
|
|
1314
|
+
const mode = records.find((record) => record.mode)?.mode;
|
|
1315
|
+
for (const record of records) {
|
|
1316
|
+
files.push({
|
|
1317
|
+
path: slicePartPath(logicalPath, record.slice!.index),
|
|
1318
|
+
content: record.content,
|
|
1319
|
+
...(encoding ? { encoding } : {}),
|
|
1320
|
+
});
|
|
1321
|
+
}
|
|
1322
|
+
files.push({
|
|
1323
|
+
path: sliceManifestPath(logicalPath),
|
|
1324
|
+
content: JSON.stringify(sliceManifest(
|
|
1325
|
+
{
|
|
1326
|
+
path: logicalPath,
|
|
1327
|
+
...(encoding ? { encoding } : {}),
|
|
1328
|
+
...(mode ? { mode } : {}),
|
|
1329
|
+
},
|
|
1330
|
+
records.map((record) => ({
|
|
1331
|
+
index: record.slice!.index,
|
|
1332
|
+
bytes: record.slice!.bytes,
|
|
1333
|
+
sha256: record.slice!.sha256,
|
|
1334
|
+
})),
|
|
1335
|
+
records[0].slice!.total,
|
|
1336
|
+
)),
|
|
1337
|
+
});
|
|
1338
|
+
}
|
|
1339
|
+
files.sort((left, right) => comparePaths(left.path, right.path));
|
|
806
1340
|
files.push({
|
|
807
|
-
path:
|
|
808
|
-
content: JSON.stringify({ schema_version: 1, session_id: state.id, entries
|
|
1341
|
+
path: `${AGENT_DIR}/changes.json`,
|
|
1342
|
+
content: JSON.stringify({ schema_version: 1, session_id: state.id, entries }),
|
|
809
1343
|
});
|
|
1344
|
+
return files;
|
|
1345
|
+
};
|
|
1346
|
+
const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
|
|
1347
|
+
await this.packAndUpload(state, "change", snapshot, {
|
|
1348
|
+
sizeOf: (entry) =>
|
|
1349
|
+
Buffer.byteLength(entry.content ?? entry.target ?? "") + entry.path.length + 96,
|
|
1350
|
+
measure: async (batch) => {
|
|
1351
|
+
const files = filesForSlice(batch);
|
|
1352
|
+
const payload = buildEnvelopePayload(bySequence(), "change", files);
|
|
1353
|
+
return { files, payload, compressed: await compressZstd(payload) };
|
|
1354
|
+
},
|
|
1355
|
+
upload: async (measured, batch) => {
|
|
1356
|
+
const ok = await this.uploadEnvelope(state, "change", measured.files, measured);
|
|
1357
|
+
if (ok) uploadedSlices.push(batch);
|
|
1358
|
+
return ok;
|
|
1359
|
+
},
|
|
1360
|
+
drop: (entry) => {
|
|
1361
|
+
this.log("warn", "OmniRush change record exceeds the upload limits and cannot be shipped; skipped", {
|
|
1362
|
+
sessionId: state.id,
|
|
1363
|
+
path: entry.path,
|
|
1364
|
+
bytes: Buffer.byteLength(entry.content ?? entry.target ?? ""),
|
|
1365
|
+
});
|
|
1366
|
+
state.trace.push({
|
|
1367
|
+
at: new Date().toISOString(),
|
|
1368
|
+
type: "file.unshippable",
|
|
1369
|
+
data: { path: entry.path, reason: "change_record_over_limit" },
|
|
1370
|
+
});
|
|
1371
|
+
},
|
|
1372
|
+
}, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
|
|
1373
|
+
return uploadedSlices;
|
|
1374
|
+
}
|
|
1375
|
+
|
|
1376
|
+
/**
|
|
1377
|
+
* Streaming packer shared by tree and change uploads: accumulate units
|
|
1378
|
+
* in order, compress-check every partCheckBytes of uncompressed growth,
|
|
1379
|
+
* flush parts targeting partCompressedBytes, and respect the hard rail
|
|
1380
|
+
* (a lone unit over the rail is skipped via hooks.drop — never split).
|
|
1381
|
+
* fit() mutates the batch it is given: the returned `measured` always
|
|
1382
|
+
* corresponds to the (mutated) batch passed to hooks.upload, and popped
|
|
1383
|
+
* units come back as `carry` to lead the next part. hooks.upload must
|
|
1384
|
+
* persist sequence bookkeeping via uploadEnvelope.
|
|
1385
|
+
*/
|
|
1386
|
+
private async packAndUpload<U>(
|
|
1387
|
+
state: SessionState,
|
|
1388
|
+
snapshotType: SnapshotType,
|
|
1389
|
+
units: Iterable<U> | AsyncIterable<U>,
|
|
1390
|
+
hooks: {
|
|
1391
|
+
sizeOf: (unit: U) => number;
|
|
1392
|
+
measure: (batch: U[]) => Promise<{ files: CollectorFile[]; payload: Buffer; compressed: Buffer }>;
|
|
1393
|
+
upload: (measured: { files: CollectorFile[]; payload: Buffer; compressed: Buffer }, batch: U[]) => Promise<boolean>;
|
|
1394
|
+
drop?: (unit: U) => void;
|
|
1395
|
+
},
|
|
1396
|
+
opts: { checkBytes?: number; flushPlainBytes?: number } = {},
|
|
1397
|
+
): Promise<void> {
|
|
1398
|
+
const checkBytes = Math.max(1024, opts.checkBytes ?? this.partCheckBytes);
|
|
1399
|
+
const flushPlainBytes = Math.max(checkBytes, opts.flushPlainBytes ?? PART_BATCH_UNCOMPRESSED_MAX);
|
|
1400
|
+
const hard = MAX_PART_COMPRESSED_HARD;
|
|
1401
|
+
const fit = async (batch: U[]) => {
|
|
1402
|
+
let measured = await hooks.measure(batch);
|
|
1403
|
+
const carry: U[] = [];
|
|
1404
|
+
// Multi-file parts split at the compressed target. A lone unit
|
|
1405
|
+
// (e.g. one slice entry) may use the body headroom up to the hard
|
|
1406
|
+
// rail; beyond that it is dropped loudly rather than split
|
|
1407
|
+
// mid-content.
|
|
1408
|
+
while (measured.compressed.length > this.partCompressedBytes && batch.length > 1) {
|
|
1409
|
+
carry.unshift(batch.pop()!);
|
|
1410
|
+
measured = await hooks.measure(batch);
|
|
1411
|
+
}
|
|
1412
|
+
if (measured.compressed.length > hard && batch.length === 1) {
|
|
1413
|
+
const lone = batch.pop()!;
|
|
1414
|
+
hooks.drop?.(lone);
|
|
1415
|
+
return { measured: batch.length > 0 ? await hooks.measure(batch) : null, carry };
|
|
1416
|
+
}
|
|
1417
|
+
return { measured, carry };
|
|
1418
|
+
};
|
|
1419
|
+
|
|
1420
|
+
let batch: U[] = [];
|
|
1421
|
+
let batchBytes = 0;
|
|
1422
|
+
let nextCheck = checkBytes;
|
|
1423
|
+
const resetBatch = (unitsToKeep: U[]) => {
|
|
1424
|
+
batch = unitsToKeep;
|
|
1425
|
+
batchBytes = batch.reduce((total, unit) => total + hooks.sizeOf(unit), 0);
|
|
1426
|
+
nextCheck = batchBytes + checkBytes;
|
|
1427
|
+
};
|
|
1428
|
+
const shouldFlush = (measured: { compressed: Buffer }) =>
|
|
1429
|
+
measured.compressed.length >= PART_READY_COMPRESSED_BYTES || batchBytes >= flushPlainBytes;
|
|
1430
|
+
|
|
1431
|
+
for await (const unit of units as AsyncIterable<U>) {
|
|
1432
|
+
if (state.budgetExhausted) return;
|
|
1433
|
+
batch.push(unit);
|
|
1434
|
+
batchBytes += hooks.sizeOf(unit);
|
|
1435
|
+
if (batchBytes < nextCheck) continue;
|
|
1436
|
+
const { measured, carry } = await fit(batch);
|
|
1437
|
+
if (measured && shouldFlush(measured)) {
|
|
1438
|
+
if (!(await hooks.upload(measured, batch))) return;
|
|
1439
|
+
resetBatch(carry);
|
|
1440
|
+
} else if (measured) {
|
|
1441
|
+
// Under the flush thresholds — keep accumulating (carry is empty
|
|
1442
|
+
// here: popping only happens above the flush levels).
|
|
1443
|
+
batch = [...batch, ...carry];
|
|
1444
|
+
nextCheck = batchBytes + checkBytes;
|
|
1445
|
+
} else {
|
|
1446
|
+
resetBatch(carry);
|
|
1447
|
+
}
|
|
1448
|
+
}
|
|
1449
|
+
while (batch.length > 0 && !state.budgetExhausted) {
|
|
1450
|
+
const { measured, carry } = await fit(batch);
|
|
1451
|
+
if (measured) {
|
|
1452
|
+
if (!(await hooks.upload(measured, batch))) return;
|
|
1453
|
+
}
|
|
1454
|
+
if (carry.length === 0) break;
|
|
1455
|
+
batch = carry;
|
|
1456
|
+
batchBytes = batch.reduce((total, unit) => total + hooks.sizeOf(unit), 0);
|
|
810
1457
|
}
|
|
811
|
-
const uploaded = await this.uploadEnvelope(state, type, files);
|
|
812
|
-
if (uploaded && journal.length > 0) this.acknowledgeJournal(state, journal);
|
|
813
|
-
state.lastSignature = await workspaceSignature(state.root);
|
|
814
1458
|
}
|
|
815
1459
|
|
|
1460
|
+
/** Chunked trace upload: event subsets per part, ordered by sequence. */
|
|
816
1461
|
private async uploadTrace(state: SessionState, traceEvents: TraceEvent[]): Promise<void> {
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
1462
|
+
if (state.budgetExhausted || traceEvents.length === 0) return;
|
|
1463
|
+
const remaining = this.sessionBudgetBytes - state.sentBytes;
|
|
1464
|
+
if (remaining <= 1024) {
|
|
1465
|
+
this.exhaustSessionBudget(state);
|
|
1466
|
+
return;
|
|
1467
|
+
}
|
|
1468
|
+
// Budget wall: truncate (loudly, trace_truncated=true) only here.
|
|
1469
|
+
const selected = selectTraceEvents(traceEvents, remaining - 1024);
|
|
1470
|
+
const meta = {
|
|
1471
|
+
truncated: selected.truncated,
|
|
1472
|
+
dropped: selected.droppedCount,
|
|
1473
|
+
};
|
|
1474
|
+
if (selected.truncated) {
|
|
1475
|
+
this.log("warn", "OmniRush trace hit the session budget wall; older events dropped", {
|
|
1476
|
+
sessionId: state.id,
|
|
1477
|
+
droppedEventCount: selected.droppedCount,
|
|
1478
|
+
});
|
|
1479
|
+
}
|
|
1480
|
+
let partIndex = 0;
|
|
1481
|
+
const bySequence = () => ({ id: state.id, segment: state.segment, resumed: state.resumed, sequence: state.sequence });
|
|
1482
|
+
await this.packAndUpload(state, "trace", selected.events, {
|
|
1483
|
+
sizeOf: (event) => Buffer.byteLength(JSON.stringify(event)) + 32,
|
|
1484
|
+
measure: async (batch) => {
|
|
1485
|
+
const content = JSON.stringify({
|
|
1486
|
+
schema_version: 1,
|
|
1487
|
+
schema: AGENT_TRACE_SCHEMA,
|
|
1488
|
+
session_id: state.id,
|
|
1489
|
+
session_segment: state.segment,
|
|
1490
|
+
session_resumed: state.resumed,
|
|
1491
|
+
trace_truncated: meta.truncated,
|
|
1492
|
+
dropped_event_count: meta.dropped,
|
|
1493
|
+
trace_part: { index: partIndex },
|
|
1494
|
+
events: batch,
|
|
1495
|
+
});
|
|
1496
|
+
const files = [{ path: `${AGENT_DIR}/trace.json`, content }];
|
|
1497
|
+
const payload = buildEnvelopePayload(bySequence(), "trace", files);
|
|
1498
|
+
return { files, payload, compressed: await compressZstd(payload) };
|
|
1499
|
+
},
|
|
1500
|
+
upload: async (measured) => {
|
|
1501
|
+
const ok = await this.uploadEnvelope(state, "trace", measured.files, measured);
|
|
1502
|
+
if (ok) partIndex += 1;
|
|
1503
|
+
return ok;
|
|
1504
|
+
},
|
|
1505
|
+
drop: (event) => {
|
|
1506
|
+
this.log("warn", "OmniRush trace event exceeds the upload limits; skipped", {
|
|
1507
|
+
sessionId: state.id,
|
|
1508
|
+
type: event.type,
|
|
1509
|
+
bytes: Buffer.byteLength(JSON.stringify(event)),
|
|
1510
|
+
});
|
|
1511
|
+
},
|
|
1512
|
+
}, { checkBytes: PART_SIDECAR_PLAIN_MAX, flushPlainBytes: PART_SIDECAR_PLAIN_MAX });
|
|
821
1513
|
}
|
|
822
1514
|
|
|
823
|
-
|
|
1515
|
+
/**
|
|
1516
|
+
* Send one envelope (one part). `precompressed` skips the rebuild +
|
|
1517
|
+
* recompress when the caller already measured this exact batch.
|
|
1518
|
+
*/
|
|
1519
|
+
private async uploadEnvelope(
|
|
1520
|
+
state: SessionState,
|
|
1521
|
+
snapshotType: SnapshotType,
|
|
1522
|
+
files: CollectorFile[],
|
|
1523
|
+
precompressed?: { payload: Buffer; compressed: Buffer },
|
|
1524
|
+
): Promise<boolean> {
|
|
824
1525
|
if (!this.collectUrl && !this.uploader) return false;
|
|
825
|
-
|
|
826
|
-
|
|
827
|
-
|
|
828
|
-
|
|
829
|
-
|
|
1526
|
+
if (state.budgetExhausted) return false;
|
|
1527
|
+
const payload = precompressed?.payload ?? buildEnvelopePayload(state, snapshotType, files);
|
|
1528
|
+
if (state.sentBytes + payload.length > this.sessionBudgetBytes) {
|
|
1529
|
+
this.exhaustSessionBudget(state);
|
|
1530
|
+
return false;
|
|
1531
|
+
}
|
|
1532
|
+
const compressed = precompressed?.compressed ?? await compressZstd(payload);
|
|
1533
|
+
if (compressed.length > MAX_PART_COMPRESSED_HARD) {
|
|
1534
|
+
this.log("warn", "OmniRush part exceeded the backend body rail; not sent", {
|
|
1535
|
+
sessionId: state.id,
|
|
1536
|
+
snapshotType,
|
|
1537
|
+
compressedBytes: compressed.length,
|
|
1538
|
+
railBytes: MAX_PART_COMPRESSED_HARD,
|
|
1539
|
+
});
|
|
830
1540
|
return false;
|
|
831
1541
|
}
|
|
832
1542
|
const send = () => this.uploader
|
|
@@ -839,7 +1549,7 @@ export class WorkspaceCollector {
|
|
|
839
1549
|
"X-Omnirush-Session-ID": state.id,
|
|
840
1550
|
},
|
|
841
1551
|
body: compressed.buffer.slice(compressed.byteOffset, compressed.byteOffset + compressed.byteLength) as ArrayBuffer,
|
|
842
|
-
signal: AbortSignal.timeout(
|
|
1552
|
+
signal: AbortSignal.timeout(120_000),
|
|
843
1553
|
});
|
|
844
1554
|
let response = await send();
|
|
845
1555
|
if (response.status === 401 && this.refresh) {
|
|
@@ -867,6 +1577,7 @@ export class WorkspaceCollector {
|
|
|
867
1577
|
snapshotType,
|
|
868
1578
|
fileCount: files.length,
|
|
869
1579
|
compressedBytes: compressed.length,
|
|
1580
|
+
sequence: state.sequence,
|
|
870
1581
|
});
|
|
871
1582
|
return true;
|
|
872
1583
|
}
|