mixdog 0.9.97 → 0.9.99
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +9 -2
- package/src/rules/agent/30-explorer.md +43 -32
- package/src/runtime/agent/orchestrator/agent-trace.mjs +19 -5
- package/src/runtime/agent/orchestrator/providers/grok-oauth.mjs +7 -0
- package/src/runtime/agent/orchestrator/providers/media-normalization.mjs +30 -3
- package/src/runtime/agent/orchestrator/providers/model-catalog.mjs +5 -3
- package/src/runtime/agent/orchestrator/providers/provider-catalog-cache.mjs +23 -1
- package/src/runtime/agent/orchestrator/session/manager/ask-session.mjs +2 -1
- package/src/runtime/agent/orchestrator/session/manager/pending-messages.mjs +44 -6
- package/src/runtime/agent/orchestrator/session/manager/prompt-utils.mjs +2 -1
- package/src/runtime/agent/orchestrator/session/store/serialize.mjs +3 -6
- package/src/runtime/agent/orchestrator/tools/builtin/read-image-resize.mjs +103 -12
- package/src/runtime/agent/orchestrator/tools/builtin/read-special-files.mjs +10 -30
- package/src/runtime/agent/orchestrator/tools/patch/orchestrator.mjs +27 -84
- package/src/runtime/attachments/pdf-extract.mjs +83 -0
- package/src/runtime/attachments/store.mjs +515 -0
- package/src/runtime/attachments/store.test.mjs +188 -0
- package/src/runtime/channels/lib/interaction-handlers.mjs +1 -191
- package/src/runtime/channels/lib/runtime-paths.mjs +1 -4
- package/src/runtime/channels/lib/worker-ipc.mjs +0 -66
- package/src/runtime/channels/lib/worker-main.mjs +1 -20
- package/src/runtime/memory/lib/http-wire.mjs +39 -4
- package/src/runtime/shared/json-metrics.mjs +94 -0
- package/src/runtime/shared/turn-snapshot.mjs +13 -0
- package/src/runtime/shared/turn-worktree-snapshot.mjs +19 -0
- package/src/session-runtime/provider-models.mjs +11 -1
- package/src/session-runtime/quick-model-rows.mjs +38 -7
- package/src/session-runtime/quick-search-models.mjs +15 -17
- package/src/session-runtime/runtime-core.mjs +9 -4
- package/src/session-runtime/runtime-tunables.mjs +9 -6
- package/src/session-runtime/session-title.mjs +27 -3
- package/src/standalone/channel-transport.mjs +28 -12
- package/src/standalone/daemon.mjs +3 -0
- package/src/standalone/session-protocol.mjs +1 -0
- package/src/standalone/session-service.mjs +23 -1
- package/src/standalone/session-transport.mjs +3 -2
- package/src/tui/app/prompt-submit.mjs +7 -9
- package/src/tui/app/use-prompt-handlers.mjs +7 -3
- package/src/tui/dist/index.mjs +249 -43
- package/src/tui/paste-attachments.mjs +13 -7
- package/src/tui/prompt-history-store.mjs +52 -4
- package/src/tui/session/agent-job-feed.mjs +5 -1
- package/src/tui/session/live-share.mjs +21 -7
- package/src/tui/session/queue-helpers.mjs +6 -6
- package/src/tui/session/session-api-ext.mjs +12 -4
- package/src/tui/session/session-api.mjs +10 -1
- package/src/tui/session/session-flow.mjs +4 -2
- package/src/tui/session-local.mjs +15 -5
|
@@ -3,10 +3,11 @@
|
|
|
3
3
|
// readImageWithTokenBudget) so a `read` on an image returns a viewable,
|
|
4
4
|
// budget-bounded image block instead of refusing oversized originals.
|
|
5
5
|
//
|
|
6
|
-
// sharp is
|
|
7
|
-
//
|
|
8
|
-
//
|
|
9
|
-
|
|
6
|
+
// sharp is a direct runtime dependency. Entry points still degrade to `null`
|
|
7
|
+
// when a platform-native binding cannot load so a damaged install reports the
|
|
8
|
+
// existing bounded fallback instead of crashing the whole daemon.
|
|
9
|
+
|
|
10
|
+
import { createHash } from 'node:crypto';
|
|
10
11
|
|
|
11
12
|
// Anthropic inline-image input is capped near 5MB base64 (API rejects on the
|
|
12
13
|
// base64 LENGTH, not raw bytes). IMAGE_TARGET_RAW_SIZE is the raw-byte target
|
|
@@ -20,6 +21,70 @@ const IMAGE_MAX_HEIGHT = 2000;
|
|
|
20
21
|
// dimension/raw-size resize governs the common case and the token gate only
|
|
21
22
|
// fires on pathologically dense images.
|
|
22
23
|
const DEFAULT_IMAGE_MAX_TOKENS = Math.ceil(API_IMAGE_MAX_BASE64_SIZE * 0.125);
|
|
24
|
+
export const OPENAI_IMAGE_MAX_DIMENSION = 2048;
|
|
25
|
+
export const OPENAI_IMAGE_PATCH_SIZE = 32;
|
|
26
|
+
export const OPENAI_IMAGE_MAX_PATCHES = 1536;
|
|
27
|
+
const IMAGE_RESIZE_CACHE_MAX_BYTES = 64 * 1024 * 1024;
|
|
28
|
+
const imageResizeCache = new Map();
|
|
29
|
+
let imageResizeCacheBytes = 0;
|
|
30
|
+
let imageResizeCacheHits = 0;
|
|
31
|
+
let imageResizeCacheMisses = 0;
|
|
32
|
+
|
|
33
|
+
export function imageProfileForProvider(provider) {
|
|
34
|
+
const value = String(provider || '').trim().toLowerCase();
|
|
35
|
+
return /^(?:openai|xai|grok|deepseek|opencode-go|ollama|lmstudio)(?:-|$)/.test(value)
|
|
36
|
+
? 'openai'
|
|
37
|
+
: 'anthropic';
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function openAIImagePatchCount(width, height) {
|
|
41
|
+
return Math.ceil(Math.max(1, Number(width) || 1) / OPENAI_IMAGE_PATCH_SIZE)
|
|
42
|
+
* Math.ceil(Math.max(1, Number(height) || 1) / OPENAI_IMAGE_PATCH_SIZE);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function resizeCacheKey(buffer, ext, maxTokens, profile) {
|
|
46
|
+
return createHash('sha256')
|
|
47
|
+
.update(String(ext || '')).update('\0')
|
|
48
|
+
.update(String(maxTokens || 0)).update('\0')
|
|
49
|
+
.update(String(profile || '')).update('\0')
|
|
50
|
+
.update(buffer)
|
|
51
|
+
.digest('hex');
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
function cloneResizeResult(result) {
|
|
55
|
+
return {
|
|
56
|
+
...result,
|
|
57
|
+
...(result?.dimensions ? { dimensions: { ...result.dimensions } } : {}),
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function rememberResizeResult(key, result) {
|
|
62
|
+
const bytes = Buffer.byteLength(String(result?.data || ''), 'base64');
|
|
63
|
+
if (bytes <= 0 || bytes > IMAGE_RESIZE_CACHE_MAX_BYTES) return;
|
|
64
|
+
const existing = imageResizeCache.get(key);
|
|
65
|
+
if (existing) {
|
|
66
|
+
imageResizeCacheBytes -= existing.bytes;
|
|
67
|
+
imageResizeCache.delete(key);
|
|
68
|
+
}
|
|
69
|
+
imageResizeCache.set(key, { result: cloneResizeResult(result), bytes });
|
|
70
|
+
imageResizeCacheBytes += bytes;
|
|
71
|
+
while (imageResizeCacheBytes > IMAGE_RESIZE_CACHE_MAX_BYTES && imageResizeCache.size > 0) {
|
|
72
|
+
const oldest = imageResizeCache.keys().next().value;
|
|
73
|
+
const evicted = imageResizeCache.get(oldest);
|
|
74
|
+
imageResizeCache.delete(oldest);
|
|
75
|
+
imageResizeCacheBytes -= evicted?.bytes || 0;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
export function imageResizeCacheStats() {
|
|
80
|
+
return {
|
|
81
|
+
entries: imageResizeCache.size,
|
|
82
|
+
bytes: imageResizeCacheBytes,
|
|
83
|
+
maxBytes: IMAGE_RESIZE_CACHE_MAX_BYTES,
|
|
84
|
+
hits: imageResizeCacheHits,
|
|
85
|
+
misses: imageResizeCacheMisses,
|
|
86
|
+
};
|
|
87
|
+
}
|
|
23
88
|
|
|
24
89
|
// Cached dynamic import. Resolves to the sharp factory or null (absent /
|
|
25
90
|
// failed). Cached so repeated reads don't re-attempt a failing import.
|
|
@@ -38,6 +103,10 @@ async function loadSharp() {
|
|
|
38
103
|
return _sharpPromise;
|
|
39
104
|
}
|
|
40
105
|
|
|
106
|
+
export function prewarmImageResizer() {
|
|
107
|
+
return loadSharp();
|
|
108
|
+
}
|
|
109
|
+
|
|
41
110
|
// True when sharp resolved; used for the per-file change summary / fallback note.
|
|
42
111
|
async function sharpAvailable() {
|
|
43
112
|
return (await loadSharp()) !== null;
|
|
@@ -88,8 +157,21 @@ export function imageMetadataText(dims, sourcePath) {
|
|
|
88
157
|
// Returns { data (base64), mimeType ("image/..."), dimensions } on success,
|
|
89
158
|
// or null when sharp is unavailable OR any sharp op threw (caller falls back
|
|
90
159
|
// to legacy pass-through-with-cap).
|
|
91
|
-
export async function resizeImageBuffer(buffer, ext, {
|
|
160
|
+
export async function resizeImageBuffer(buffer, ext, {
|
|
161
|
+
maxTokens = DEFAULT_IMAGE_MAX_TOKENS,
|
|
162
|
+
profile = 'anthropic',
|
|
163
|
+
} = {}) {
|
|
92
164
|
if (!Buffer.isBuffer(buffer) || buffer.length === 0) return null;
|
|
165
|
+
const normalizedProfile = profile === 'openai' ? 'openai' : 'anthropic';
|
|
166
|
+
const cacheKey = resizeCacheKey(buffer, ext, maxTokens, normalizedProfile);
|
|
167
|
+
const cached = imageResizeCache.get(cacheKey);
|
|
168
|
+
if (cached) {
|
|
169
|
+
imageResizeCacheHits += 1;
|
|
170
|
+
imageResizeCache.delete(cacheKey);
|
|
171
|
+
imageResizeCache.set(cacheKey, cached);
|
|
172
|
+
return cloneResizeResult(cached.result);
|
|
173
|
+
}
|
|
174
|
+
imageResizeCacheMisses += 1;
|
|
93
175
|
const sharp = await loadSharp();
|
|
94
176
|
if (!sharp) return null;
|
|
95
177
|
try {
|
|
@@ -108,13 +190,20 @@ export async function resizeImageBuffer(buffer, ext, { maxTokens = DEFAULT_IMAGE
|
|
|
108
190
|
// Constrain dimensions while preserving aspect ratio.
|
|
109
191
|
let width = originalWidth;
|
|
110
192
|
let height = originalHeight;
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
193
|
+
const maxWidth = normalizedProfile === 'openai' ? OPENAI_IMAGE_MAX_DIMENSION : IMAGE_MAX_WIDTH;
|
|
194
|
+
const maxHeight = normalizedProfile === 'openai' ? OPENAI_IMAGE_MAX_DIMENSION : IMAGE_MAX_HEIGHT;
|
|
195
|
+
let scale = Math.min(1, maxWidth / width, maxHeight / height);
|
|
196
|
+
if (normalizedProfile === 'openai') {
|
|
197
|
+
scale = Math.min(
|
|
198
|
+
scale,
|
|
199
|
+
Math.sqrt((OPENAI_IMAGE_MAX_PATCHES * OPENAI_IMAGE_PATCH_SIZE ** 2) / (width * height)),
|
|
200
|
+
);
|
|
114
201
|
}
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
202
|
+
width = Math.max(1, Math.floor(width * scale));
|
|
203
|
+
height = Math.max(1, Math.floor(height * scale));
|
|
204
|
+
while (normalizedProfile === 'openai' && openAIImagePatchCount(width, height) > OPENAI_IMAGE_MAX_PATCHES) {
|
|
205
|
+
if (width >= height) width -= 1;
|
|
206
|
+
else height -= 1;
|
|
118
207
|
}
|
|
119
208
|
const needsResize = width !== originalWidth || height !== originalHeight;
|
|
120
209
|
if (needsResize || originalSize > IMAGE_TARGET_RAW_SIZE) {
|
|
@@ -162,11 +251,13 @@ export async function resizeImageBuffer(buffer, ext, { maxTokens = DEFAULT_IMAGE
|
|
|
162
251
|
}
|
|
163
252
|
}
|
|
164
253
|
|
|
165
|
-
|
|
254
|
+
const result = {
|
|
166
255
|
data: base64,
|
|
167
256
|
mimeType: `image/${mediaType}`,
|
|
168
257
|
dimensions: { originalWidth, originalHeight, displayWidth, displayHeight },
|
|
169
258
|
};
|
|
259
|
+
rememberResizeResult(cacheKey, result);
|
|
260
|
+
return result;
|
|
170
261
|
} catch {
|
|
171
262
|
// sharp present but processing failed (corrupt header, unsupported
|
|
172
263
|
// format, OOM). Signal fallback rather than throwing.
|
|
@@ -1,16 +1,15 @@
|
|
|
1
1
|
import { readFile, stat } from 'fs/promises';
|
|
2
2
|
import { open } from 'fs/promises';
|
|
3
|
-
import { createRequire } from 'module';
|
|
4
3
|
import { READ_MAX_SIZE_BYTES } from './read-constants.mjs';
|
|
5
4
|
import { imageBlocksFromBuffer } from './read-image-resize.mjs';
|
|
5
|
+
import { inspectPdfBuffer } from '../../../../attachments/pdf-extract.mjs';
|
|
6
6
|
|
|
7
|
-
const requireCjs = createRequire(import.meta.url);
|
|
8
7
|
const DEFAULT_READ_MAX_OUTPUT_BYTES = 100 * 1024;
|
|
9
8
|
|
|
10
9
|
// PDFs at or under this size are emitted as an Anthropic base64 document
|
|
11
10
|
// block (the model reads the rendered PDF directly): 20MB raw → ~27MB
|
|
12
11
|
// base64, which stays under the 32MB request cap.
|
|
13
|
-
// Larger PDFs fall back to
|
|
12
|
+
// Larger PDFs fall back to bounded PDF.js text extraction.
|
|
14
13
|
const PDF_DOCUMENT_MAX_BYTES = 20 * 1024 * 1024;
|
|
15
14
|
|
|
16
15
|
// %PDF- magic bytes (0x25 0x50 0x44 0x46 0x2D). A document block must only be
|
|
@@ -66,37 +65,18 @@ function parsePagesArg(pagesArg) {
|
|
|
66
65
|
return { filter: { from, to } };
|
|
67
66
|
}
|
|
68
67
|
|
|
69
|
-
//
|
|
68
|
+
// PDF.js text extraction. Fallback path for PDFs over the document-block
|
|
70
69
|
// size cap, and the always-path when a page range is requested (a base64
|
|
71
70
|
// document block can't be page-filtered, so a narrowed read keeps using text).
|
|
72
71
|
async function extractPdfTextBody(fullPath, pageFilter, maxOutputBytes) {
|
|
73
|
-
const pdfParse = requireCjs('pdf-parse');
|
|
74
72
|
const buf = await readFile(fullPath);
|
|
75
|
-
const
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
// pageNumber dropped/duplicated pages when pdf.js renumbered with
|
|
81
|
-
// annotations or oddball page trees. Use pageNumber first.
|
|
82
|
-
const pageNum = (typeof pageData.pageNumber === 'number')
|
|
83
|
-
? pageData.pageNumber
|
|
84
|
-
: ((pageData._pageIndex ?? pageData.pageIndex ?? 0) + 1);
|
|
85
|
-
if (pageFilter && (pageNum < pageFilter.from || pageNum > pageFilter.to)) return Promise.resolve('');
|
|
86
|
-
return pageData.getTextContent().then((tc) => {
|
|
87
|
-
const text = tc.items.map((i) => i.str).join(' ');
|
|
88
|
-
pageTexts.push({ page: pageNum, text });
|
|
89
|
-
return text;
|
|
90
|
-
});
|
|
91
|
-
},
|
|
73
|
+
const result = await inspectPdfBuffer(buf, {
|
|
74
|
+
extractText: true,
|
|
75
|
+
maxPages: Infinity,
|
|
76
|
+
maxOutputBytes,
|
|
77
|
+
pageRange: pageFilter,
|
|
92
78
|
});
|
|
93
|
-
|
|
94
|
-
? pageTexts.map((p) => `--- Page ${p.page} ---\n${p.text}`).join('\n\n')
|
|
95
|
-
: (data.text || '');
|
|
96
|
-
if (out.length > maxOutputBytes) {
|
|
97
|
-
out = out.slice(0, maxOutputBytes) + `\n\n... [PDF output truncated at ${Math.round(maxOutputBytes / 1024)} KB; use pages param to narrow]`;
|
|
98
|
-
}
|
|
99
|
-
return out || '(no text content extracted from PDF)';
|
|
79
|
+
return result.text || '(no text content extracted from PDF)';
|
|
100
80
|
}
|
|
101
81
|
|
|
102
82
|
export async function extractPdfText(fullPath, pagesArg, { maxOutputBytes = DEFAULT_READ_MAX_OUTPUT_BYTES, textOnly = false } = {}) {
|
|
@@ -136,7 +116,7 @@ export async function extractPdfText(fullPath, pagesArg, { maxOutputBytes = DEFA
|
|
|
136
116
|
// TEXT fallback: >20MB PDFs, page-filtered reads, or non-magic files.
|
|
137
117
|
return await extractPdfTextBody(fullPath, pages.filter, maxOutputBytes);
|
|
138
118
|
} catch (err) {
|
|
139
|
-
return `Error:
|
|
119
|
+
return `Error: PDF extraction failed — ${err instanceof Error ? err.message : String(err)}`;
|
|
140
120
|
}
|
|
141
121
|
}
|
|
142
122
|
|
|
@@ -200,7 +200,7 @@ async function applyParsedWave({ parsed: wparsed, entries: wentries, headerRewri
|
|
|
200
200
|
async function applyPatchSequence(patchStr, requestedFormat, basePath, ctx) {
|
|
201
201
|
const {
|
|
202
202
|
v4aConvertOpts, dryRun, fuzz, fuzzy, rejectPartial,
|
|
203
|
-
readStateScope, abortSignal, mutationPlan,
|
|
203
|
+
readStateScope, abortSignal, mutationPlan,
|
|
204
204
|
toolCallId, sessionId,
|
|
205
205
|
} = ctx;
|
|
206
206
|
|
|
@@ -328,52 +328,8 @@ async function applyPatchSequence(patchStr, requestedFormat, basePath, ctx) {
|
|
|
328
328
|
|
|
329
329
|
return withBuiltinPathLocks(lockPaths, () =>
|
|
330
330
|
withAdvisoryLocks(lockPaths, async () => {
|
|
331
|
-
// Most multi-file failures are stale context in a later, independent
|
|
332
|
-
// file. For unique targets, validate every section against current bytes
|
|
333
|
-
// before the first write. Same-target sequences retain the existing
|
|
334
|
-
// ordered conversion/rollback path because later sections may
|
|
335
|
-
// intentionally depend on earlier edits.
|
|
336
|
-
const preparedUnits = new Map();
|
|
337
|
-
const uniqueTargets = new Set(lockPaths.map(patchPathKey)).size === units.length;
|
|
338
|
-
if (!dryRun && rollbackOnFailure && uniqueTargets) {
|
|
339
|
-
for (let i = 0; i < units.length; i++) {
|
|
340
|
-
const unit = units[i];
|
|
341
|
-
if (abortSignal?.aborted) return 'Error: apply_patch aborted during preflight; no files were written';
|
|
342
|
-
let parsed;
|
|
343
|
-
try {
|
|
344
|
-
parsed = await unit.buildParsed();
|
|
345
|
-
} catch (err) {
|
|
346
|
-
return `Error: apply_patch preflight rejected section ${i + 1}/${units.length} (${unit.displayPath}); no files were written.\n${err?.message || String(err)}`;
|
|
347
|
-
}
|
|
348
|
-
if (!Array.isArray(parsed) || parsed.length === 0) {
|
|
349
|
-
preparedUnits.set(i, { parsed, wave: null });
|
|
350
|
-
continue;
|
|
351
|
-
}
|
|
352
|
-
let wave;
|
|
353
|
-
try {
|
|
354
|
-
const { entries, headerRewrites } = await preValidateNativeBatch(parsed, basePath);
|
|
355
|
-
wave = { parsed, entries, headerRewrites };
|
|
356
|
-
} catch (err) {
|
|
357
|
-
return `Error: apply_patch preflight rejected section ${i + 1}/${units.length} (${unit.displayPath}); no files were written.\n${err?.message || String(err)}`;
|
|
358
|
-
}
|
|
359
|
-
const checked = await applyParsedWave(wave, basePath, { ...waveOpts, dryRun: true });
|
|
360
|
-
if (checked.error) {
|
|
361
|
-
const detail = checked.error.replace(/^Error:\s*/, '');
|
|
362
|
-
return `Error: apply_patch preflight rejected section ${i + 1}/${units.length} (${unit.displayPath}); no files were written.\n${detail}`;
|
|
363
|
-
}
|
|
364
|
-
preparedUnits.set(i, { parsed, wave });
|
|
365
|
-
}
|
|
366
|
-
}
|
|
367
|
-
let rollbackSnapshots = [];
|
|
368
331
|
let uiBeforeSnapshots = [];
|
|
369
|
-
if (!dryRun &&
|
|
370
|
-
try {
|
|
371
|
-
rollbackSnapshots = capturePatchRollbackState(lockPaths);
|
|
372
|
-
uiBeforeSnapshots = rollbackSnapshots;
|
|
373
|
-
} catch (err) {
|
|
374
|
-
return `Error: ${err?.message || String(err)}`;
|
|
375
|
-
}
|
|
376
|
-
} else if (!dryRun && toolCallId && sessionId) {
|
|
332
|
+
if (!dryRun && toolCallId && sessionId) {
|
|
377
333
|
try {
|
|
378
334
|
uiBeforeSnapshots = capturePatchRollbackState(lockPaths);
|
|
379
335
|
} catch {
|
|
@@ -393,15 +349,13 @@ async function applyPatchSequence(patchStr, requestedFormat, basePath, ctx) {
|
|
|
393
349
|
failedIndex = i;
|
|
394
350
|
continue;
|
|
395
351
|
}
|
|
396
|
-
let parsed
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
continue;
|
|
404
|
-
}
|
|
352
|
+
let parsed;
|
|
353
|
+
try {
|
|
354
|
+
parsed = await unit.buildParsed();
|
|
355
|
+
} catch (err) {
|
|
356
|
+
failed = { displayPath: unit.displayPath, error: `Error: ${err?.message || String(err)}` };
|
|
357
|
+
failedIndex = i;
|
|
358
|
+
continue;
|
|
405
359
|
}
|
|
406
360
|
if (!Array.isArray(parsed) || parsed.length === 0) {
|
|
407
361
|
// Section produced no applicable hunks (all skipped / no-op). Nothing
|
|
@@ -409,16 +363,14 @@ async function applyPatchSequence(patchStr, requestedFormat, basePath, ctx) {
|
|
|
409
363
|
applied.push({ displayPath: unit.displayPath, text: `(no changes) ${unit.displayPath}` });
|
|
410
364
|
continue;
|
|
411
365
|
}
|
|
412
|
-
let wave
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
continue;
|
|
421
|
-
}
|
|
366
|
+
let wave;
|
|
367
|
+
try {
|
|
368
|
+
const { entries, headerRewrites } = await preValidateNativeBatch(parsed, basePath);
|
|
369
|
+
wave = { parsed, entries, headerRewrites };
|
|
370
|
+
} catch (err) {
|
|
371
|
+
failed = { displayPath: unit.displayPath, error: `Error: ${err?.message || String(err)}` };
|
|
372
|
+
failedIndex = i;
|
|
373
|
+
continue;
|
|
422
374
|
}
|
|
423
375
|
const res = await applyParsedWave(wave, basePath, waveOpts);
|
|
424
376
|
executor = res.executor;
|
|
@@ -462,13 +414,10 @@ async function applyPatchSequence(patchStr, requestedFormat, basePath, ctx) {
|
|
|
462
414
|
return wrapPatchMutationOutput(body, mutationPlan, { executor });
|
|
463
415
|
}
|
|
464
416
|
const failMsg = failed.error.replace(/^Error:\s*/, '');
|
|
465
|
-
const rollbackErrors = (!dryRun && rollbackOnFailure)
|
|
466
|
-
? restorePatchRollbackState(rollbackSnapshots, readStateScope)
|
|
467
|
-
: [];
|
|
468
417
|
if (
|
|
469
418
|
!dryRun
|
|
470
419
|
&& uiBeforeSnapshots.length > 0
|
|
471
|
-
&&
|
|
420
|
+
&& applied.length > 0
|
|
472
421
|
) {
|
|
473
422
|
registerCommittedPatchUiDiff({
|
|
474
423
|
callId: toolCallId,
|
|
@@ -480,22 +429,18 @@ async function applyPatchSequence(patchStr, requestedFormat, basePath, ctx) {
|
|
|
480
429
|
}
|
|
481
430
|
const committedPhrase = dryRun
|
|
482
431
|
? `${applied.length} earlier section(s) were validated`
|
|
483
|
-
: (
|
|
484
|
-
? (rollbackErrors.length === 0
|
|
485
|
-
? `${applied.length} earlier section(s) were applied, then all touched paths were rolled back to their pre-patch state`
|
|
486
|
-
: `${applied.length} earlier section(s) were applied, but rollback was incomplete`)
|
|
487
|
-
: `${applied.length} earlier section(s) were applied to disk (committed) and left in place`);
|
|
432
|
+
: `${applied.length} earlier section(s) were applied to disk (committed) and left in place`;
|
|
488
433
|
const lines = [
|
|
489
434
|
`Error: apply_patch sequence stopped at section ${failedIndex + 1}/${units.length} (${failed.displayPath}); `
|
|
490
435
|
+ `${committedPhrase}; ${skipped.length} later section(s) were skipped (not attempted).`,
|
|
491
436
|
];
|
|
437
|
+
if (!dryRun) {
|
|
438
|
+
lines.push('Retry only the failed and skipped sections; do not resend committed sections.');
|
|
439
|
+
}
|
|
492
440
|
if (appliedTexts) {
|
|
493
|
-
lines.push(`--- ${dryRun ? 'validated' :
|
|
441
|
+
lines.push(`--- ${dryRun ? 'validated' : 'applied (committed to disk)'} ---`, appliedTexts);
|
|
494
442
|
}
|
|
495
443
|
lines.push(`--- failed section: ${failed.displayPath} ---`, failMsg);
|
|
496
|
-
if (rollbackErrors.length > 0) {
|
|
497
|
-
lines.push('--- rollback incomplete ---', ...rollbackErrors);
|
|
498
|
-
}
|
|
499
444
|
if (skipped.length > 0) {
|
|
500
445
|
lines.push(`--- skipped (not attempted): ${skipped.join(', ')} ---`);
|
|
501
446
|
}
|
|
@@ -830,11 +775,10 @@ async function apply_patch(args, cwd, options = {}) {
|
|
|
830
775
|
let inputPatchStr = patchStr;
|
|
831
776
|
const rejectedV4AHunks = [];
|
|
832
777
|
const v4aConvertOpts = { rejectPartial, rejectedHunks: rejectedV4AHunks, fuzzy, dryRun, readStateScope };
|
|
833
|
-
// Default
|
|
834
|
-
//
|
|
835
|
-
//
|
|
836
|
-
//
|
|
837
|
-
// The legacy bulk/atomic path remains available as an explicit escape hatch.
|
|
778
|
+
// Default ordered mode mirrors Codex's lower executor: apply sections in
|
|
779
|
+
// listed order, stop at the first failure, and keep the committed prefix.
|
|
780
|
+
// The model-visible schema stays unchanged; legacy bulk/atomic rollback
|
|
781
|
+
// remains an internal escape hatch.
|
|
838
782
|
const patchMode = String(args?.mode || '').toLowerCase();
|
|
839
783
|
const legacyBulkMode = args?.sequence === false
|
|
840
784
|
|| ['atomic', 'bulk'].includes(patchMode);
|
|
@@ -842,7 +786,6 @@ async function apply_patch(args, cwd, options = {}) {
|
|
|
842
786
|
const seqOut = await applyPatchSequence(patchStr, requestedFormat, basePath, {
|
|
843
787
|
v4aConvertOpts, dryRun, fuzz, fuzzy, rejectPartial,
|
|
844
788
|
readStateScope, abortSignal, mutationPlan,
|
|
845
|
-
rollbackOnFailure: patchMode !== 'partial',
|
|
846
789
|
toolCallId: options?.toolCallId || null,
|
|
847
790
|
sessionId: options?.sessionId || null,
|
|
848
791
|
});
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bounded PDF inspection/text extraction shared by prompt intake and `read`.
|
|
3
|
+
*
|
|
4
|
+
* Native PDF-capable providers receive the original content-addressed file.
|
|
5
|
+
* OpenAI-compatible providers without a document contract receive page-ordered
|
|
6
|
+
* text extracted here, so they never see an unsupported inline Base64 block.
|
|
7
|
+
*/
|
|
8
|
+
const DEFAULT_MAX_PAGES = 100;
|
|
9
|
+
const DEFAULT_MAX_OUTPUT_BYTES = 1024 * 1024;
|
|
10
|
+
|
|
11
|
+
function boundedUtf8(value, maxBytes) {
|
|
12
|
+
const buffer = Buffer.from(String(value || ''), 'utf8');
|
|
13
|
+
if (buffer.length <= maxBytes) return { text: buffer.toString('utf8'), bytes: buffer.length, truncated: false };
|
|
14
|
+
if (maxBytes <= 0) return { text: '', bytes: 0, truncated: true };
|
|
15
|
+
const text = buffer.subarray(0, maxBytes).toString('utf8').replace(/\uFFFD+$/g, '');
|
|
16
|
+
return { text, bytes: Buffer.byteLength(text, 'utf8'), truncated: true };
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
export async function inspectPdfBuffer(buffer, {
|
|
20
|
+
extractText = false,
|
|
21
|
+
maxPages = DEFAULT_MAX_PAGES,
|
|
22
|
+
maxOutputBytes = DEFAULT_MAX_OUTPUT_BYTES,
|
|
23
|
+
pageRange = null,
|
|
24
|
+
} = {}) {
|
|
25
|
+
if (!Buffer.isBuffer(buffer) || buffer.length === 0) throw new TypeError('PDF payload is empty');
|
|
26
|
+
const { getDocumentProxy } = await import('unpdf');
|
|
27
|
+
const pdf = await getDocumentProxy(new Uint8Array(buffer));
|
|
28
|
+
try {
|
|
29
|
+
const pageCount = Math.max(0, Number(pdf?.numPages) || 0);
|
|
30
|
+
const pageLimit = Number.isFinite(Number(maxPages)) ? Math.max(1, Math.floor(Number(maxPages))) : Infinity;
|
|
31
|
+
if (pageCount > pageLimit) {
|
|
32
|
+
throw new RangeError(`PDF has ${pageCount} pages; maximum supported attachment is ${pageLimit} pages`);
|
|
33
|
+
}
|
|
34
|
+
if (!extractText) return { pageCount, text: '', truncated: false };
|
|
35
|
+
|
|
36
|
+
const from = Math.max(1, Number(pageRange?.from) || 1);
|
|
37
|
+
const to = Math.min(pageCount, Math.max(from, Number(pageRange?.to) || pageCount));
|
|
38
|
+
const byteLimit = Math.max(1, Number(maxOutputBytes) || DEFAULT_MAX_OUTPUT_BYTES);
|
|
39
|
+
const chunks = [];
|
|
40
|
+
let bytes = 0;
|
|
41
|
+
let truncated = false;
|
|
42
|
+
for (let pageNumber = from; pageNumber <= to; pageNumber += 1) {
|
|
43
|
+
const page = await pdf.getPage(pageNumber);
|
|
44
|
+
try {
|
|
45
|
+
const content = await page.getTextContent();
|
|
46
|
+
const body = (content?.items || [])
|
|
47
|
+
.map((item) => typeof item?.str === 'string' ? item.str : '')
|
|
48
|
+
.filter(Boolean)
|
|
49
|
+
.join(' ')
|
|
50
|
+
.trim();
|
|
51
|
+
const block = `--- Page ${pageNumber} ---\n${body || '(no extractable text on this page)'}`;
|
|
52
|
+
const separatorBytes = chunks.length ? 2 : 0;
|
|
53
|
+
const remaining = byteLimit - bytes - separatorBytes;
|
|
54
|
+
if (remaining <= 0) {
|
|
55
|
+
truncated = true;
|
|
56
|
+
break;
|
|
57
|
+
}
|
|
58
|
+
const bounded = boundedUtf8(block, remaining);
|
|
59
|
+
if (chunks.length) {
|
|
60
|
+
chunks.push('\n\n');
|
|
61
|
+
bytes += 2;
|
|
62
|
+
}
|
|
63
|
+
chunks.push(bounded.text);
|
|
64
|
+
bytes += bounded.bytes;
|
|
65
|
+
if (bounded.truncated) {
|
|
66
|
+
truncated = true;
|
|
67
|
+
break;
|
|
68
|
+
}
|
|
69
|
+
} finally {
|
|
70
|
+
try { page.cleanup?.(); } catch {}
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
if (truncated) {
|
|
74
|
+
const suffix = '\n\n... [PDF text truncated to the prompt input budget]';
|
|
75
|
+
const room = Math.max(0, byteLimit - Buffer.byteLength(suffix, 'utf8'));
|
|
76
|
+
const bounded = boundedUtf8(chunks.join(''), room);
|
|
77
|
+
return { pageCount, text: `${bounded.text}${suffix}`, truncated: true };
|
|
78
|
+
}
|
|
79
|
+
return { pageCount, text: chunks.join(''), truncated: false };
|
|
80
|
+
} finally {
|
|
81
|
+
try { await pdf.destroy?.(); } catch {}
|
|
82
|
+
}
|
|
83
|
+
}
|