@bevel-software/platform-core-backend 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/THIRD-PARTY-NOTICES.md +1163 -425
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +2 -0
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +5 -0
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  13. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  14. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  15. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  16. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  17. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  18. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  19. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  20. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  21. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  22. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  23. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  24. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  25. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  26. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  27. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  28. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  29. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  30. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  31. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  32. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  33. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  34. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  35. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  36. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  37. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  38. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  39. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  40. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  41. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  42. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  43. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  44. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  45. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  46. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  47. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  48. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  49. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  50. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  51. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  52. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  53. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  54. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  55. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  56. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  57. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  58. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  59. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  60. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  61. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  62. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  63. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  64. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  65. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  66. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  67. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  68. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  69. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  70. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  71. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  72. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  73. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  74. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  75. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  76. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  77. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  78. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  79. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  80. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  81. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  82. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  83. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  84. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  85. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  86. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  87. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  88. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  89. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  90. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  91. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  92. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  93. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  94. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  95. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  96. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  97. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  98. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  99. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  100. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  101. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  102. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  103. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  104. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  105. package/dist/modules/workspace/workspace.tools.js +158 -15
  106. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  107. package/package.json +9 -4
  108. package/src/core/create-core-server.ts +1 -1
  109. package/src/core/create-core-services.ts +6 -0
  110. package/src/core-config.ts +9 -0
  111. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  112. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  113. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  114. package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
  115. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  116. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  117. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  118. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  119. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  120. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  121. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  122. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  123. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  124. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  125. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  126. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  127. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  128. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  129. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  130. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  131. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  132. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  133. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  134. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  135. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  136. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  137. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  138. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  139. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  140. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  141. package/src/modules/workspace/workspace.tools.ts +174 -12
@@ -24,6 +24,11 @@ import type { ISessionSink } from './session-sink.js';
24
24
  import type { IAccessControl } from '../access/access-control.interface.js';
25
25
  import { toKbRelative, resolveReadableMap } from '../access-model/kb-read-filter.js';
26
26
  import type { SpillStore } from './spill-store.js';
27
+ import type { DocExtractService } from './file-readers/doc-extract.service.js';
28
+ import type { FileReaderRegistry } from './file-readers/file-reader.js';
29
+ import { createFileReaderRegistry } from './file-readers/file-reader.registry.js';
30
+ import { DocumentReader } from './file-readers/document-reader.js';
31
+ import { mcpImageResult } from '@bevel-software/platform-mcp-core';
27
32
 
28
33
  /** A directory entry as returned by `LocalFilesystem.readdir`. */
29
34
  interface DirEntry {
@@ -110,6 +115,85 @@ function asText(content: string | Buffer): string {
110
115
  return typeof content === 'string' ? content : content.toString('utf8');
111
116
  }
112
117
 
118
+ function asBytes(content: string | Buffer): Buffer {
119
+ return Buffer.isBuffer(content) ? content : Buffer.from(content, 'utf8');
120
+ }
121
+
122
+ /**
123
+ * How many NOT-yet-cached documents one `grep` call will extract. Extraction
124
+ * is the expensive step (unzip + parse; pdf.js for PDFs) and a cold grep over
125
+ * a document-heavy KB would otherwise extract the whole tree inside a single
126
+ * walk. Cached extractions are always searched (a cache hit costs one small
127
+ * JSON read); beyond the budget the walk skips the document and the result
128
+ * carries a note with the count, so the caller knows a re-run (or a
129
+ * `read_file`, which extracts unbudgeted) will cover the rest.
130
+ */
131
+ const UNCACHED_DOCS_PER_GREP = 20;
132
+
133
+ /** Per-grep-walk extraction state: the shared budget + how many documents it left unsearched. */
134
+ interface DocGrepState {
135
+ readers: FileReaderRegistry;
136
+ uncachedBudget: number;
137
+ skippedUncached: number;
138
+ }
139
+
140
+ /**
141
+ * The write-refusal for office documents/PDFs on the agent TEXT-editing tools
142
+ * (write_file / write_files / edit_file). `read_file` returns an EXTRACTION
143
+ * for these types — their reader declares `textEditable: false` — so an agent
144
+ * that "read" one and writes text back would
145
+ * silently destroy the real document. Uploads and the plain HTTP write routes
146
+ * are untouched — humans replacing a document is exactly the right move — and
147
+ * `unzip` stays the raw-access escape hatch.
148
+ */
149
+ function assertNotDocumentEdit(readers: FileReaderRegistry, path: string): void {
150
+ const reader = readers.readerFor(path);
151
+ if (reader.textEditable) return;
152
+ throw new ToolError(
153
+ reader.editRefusal?.(path) ??
154
+ `"${path}" is an office document/PDF. read_file returns EXTRACTED text for it — not the file's real ` +
155
+ 'content — so text written back cannot round-trip and would corrupt the document. To change it, ' +
156
+ 'replace the document by uploading a new version.',
157
+ 400,
158
+ );
159
+ }
160
+
161
+ /**
162
+ * The write-refusal for what a path ALREADY holds, as opposed to what its
163
+ * extension means (see `assertNotDocumentEdit` for that half). Only the
164
+ * fallback reader answers here today: an extensionless file may hold anything,
165
+ * and `read_file` refuses binary content — so a write gate that never asked
166
+ * would let an agent overwrite bytes it was not allowed to read.
167
+ *
168
+ * Costs one read of the existing file, and only for readers that ask the
169
+ * question. A path with nothing at it is a CREATE: there is nothing to destroy.
170
+ * Returns the bytes it read (so a caller that needs the content next —
171
+ * `edit_file` — does not read the file a second time), or undefined when it
172
+ * had no reason to read or nothing existed.
173
+ */
174
+ async function assertNotBinaryOverwrite(
175
+ readers: FileReaderRegistry,
176
+ path: string,
177
+ fs: { readFile(p: string): Promise<string | Buffer> },
178
+ ): Promise<Buffer | undefined> {
179
+ const reader = readers.readerFor(path);
180
+ if (reader.editRefusalForExisting === undefined) return undefined;
181
+ let existing: Buffer;
182
+ try {
183
+ existing = asBytes(await fs.readFile(path));
184
+ } catch (err) {
185
+ // Only a MISSING file is a create (both raw Node errors and Mastra's
186
+ // FileNotFoundError carry code 'ENOENT'). Any other failure — permissions,
187
+ // I/O — means the existing content could not be inspected: propagate it
188
+ // rather than let the write destroy bytes the gate never saw.
189
+ if ((err as { code?: unknown } | null)?.code === 'ENOENT') return undefined; // nothing there yet
190
+ throw err;
191
+ }
192
+ const refusal = reader.editRefusalForExisting(existing, path);
193
+ if (refusal !== null) throw new ToolError(refusal, 400);
194
+ return existing;
195
+ }
196
+
113
197
  /** JS grep over the workspace tree (read methods only) — bounded by match + depth caps. */
114
198
  async function grepWalk(
115
199
  fs: LocalFilesystem,
@@ -120,6 +204,7 @@ async function grepWalk(
120
204
  depth: number,
121
205
  gate: ReadGate,
122
206
  recordOntologyRead: (path: string) => Promise<void>,
207
+ docs: DocGrepState,
123
208
  ): Promise<void> {
124
209
  if (out.length >= max || depth > 12) return;
125
210
  let entries;
@@ -136,19 +221,39 @@ async function grepWalk(
136
221
  if (e.name === '.git' || e.name === 'node_modules') continue;
137
222
  const p = dir ? `${dir}/${e.name}` : e.name;
138
223
  if (e.type === 'directory') {
139
- await grepWalk(fs, p, re, out, max, depth + 1, gate, recordOntologyRead);
224
+ await grepWalk(fs, p, re, out, max, depth + 1, gate, recordOntologyRead, docs);
140
225
  } else {
141
226
  // Opening a file under a named ontology is a read of that ontology — even
142
227
  // for a root-level grep that resolves to a neutral root. Record it so a
143
228
  // cross-ontology grep poisons later writes (closes the read-leak).
144
229
  await recordOntologyRead(p);
145
- let content;
230
+ // What grep searches is the file's reader's business: text content for
231
+ // the text reader (null on NUL bytes — binary is not searchable), the
232
+ // EXTRACTION for a document reader (marker line included, so the
233
+ // [slide N]/[sheet: …]/[page N] lines are themselves searchable and
234
+ // line numbers match what read_file returns), nothing for images.
235
+ const reader = docs.readers.readerFor(p);
236
+ let content: string | null;
146
237
  try {
147
- content = asText(await fs.readFile(p));
238
+ const bytes = asBytes(await fs.readFile(p));
239
+ content = reader.greppableText ? await reader.greppableText(bytes, p) : null;
240
+ if (content === null && reader instanceof DocumentReader) {
241
+ // Cold document (extraction not yet cached): cached ones above are
242
+ // free, extracting draws on the per-walk budget (see
243
+ // UNCACHED_DOCS_PER_GREP) — beyond it the walk skips and counts.
244
+ if (docs.uncachedBudget <= 0) {
245
+ docs.skippedUncached++;
246
+ continue;
247
+ }
248
+ docs.uncachedBudget--;
249
+ const res = await reader.read(bytes, p);
250
+ // A non-text outcome is a corrupt document — nothing searchable.
251
+ content = res.kind === 'text' ? res.text : null;
252
+ }
148
253
  } catch {
149
254
  continue;
150
255
  }
151
- if (content.includes(String.fromCharCode(0))) continue; // skip binary (NUL)
256
+ if (content === null) continue;
152
257
  const lines = content.split('\n');
153
258
  for (let i = 0; i < lines.length && out.length < max; i++) {
154
259
  re.lastIndex = 0;
@@ -173,12 +278,21 @@ export function registerWorkspaceTools(
173
278
  toolAuth: RequestHandler,
174
279
  toolHandler: ToolHandlerFactory,
175
280
  spillStore: SpillStore,
281
+ docExtract: DocExtractService,
176
282
  accessControl: IAccessControl,
177
283
  kbDirName: string,
178
284
  sessionOntologyGate: SessionOntologyGate,
179
285
  writePolicy: IRoutineWritePolicy,
180
286
  sessionSink: ISessionSink,
181
287
  ): void {
288
+ /**
289
+ * The one extension→reader registry every read-shaped decision routes
290
+ * through: read_file dispatches on it, grep asks it for searchable text,
291
+ * and the write-refusal consults its `textEditable`. Built once per mount
292
+ * around the shared extraction cache.
293
+ */
294
+ const readers = createFileReaderRegistry(docExtract);
295
+
182
296
  /** Build the per-call read gate from the tool's branch input + caller identity. */
183
297
  const readGateFor = (branch: string, ctx: ToolContext): ReadGate => ({
184
298
  accessControl,
@@ -274,7 +388,7 @@ export function registerWorkspaceTools(
274
388
  mount({
275
389
  name: 'read_file',
276
390
  description:
277
- 'Read a workspace file as text. Returns `{ path, content }`. Optional `offset`/`limit` slice the content (characters for a file, bytes for a `__tool_chain_spill__/…` ref) — use them to page through large files or a `call_tool_chain` spill rather than reading multi-MB in full. A spill ref is workspace-independent: `branch` is ignored for it.' +
391
+ 'Read a workspace file as text. Returns `{ path, content }`. Images (.png/.jpg/.jpeg/.gif/.webp) return the IMAGE ITSELF as native MCP image content (plus a one-line text note naming the file), so you can look at the picture — up to 3.5 MB of raw image data; a larger image gets an honest refusal asking for a locally downscaled copy or a smaller export (`.svg` is text and reads as text). Images come back only on a DIRECT call: inside `call_tool_chain` an image read yields an `{ image_omitted, note }` stub instead. Office and OpenDocument files (.docx/.pptx/.xlsx, .odt/.odp/.ods) and PDFs return their EXTRACTED text under an honest `[extracted text of …]` header, with `[slide N]`/`[sheet: Name]`/`[page N]` markers — the extraction is READ-ONLY (layout/images omitted; such files cannot be edited as text, only replaced by uploading a new version). Email files (.eml/.msg) return their EXTRACTED text the same way: a `[from]`/`[to]`/`[subject]`/`[date]` header block, the body (plain-text part preferred; an HTML-only body is stripped to text), and an `[attachments]` name list — attachments are listed, never extracted. Other binary files return a one-line description instead of raw bytes. Optional `offset`/`limit` slice the content (characters for a file, bytes for a `__tool_chain_spill__/…` ref; ignored for an image) — use them to page through large files or a `call_tool_chain` spill rather than reading multi-MB in full. A spill ref is workspace-independent: `branch` is ignored for it.' +
278
392
  ONTOLOGY_BOUNDARY_NOTE,
279
393
  inputs: {
280
394
  type: 'object',
@@ -304,7 +418,30 @@ export function registerWorkspaceTools(
304
418
  await recordOntologyRead(sessionOntologyGate, ctx, p);
305
419
  await assertCanRead(readGateFor(a.branch as string, ctx), p);
306
420
  const fs = await ctx.getFilesystem(a.branch as string);
307
- const content = asText(await fs.readFile(p));
421
+ // Reading (extraction, image and binary handling included) happens AFTER
422
+ // the access gate and the ontology-read recording above — a document
423
+ // read is still a KB read. ONE registry dispatch picks the reader by
424
+ // extension; everything below just maps its ReadResult onto the tool's
425
+ // result shape.
426
+ const result = await readers.readerFor(p).read(asBytes(await fs.readFile(p)), p);
427
+ // Images return the picture itself as an MCP image content block, so a
428
+ // multimodal model SEES it. The handler returns the `McpImageResult`
429
+ // sentinel; the MCP result shaping (`toCallToolResult` in
430
+ // platform-mcp-core) turns it into `content: [image, text-note]`. The
431
+ // declared `outputs` schema (`{ path, content }`) intentionally does NOT
432
+ // cover this shape: `outputs` is advisory documentation — never enforced
433
+ // at the route, and not advertised over MCP (`toListedTool` exposes
434
+ // `inputSchema` only) — and the image result is replaced wholesale by
435
+ // content blocks before any client could try to validate it, so the
436
+ // schema keeps describing the text path it has always described.
437
+ // `offset`/`limit` are meaningless on a picture and are ignored.
438
+ if (result.kind === 'image') {
439
+ return mcpImageResult(result.data, result.mimeType, result.note);
440
+ }
441
+ // Text and refusals alike land in `content` — a refusal (corrupt
442
+ // document, unreadable binary, oversized image) IS the file's honest
443
+ // textual answer, sliced like any other content.
444
+ const content = result.kind === 'text' ? result.text : result.message;
308
445
  const start = offset && offset > 0 ? offset : 0;
309
446
  const sliced = offset !== undefined || limit !== undefined
310
447
  ? content.slice(start, limit !== undefined ? start + limit : undefined)
@@ -388,7 +525,7 @@ export function registerWorkspaceTools(
388
525
  mount({
389
526
  name: 'grep',
390
527
  description:
391
- 'Regex content search across the workspace. Returns `{ matches: [{ path, line, text }] }` (capped). Use to find where something is defined/referenced.' +
528
+ 'Regex content search across the workspace. Returns `{ matches: [{ path, line, text }] }` (capped). Use to find where something is defined/referenced. Searches INSIDE Office and OpenDocument files (.docx/.pptx/.xlsx, .odt/.odp/.ods), PDFs and email files (.eml/.msg) via their extracted text — matches there carry the extraction\'s line numbers, and the `[slide N]`/`[sheet: Name]`/`[page N]`/`[from]`/`[subject]` marker lines locate them; a bounded number of not-yet-extracted documents is extracted per call, and the result notes how many were skipped (re-run to cover them).' +
392
529
  ONTOLOGY_BOUNDARY_NOTE,
393
530
  inputs: {
394
531
  type: 'object',
@@ -416,6 +553,7 @@ export function registerWorkspaceTools(
416
553
  },
417
554
  },
418
555
  truncated: { type: 'boolean', description: 'True if the match cap was hit and results may be incomplete.' },
556
+ note: str('Present when some documents (office/PDF/email files) were not searched because their text was not yet extracted and the per-call extraction budget ran out — re-run grep to cover them.'),
419
557
  },
420
558
  required: ['matches', 'truncated'],
421
559
  },
@@ -435,6 +573,7 @@ export function registerWorkspaceTools(
435
573
  await recordOntologyRead(sessionOntologyGate, ctx, searchRoot);
436
574
  const out: { path: string; line: number; text: string }[] = [];
437
575
  const max = typeof a.max_results === 'number' ? Math.min(a.max_results, 1000) : 200;
576
+ const docs: DocGrepState = { readers, uncachedBudget: UNCACHED_DOCS_PER_GREP, skippedUncached: 0 };
438
577
  await grepWalk(
439
578
  await ctx.getFilesystem(a.branch as string),
440
579
  searchRoot,
@@ -444,8 +583,20 @@ export function registerWorkspaceTools(
444
583
  0,
445
584
  readGateFor(a.branch as string, ctx),
446
585
  (p) => recordOntologyRead(sessionOntologyGate, ctx, p),
586
+ docs,
447
587
  );
448
- return { matches: out, truncated: out.length >= max };
588
+ return {
589
+ matches: out,
590
+ truncated: out.length >= max,
591
+ ...(docs.skippedUncached > 0
592
+ ? {
593
+ note:
594
+ `${docs.skippedUncached} document(s) (office/PDF/email files) were not searched: their text was not yet ` +
595
+ `extracted and this call's extraction budget (${UNCACHED_DOCS_PER_GREP}) ran out. Re-run the ` +
596
+ 'same grep to extract and search the next batch.',
597
+ }
598
+ : {}),
599
+ };
449
600
  },
450
601
  });
451
602
 
@@ -453,7 +604,7 @@ export function registerWorkspaceTools(
453
604
  mount({
454
605
  name: 'write_file',
455
606
  description:
456
- 'Write (create or overwrite) a workspace file. The change is committed + pushed as you. Returns `{ path, bytes }`.' +
607
+ 'Write (create or overwrite) a workspace file. The change is committed + pushed as you. Returns `{ path, bytes }`. Refuses document formats: Office/OpenDocument files and PDFs (.docx/.pptx/.xlsx/.odt/.odp/.ods/.pdf) and email files (.eml/.msg), whose reads are text EXTRACTIONS that cannot round-trip, and legacy binary Office files (.doc/.ppt/.xls), which cannot be extracted at all; replace such a file by uploading a new version instead.' +
457
608
  ONTOLOGY_BOUNDARY_NOTE,
458
609
  inputs: {
459
610
  type: 'object',
@@ -473,6 +624,7 @@ export function registerWorkspaceTools(
473
624
  },
474
625
  write: true,
475
626
  handler: async (a, ctx: ToolContext) => {
627
+ assertNotDocumentEdit(readers, a.path as string);
476
628
  // NB: this is a no-op for chat + `ontology_ingest` — it only bites when a
477
629
  // routine executor has explicitly restricted THIS session's `ctx.sessionId`
478
630
  // (today only `watchlist_check`, to `.html`). Unrestricted sessions pass straight
@@ -480,6 +632,7 @@ export function registerWorkspaceTools(
480
632
  writePolicy.assertPathWritable(ctx.sessionId, a.path as string);
481
633
  await assertOntologyWriteAllowed(sessionOntologyGate, ctx, a.path as string);
482
634
  const fs = await ctx.getFilesystem(a.branch as string);
635
+ await assertNotBinaryOverwrite(readers, a.path as string, fs);
483
636
  await fs.writeFile(a.path as string, a.content as string);
484
637
  return { path: a.path, bytes: Buffer.byteLength(a.content as string, 'utf8') };
485
638
  },
@@ -491,7 +644,9 @@ export function registerWorkspaceTools(
491
644
  'Batch-write many files in ONE commit — far faster than calling write_file once per file when ' +
492
645
  'creating many files at once (e.g. seeding a knowledge base). Each entry is `{ path, content }`; all ' +
493
646
  'are created/overwritten and committed + pushed together as you. Prefer this over many write_file ' +
494
- 'calls. All files must be in the SAME ontology (the boundary below applies to the batch). Returns `{ count }`.' +
647
+ 'calls. All files must be in the SAME ontology (the boundary below applies to the batch). Refuses document formats: ' +
648
+ 'Office/OpenDocument files and PDFs (.docx/.pptx/.xlsx/.odt/.odp/.ods/.pdf) and email files (.eml/.msg), whose reads are text EXTRACTIONS that cannot ' +
649
+ 'round-trip, and legacy binary Office files (.doc/.ppt/.xls); replace such a file by uploading a new version instead. Returns `{ count }`.' +
495
650
  ONTOLOGY_BOUNDARY_NOTE,
496
651
  inputs: {
497
652
  type: 'object',
@@ -523,13 +678,16 @@ export function registerWorkspaceTools(
523
678
  if (files.length === 0) return { count: 0 };
524
679
  // Gate every path first (records ontology touches; a cross-ontology batch
525
680
  // is blocked exactly like the per-file write tools).
681
+ for (const f of files) assertNotDocumentEdit(readers, f.path);
526
682
  for (const f of files) writePolicy.assertPathWritable(ctx.sessionId, f.path);
527
683
  for (const f of files) await assertOntologyWriteAllowed(sessionOntologyGate, ctx, f.path);
528
684
  // `write: true` guarantees a LockingFilesystem here; `writeFiles` lands the
529
685
  // whole batch as one commit. Structural cast avoids a workflow-internal import.
530
686
  const fs = (await ctx.getFilesystem(a.branch as string)) as unknown as {
531
687
  writeFiles(writes: { path: string; content: string }[], summary: string): Promise<void>;
688
+ readFile(p: string): Promise<string | Buffer>;
532
689
  };
690
+ for (const f of files) await assertNotBinaryOverwrite(readers, f.path, fs);
533
691
  await fs.writeFiles(
534
692
  files.map((f) => ({ path: f.path, content: f.content })),
535
693
  `Write ${files.length} file(s)`,
@@ -541,7 +699,7 @@ export function registerWorkspaceTools(
541
699
  mount({
542
700
  name: 'edit_file',
543
701
  description:
544
- 'Replace an exact string in a workspace file. `old_string` must appear exactly once unless `replace_all`. Committed + pushed as you.' +
702
+ 'Replace an exact string in a workspace file. `old_string` must appear exactly once unless `replace_all`. Committed + pushed as you. Refuses document formats: Office/OpenDocument files and PDFs (.docx/.pptx/.xlsx/.odt/.odp/.ods/.pdf) and email files (.eml/.msg), whose reads are text EXTRACTIONS that cannot round-trip, and legacy binary Office files (.doc/.ppt/.xls), which cannot be extracted at all; replace such a file by uploading a new version instead.' +
545
703
  ONTOLOGY_BOUNDARY_NOTE,
546
704
  inputs: {
547
705
  type: 'object',
@@ -563,13 +721,17 @@ export function registerWorkspaceTools(
563
721
  },
564
722
  write: true,
565
723
  handler: async (a, ctx: ToolContext) => {
724
+ assertNotDocumentEdit(readers, a.path as string);
566
725
  writePolicy.assertPathWritable(ctx.sessionId, a.path as string);
567
726
  await assertOntologyWriteAllowed(sessionOntologyGate, ctx, a.path as string);
568
727
  const fs = await ctx.getFilesystem(a.branch as string);
569
728
  const path = a.path as string;
570
729
  const oldStr = a.old_string as string;
571
730
  const newStr = a.new_string as string;
572
- const content = asText(await fs.readFile(path));
731
+ // The overwrite gate already read the file when its reader asked the
732
+ // binary question — reuse those bytes instead of reading twice.
733
+ const existing = await assertNotBinaryOverwrite(readers, path, fs);
734
+ const content = asText(existing ?? (await fs.readFile(path)));
573
735
  const count = oldStr ? content.split(oldStr).length - 1 : 0;
574
736
  if (count === 0) throw new ToolError('old_string not found in the file.', 400);
575
737
  if (count > 1 && a.replace_all !== true) {