@bevel-software/platform-core-backend 0.11.2 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/THIRD-PARTY-NOTICES.md +1163 -425
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +2 -0
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +5 -0
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  13. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  14. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  15. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  16. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  17. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  18. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  19. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  20. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  21. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  22. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  23. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  24. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  25. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  26. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  27. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  28. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  29. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  30. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  31. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  32. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  33. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  34. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  35. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  36. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  37. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  38. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  39. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  40. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  41. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  42. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  43. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  44. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  45. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  46. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  47. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  48. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  49. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  50. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  51. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  52. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  53. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  54. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  55. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  56. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  57. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  58. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  59. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  60. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  61. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  62. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  63. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  64. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  65. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  66. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  67. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  68. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  69. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  70. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  71. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  72. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  73. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  74. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  75. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  76. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  77. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  78. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  79. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  80. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  81. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  82. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  83. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  84. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  85. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  86. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  87. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  88. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  89. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  90. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  91. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  92. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  93. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  94. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  95. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  96. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  97. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  98. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  99. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  100. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  101. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  102. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  103. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  104. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  105. package/dist/modules/workspace/workspace.tools.js +158 -15
  106. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  107. package/package.json +9 -4
  108. package/src/core/create-core-server.ts +1 -1
  109. package/src/core/create-core-services.ts +6 -0
  110. package/src/core-config.ts +9 -0
  111. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  112. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  113. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  114. package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
  115. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  116. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  117. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  118. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  119. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  120. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  121. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  122. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  123. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  124. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  125. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  126. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  127. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  128. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  129. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  130. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  131. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  132. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  133. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  134. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  135. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  136. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  137. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  138. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  139. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  140. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  141. package/src/modules/workspace/workspace.tools.ts +174 -12
@@ -0,0 +1,61 @@
1
+ import type { ExtractedDoc } from './doc-extract.types.js';
2
+ /**
3
+ * Git BLOB sha of `bytes` — sha1 over `"blob <len>\0" + bytes`, exactly what
4
+ * `git hash-object` computes. Chosen as the cache key because the workspace is
5
+ * a git clone: for a clean tracked file this equals the sha `git ls-files -s`
6
+ * would report, WITHOUT spawning git — and because it is computed from the
7
+ * bytes actually read, it stays correct (it just stops matching the index)
8
+ * when the file is untracked or dirty. One code path, no fallback branch.
9
+ */
10
+ export declare function gitBlobSha(bytes: Buffer): string;
11
+ /**
12
+ * On-disk cache of document extractions. The KEY is the caller's business —
13
+ * `DocExtractService` passes git blob sha + extension (content hash, so a
14
+ * re-upload of identical bytes, or the same document on another branch or
15
+ * path, hits the same entry; extension, so identical bytes under another
16
+ * FORMAT never return the wrong extractor's output). One JSON file per entry
17
+ * (`<key>.json` holding the `{ summary, text }`), under a dedicated root
18
+ * that — like the spill
19
+ * store's — sits BESIDE the workspaces root, never inside a workspace and
20
+ * never committed.
21
+ *
22
+ * Bounding: writes best-effort prune OLDEST-MTIME entries until the total
23
+ * size fits `maxTotalBytes` (simple LRU-ish eviction; reads don't touch
24
+ * mtime, so it is closer to FIFO — good enough for a cache whose entries are
25
+ * cheap to rebuild). The full scan is deferred until the writes since the
26
+ * last one could plausibly have reached the bound (see `writtenSinceScan`).
27
+ * Every filesystem error here is swallowed: the cache is an accelerator,
28
+ * never a reason for a read to fail.
29
+ */
30
+ export declare class DocExtractionCache {
31
+ private readonly root;
32
+ private readonly maxTotalBytes;
33
+ constructor(root: string, maxTotalBytes?: number);
34
+ /** The cached extraction for `sha`, or undefined on miss/corrupt entry. */
35
+ get(sha: string): Promise<ExtractedDoc | undefined>;
36
+ /**
37
+ * Bytes written since the last full scan. The scan costs a `readdir` plus a
38
+ * `stat` per entry, and running it on EVERY cold extraction made a cache that
39
+ * is nowhere near its bound pay for the bound anyway. Writes are accumulated
40
+ * instead and the scan runs once they could plausibly have reached it.
41
+ */
42
+ private writtenSinceScan;
43
+ /** Total size the last scan found (after eviction); undefined before the first scan, so the first write scans. */
44
+ private totalAtLastScan;
45
+ /** The prune in flight, so concurrent puts share ONE scan instead of racing their own. */
46
+ private pruning;
47
+ /** Store an extraction under `sha`; prunes towards the size bound. Never throws. */
48
+ put(sha: string, doc: ExtractedDoc): Promise<void>;
49
+ /**
50
+ * The file backing `key`, always INSIDE the cache root.
51
+ *
52
+ * The key is built from a content hash and an extension taken from a
53
+ * workspace path, so a path spelling `../` — or a Windows backslash — would
54
+ * otherwise resolve outside the root and let a read or a write reach an
55
+ * arbitrary file. Only the characters a real key uses survive.
56
+ */
57
+ private entryPath;
58
+ /** Delete oldest-mtime entries until the total size fits the bound. */
59
+ private prune;
60
+ }
61
+ //# sourceMappingURL=extraction-cache.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extraction-cache.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extraction-cache.ts"],"names":[],"mappings":"AAGA,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,wBAAwB,CAAC;AAE3D;;;;;;;GAOG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,GAAG,MAAM,CAEhD;AAKD;;;;;;;;;;;;;;;;;;GAkBG;AACH,qBAAa,kBAAkB;IAE3B,OAAO,CAAC,QAAQ,CAAC,IAAI;IACrB,OAAO,CAAC,QAAQ,CAAC,aAAa;gBADb,IAAI,EAAE,MAAM,EACZ,aAAa,GAAE,MAAgC;IAGlE,2EAA2E;IACrE,GAAG,CAAC,GAAG,EAAE,MAAM,GAAG,OAAO,CAAC,YAAY,GAAG,SAAS,CAAC;IAgBzD;;;;;OAKG;IACH,OAAO,CAAC,gBAAgB,CAAK;IAE7B,kHAAkH;IAClH,OAAO,CAAC,eAAe,CAAqB;IAE5C,0FAA0F;IAC1F,OAAO,CAAC,OAAO,CAA4B;IAE3C,oFAAoF;IAC9E,GAAG,CAAC,GAAG,EAAE,MAAM,EAAE,GAAG,EAAE,YAAY,GAAG,OAAO,CAAC,IAAI,CAAC;IAoBxD;;;;;;;OAOG;IACH,OAAO,CAAC,SAAS;IAKjB,uEAAuE;YACzD,KAAK;CA8BpB"}
@@ -0,0 +1,135 @@
1
+ import { createHash } from 'node:crypto';
2
+ import { promises as fs } from 'node:fs';
3
+ import path from 'node:path';
4
+ /**
5
+ * Git BLOB sha of `bytes` — sha1 over `"blob <len>\0" + bytes`, exactly what
6
+ * `git hash-object` computes. Chosen as the cache key because the workspace is
7
+ * a git clone: for a clean tracked file this equals the sha `git ls-files -s`
8
+ * would report, WITHOUT spawning git — and because it is computed from the
9
+ * bytes actually read, it stays correct (it just stops matching the index)
10
+ * when the file is untracked or dirty. One code path, no fallback branch.
11
+ */
12
+ export function gitBlobSha(bytes) {
13
+ return createHash('sha1').update(`blob ${bytes.length}\0`).update(bytes).digest('hex');
14
+ }
15
+ /** Default size bound for the on-disk extraction cache. */
16
+ const DEFAULT_MAX_TOTAL_BYTES = 512 * 1024 * 1024; // 512MB
17
+ /**
18
+ * On-disk cache of document extractions. The KEY is the caller's business —
19
+ * `DocExtractService` passes git blob sha + extension (content hash, so a
20
+ * re-upload of identical bytes, or the same document on another branch or
21
+ * path, hits the same entry; extension, so identical bytes under another
22
+ * FORMAT never return the wrong extractor's output). One JSON file per entry
23
+ * (`<key>.json` holding the `{ summary, text }`), under a dedicated root
24
+ * that — like the spill
25
+ * store's — sits BESIDE the workspaces root, never inside a workspace and
26
+ * never committed.
27
+ *
28
+ * Bounding: writes best-effort prune OLDEST-MTIME entries until the total
29
+ * size fits `maxTotalBytes` (simple LRU-ish eviction; reads don't touch
30
+ * mtime, so it is closer to FIFO — good enough for a cache whose entries are
31
+ * cheap to rebuild). The full scan is deferred until the writes since the
32
+ * last one could plausibly have reached the bound (see `writtenSinceScan`).
33
+ * Every filesystem error here is swallowed: the cache is an accelerator,
34
+ * never a reason for a read to fail.
35
+ */
36
+ export class DocExtractionCache {
37
+ root;
38
+ maxTotalBytes;
39
+ constructor(root, maxTotalBytes = DEFAULT_MAX_TOTAL_BYTES) {
40
+ this.root = root;
41
+ this.maxTotalBytes = maxTotalBytes;
42
+ }
43
+ /** The cached extraction for `sha`, or undefined on miss/corrupt entry. */
44
+ async get(sha) {
45
+ try {
46
+ const parsed = JSON.parse(await fs.readFile(this.entryPath(sha), 'utf8'));
47
+ if (typeof parsed === 'object' && parsed !== null &&
48
+ typeof parsed.summary === 'string' &&
49
+ typeof parsed.text === 'string') {
50
+ return { summary: parsed.summary, text: parsed.text };
51
+ }
52
+ return undefined;
53
+ }
54
+ catch {
55
+ return undefined; // miss, unreadable or corrupt — treated identically
56
+ }
57
+ }
58
+ /**
59
+ * Bytes written since the last full scan. The scan costs a `readdir` plus a
60
+ * `stat` per entry, and running it on EVERY cold extraction made a cache that
61
+ * is nowhere near its bound pay for the bound anyway. Writes are accumulated
62
+ * instead and the scan runs once they could plausibly have reached it.
63
+ */
64
+ writtenSinceScan = 0;
65
+ /** Total size the last scan found (after eviction); undefined before the first scan, so the first write scans. */
66
+ totalAtLastScan;
67
+ /** The prune in flight, so concurrent puts share ONE scan instead of racing their own. */
68
+ pruning;
69
+ /** Store an extraction under `sha`; prunes towards the size bound. Never throws. */
70
+ async put(sha, doc) {
71
+ try {
72
+ await fs.mkdir(this.root, { recursive: true });
73
+ const payload = JSON.stringify({ summary: doc.summary, text: doc.text });
74
+ await fs.writeFile(this.entryPath(sha), payload, 'utf8');
75
+ this.writtenSinceScan += Buffer.byteLength(payload);
76
+ if (this.totalAtLastScan === undefined ||
77
+ this.totalAtLastScan + this.writtenSinceScan > this.maxTotalBytes) {
78
+ this.pruning ??= this.prune().finally(() => {
79
+ this.pruning = undefined;
80
+ });
81
+ await this.pruning;
82
+ }
83
+ }
84
+ catch {
85
+ // cache write failed — the extraction still returns; next read re-extracts
86
+ }
87
+ }
88
+ /**
89
+ * The file backing `key`, always INSIDE the cache root.
90
+ *
91
+ * The key is built from a content hash and an extension taken from a
92
+ * workspace path, so a path spelling `../` — or a Windows backslash — would
93
+ * otherwise resolve outside the root and let a read or a write reach an
94
+ * arbitrary file. Only the characters a real key uses survive.
95
+ */
96
+ entryPath(key) {
97
+ const safe = key.replace(/[^A-Za-z0-9._-]/g, '_');
98
+ return path.join(this.root, `${safe}.json`);
99
+ }
100
+ /** Delete oldest-mtime entries until the total size fits the bound. */
101
+ async prune() {
102
+ // Snapshot FIRST: a write counted before the scan starts is on disk when
103
+ // `readdir` runs, so the scan settles exactly those bytes. A put that
104
+ // lands DURING the scan keeps its count (resetting to zero discarded it,
105
+ // and a later put then trusted a total the scan never saw), so the next
106
+ // put still prunes instead of leaving the cache above the bound.
107
+ const scanned = this.writtenSinceScan;
108
+ const entries = [];
109
+ let total = 0;
110
+ for (const name of await fs.readdir(this.root)) {
111
+ try {
112
+ const st = await fs.stat(path.join(this.root, name));
113
+ if (!st.isFile())
114
+ continue;
115
+ entries.push({ p: path.join(this.root, name), size: st.size, mtime: st.mtimeMs });
116
+ total += st.size;
117
+ }
118
+ catch {
119
+ // vanished mid-scan — ignore
120
+ }
121
+ }
122
+ if (total > this.maxTotalBytes) {
123
+ entries.sort((a, b) => a.mtime - b.mtime);
124
+ for (const e of entries) {
125
+ if (total <= this.maxTotalBytes)
126
+ break;
127
+ await fs.rm(e.p, { force: true });
128
+ total -= e.size;
129
+ }
130
+ }
131
+ this.totalAtLastScan = total;
132
+ this.writtenSinceScan -= scanned;
133
+ }
134
+ }
135
+ //# sourceMappingURL=extraction-cache.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extraction-cache.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extraction-cache.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AACzC,OAAO,EAAE,QAAQ,IAAI,EAAE,EAAE,MAAM,SAAS,CAAC;AACzC,OAAO,IAAI,MAAM,WAAW,CAAC;AAG7B;;;;;;;GAOG;AACH,MAAM,UAAU,UAAU,CAAC,KAAa;IACtC,OAAO,UAAU,CAAC,MAAM,CAAC,CAAC,MAAM,CAAC,QAAQ,KAAK,CAAC,MAAM,IAAI,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC;AACzF,CAAC;AAED,2DAA2D;AAC3D,MAAM,uBAAuB,GAAG,GAAG,GAAG,IAAI,GAAG,IAAI,CAAC,CAAC,QAAQ;AAE3D;;;;;;;;;;;;;;;;;;GAkBG;AACH,MAAM,OAAO,kBAAkB;IAEV;IACA;IAFnB,YACmB,IAAY,EACZ,gBAAwB,uBAAuB;QAD/C,SAAI,GAAJ,IAAI,CAAQ;QACZ,kBAAa,GAAb,aAAa,CAAkC;IAC/D,CAAC;IAEJ,2EAA2E;IAC3E,KAAK,CAAC,GAAG,CAAC,GAAW;QACnB,IAAI,CAAC;YACH,MAAM,MAAM,GAAY,IAAI,CAAC,KAAK,CAAC,MAAM,EAAE,CAAC,QAAQ,CAAC,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,EAAE,MAAM,CAAC,CAAC,CAAC;YACnF,IACE,OAAO,MAAM,KAAK,QAAQ,IAAI,MAAM,KAAK,IAAI;gBAC7C,OAAQ,MAAuB,CAAC,OAAO,KAAK,QAAQ;gBACpD,OAAQ,MAAuB,CAAC,IAAI,KAAK,QAAQ,EACjD,CAAC;gBACD,OAAO,EAAE,OAAO,EAAG,MAAuB,CAAC,OAAO,EAAE,IAAI,EAAG,MAAuB,CAAC,IAAI,EAAE,CAAC;YAC5F,CAAC;YACD,OAAO,SAAS,CAAC;QACnB,CAAC;QAAC,MAAM,CAAC;YACP,OAAO,SAAS,CAAC,CAAC,oDAAoD;QACxE,CAAC;IACH,CAAC;IAED;;;;;OAKG;IACK,gBAAgB,GAAG,CAAC,CAAC;IAE7B,kHAAkH;IAC1G,eAAe,CAAqB;IAE5C,0FAA0F;IAClF,OAAO,CAA4B;IAE3C,oFAAoF;IACpF,KAAK,CAAC,GAAG,CAAC,GAAW,EAAE,GAAiB;QACtC,IAAI,CAAC;YACH,MAAM,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,IAAI,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAC/C,MAAM,OAAO,GAAG,IAAI,CAAC,SAAS,CAAC,EAAE,OAAO,EAAE,GAAG,CAAC,OAAO,EAAE,IAAI,EAAE,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC;YACzE,MAAM,EAAE,CAAC,SAAS,CAAC,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,EAAE,OAAO,EAAE,MAAM,CAAC,CAAC;YACzD,IAAI,CAAC,gBAAgB,IAAI,MAAM,CAAC,UAAU,CAAC,OAAO,CAAC,CAAC;YACpD,IACE,IAAI,CAAC,eAAe,KAAK,SAAS;gBAClC,IAAI,CAAC,eAAe,GAAG,IAAI,CAAC,gBAAgB,GAAG,IAAI,CAAC,aAAa,EACjE,CAAC;gBACD,IAAI,CAAC,OAAO,KAAK,IAAI,CAAC,KAAK,EAAE,CAAC,OAAO,CAAC,GAAG,EAAE;oBACzC,IAAI,CAAC,OAAO,GAAG,SAAS,CAAC;gBAC3B,CAAC,CAAC,CAAC;gBACH,MAAM,IAAI,CAAC,OAAO,CAAC;YACrB,CAAC;QACH,CAAC;QAAC,MAAM,CAAC;YACP,2EAA2E;QAC7E,CAAC;IACH,CAAC;IAED;;;;;;;OAOG;IACK,SAAS,CAAC,GAAW;QAC3B,MAAM,IAAI,GAAG,GAAG,CAAC,OAAO,CAAC,kBAAkB,EAAE,GAAG,CAAC,CAAC;QAClD,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,GAAG,IAAI,OAAO,CAAC,CAAC;IAC9C,CAAC;IAED,uEAAuE;IAC/D,KAAK,CAAC,KAAK;QACjB,yEAAyE;QACzE,sEAAsE;QACtE,yEAAyE;QACzE,wEAAwE;QACxE,iEAAiE;QACjE,MAAM,OAAO,GAAG,IAAI,CAAC,gBAAgB,CAAC;QACtC,MAAM,OAAO,GAAsD,EAAE,CAAC;QACtE,IAAI,KAAK,GAAG,CAAC,CAAC;QACd,KAAK,MAAM,IAAI,IAAI,MAAM,EAAE,CAAC,OAAO,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YAC/C,IAAI,CAAC;gBACH,MAAM,EAAE,GAAG,MAAM,EAAE,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,IAAI,CAAC,CAAC,CAAC;gBACrD,IAAI,CAAC,EAAE,CAAC,MAAM,EAAE;oBAAE,SAAS;gBAC3B,OAAO,CAAC,IAAI,CAAC,EAAE,CAAC,EAAE,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,IAAI,CAAC,EAAE,IAAI,EAAE,EAAE,CAAC,IAAI,EAAE,KAAK,EAAE,EAAE,CAAC,OAAO,EAAE,CAAC,CAAC;gBAClF,KAAK,IAAI,EAAE,CAAC,IAAI,CAAC;YACnB,CAAC;YAAC,MAAM,CAAC;gBACP,6BAA6B;YAC/B,CAAC;QACH,CAAC;QACD,IAAI,KAAK,GAAG,IAAI,CAAC,aAAa,EAAE,CAAC;YAC/B,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC,CAAC;YAC1C,KAAK,MAAM,CAAC,IAAI,OAAO,EAAE,CAAC;gBACxB,IAAI,KAAK,IAAI,IAAI,CAAC,aAAa;oBAAE,MAAM;gBACvC,MAAM,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,EAAE,EAAE,KAAK,EAAE,IAAI,EAAE,CAAC,CAAC;gBAClC,KAAK,IAAI,CAAC,CAAC,IAAI,CAAC;YAClB,CAAC;QACH,CAAC;QACD,IAAI,CAAC,eAAe,GAAG,KAAK,CAAC;QAC7B,IAAI,CAAC,gBAAgB,IAAI,OAAO,CAAC;IACnC,CAAC;CACF"}
@@ -0,0 +1,76 @@
1
+ /**
2
+ * What reading a file produces, before the tool layer shapes it for MCP:
3
+ * text (the honest `[extracted text of …]` marker is ALREADY prepended for
4
+ * extractions, so line numbers match between read_file and grep), an image
5
+ * payload (the tool layer wraps it in the `McpImageResult` sentinel), or a
6
+ * refusal — legacy formats, oversized images, unreadable binary, corrupt
7
+ * documents — whose message is returned as the file's text content.
8
+ */
9
+ export type ReadResult = {
10
+ kind: 'text';
11
+ text: string;
12
+ } | {
13
+ kind: 'image';
14
+ data: string;
15
+ mimeType: string;
16
+ note: string;
17
+ } | {
18
+ kind: 'refusal';
19
+ message: string;
20
+ };
21
+ /**
22
+ * `path` as interpolated into a ONE-LINE notice or refusal: CR/LF are shown
23
+ * as escapes rather than obeyed (the same rule as `extractionMarker`), so a
24
+ * filename cannot forge extra output lines.
25
+ */
26
+ export declare function displayPath(path: string): string;
27
+ /**
28
+ * Any text interpolated into a ONE-LINE notice — a path, or an extractor's
29
+ * own failure message, which quotes bytes from the file and can therefore
30
+ * carry CR/LF of the document's choosing.
31
+ */
32
+ export declare const oneLine: typeof displayPath;
33
+ /** A per-format file reader. Register implementations in `createFileReaderRegistry`. */
34
+ export interface FileReader {
35
+ /** The extensions this reader owns — lowercase, with the dot. Empty for the default (fallback) reader. */
36
+ readonly extensions: readonly string[];
37
+ /** Read `bytes` (read at `path`) into what the read tools return. Must not throw for bad file content. */
38
+ read(bytes: Buffer, path: string): Promise<ReadResult>;
39
+ /**
40
+ * Text `grep` may search, or null when there is nothing (cheaply) searchable.
41
+ * Default readers: the decoded text content (TextReader; null on NUL bytes) /
42
+ * the CACHED extraction (DocumentReader — cold extraction is grep's call,
43
+ * under its per-walk budget) / absent = never greppable (ImageReader).
44
+ */
45
+ greppableText?(bytes: Buffer, path: string): Promise<string | null>;
46
+ /** May the agent TEXT-editing tools (write_file/write_files/edit_file) touch this file? */
47
+ readonly textEditable: boolean;
48
+ /**
49
+ * Format-specific copy for the write-refusal thrown when `textEditable` is
50
+ * false (see `assertNotDocumentEdit` in workspace.tools.ts). Absent = the
51
+ * generic extracted-text/round-trip explanation.
52
+ */
53
+ editRefusal?(path: string): string;
54
+ /**
55
+ * Refusal for OVERWRITING what this file ALREADY holds, or null to allow.
56
+ * Consulted only when `textEditable` is true: that answer covers the
57
+ * FORMAT, this one covers the CONTENT. The fallback reader needs it because
58
+ * an extensionless file may hold anything — `read_file` refuses binary
59
+ * content, and a write gate that did not ask would let an agent overwrite
60
+ * bytes it was never allowed to see.
61
+ */
62
+ editRefusalForExisting?(bytes: Buffer, path: string): string | null;
63
+ }
64
+ /**
65
+ * Extension → reader, with a default fallback (the text reader). Built once
66
+ * per deployment by `createFileReaderRegistry`; `readerFor` is the ONE lookup
67
+ * every consumer routes through.
68
+ */
69
+ export declare class FileReaderRegistry {
70
+ private readonly fallback;
71
+ private readonly byExtension;
72
+ constructor(readers: readonly FileReader[], fallback: FileReader);
73
+ /** The reader owning `path`'s extension (lowercased by `fileExtension`), or the fallback. */
74
+ readerFor(path: string): FileReader;
75
+ }
76
+ //# sourceMappingURL=file-reader.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"file-reader.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/file-reader.ts"],"names":[],"mappings":"AAYA;;;;;;;GAOG;AACH,MAAM,MAAM,UAAU,GAClB;IAAE,IAAI,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAC9B;IAAE,IAAI,EAAE,OAAO,CAAC;IAAC,IAAI,EAAE,MAAM,CAAC;IAAC,QAAQ,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAC/D;IAAE,IAAI,EAAE,SAAS,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAAC;AAEzC;;;;GAIG;AACH,wBAAgB,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEhD;AAED;;;;GAIG;AACH,eAAO,MAAM,OAAO,oBAAc,CAAC;AAEnC,wFAAwF;AACxF,MAAM,WAAW,UAAU;IACzB,0GAA0G;IAC1G,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAC;IACvC,0GAA0G;IAC1G,IAAI,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,UAAU,CAAC,CAAC;IACvD;;;;;OAKG;IACH,aAAa,CAAC,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,GAAG,IAAI,CAAC,CAAC;IACpE,2FAA2F;IAC3F,QAAQ,CAAC,YAAY,EAAE,OAAO,CAAC;IAC/B;;;;OAIG;IACH,WAAW,CAAC,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAAC;IACnC;;;;;;;OAOG;IACH,sBAAsB,CAAC,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CAAC;CACrE;AAED;;;;GAIG;AACH,qBAAa,kBAAkB;IAK3B,OAAO,CAAC,QAAQ,CAAC,QAAQ;IAJ3B,OAAO,CAAC,QAAQ,CAAC,WAAW,CAAiC;gBAG3D,OAAO,EAAE,SAAS,UAAU,EAAE,EACb,QAAQ,EAAE,UAAU;IAgBvC,6FAA6F;IAC7F,SAAS,CAAC,IAAI,EAAE,MAAM,GAAG,UAAU;CAGpC"}
@@ -0,0 +1,55 @@
1
+ /**
2
+ * The FileReader contract: ONE interface every read-path consumer goes
3
+ * through, with per-extension implementations picked by a single registry
4
+ * lookup — the same shape as the frontend's renderer registry.
5
+ *
6
+ * Adding a file format is ONE new reader entry in `createFileReaderRegistry`
7
+ * (file-reader.registry.ts). The three consumers — `read_file`, `grep` and the
8
+ * write-refusal in workspace.tools.ts — dispatch through `readerFor` and never
9
+ * branch on extensions themselves.
10
+ */
11
+ import { fileExtension } from './doc-extract.types.js';
12
+ /**
13
+ * `path` as interpolated into a ONE-LINE notice or refusal: CR/LF are shown
14
+ * as escapes rather than obeyed (the same rule as `extractionMarker`), so a
15
+ * filename cannot forge extra output lines.
16
+ */
17
+ export function displayPath(path) {
18
+ return path.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
19
+ }
20
+ /**
21
+ * Any text interpolated into a ONE-LINE notice — a path, or an extractor's
22
+ * own failure message, which quotes bytes from the file and can therefore
23
+ * carry CR/LF of the document's choosing.
24
+ */
25
+ export const oneLine = displayPath;
26
+ /**
27
+ * Extension → reader, with a default fallback (the text reader). Built once
28
+ * per deployment by `createFileReaderRegistry`; `readerFor` is the ONE lookup
29
+ * every consumer routes through.
30
+ */
31
+ export class FileReaderRegistry {
32
+ fallback;
33
+ byExtension = new Map();
34
+ constructor(readers, fallback) {
35
+ this.fallback = fallback;
36
+ for (const reader of readers) {
37
+ for (const raw of reader.extensions) {
38
+ // Lookups arrive lowercased (`fileExtension`), so a key stored with any
39
+ // other casing could never match and its reader would silently fall back
40
+ // to the default — routing a .PDF-registered reader nowhere.
41
+ const ext = raw.toLowerCase();
42
+ // Duplicate claims fail LOUDLY: were the later registration to win
43
+ // silently, adding a format could disable an existing reader.
44
+ if (this.byExtension.has(ext))
45
+ throw new Error(`two file readers claim the extension "${ext}"`);
46
+ this.byExtension.set(ext, reader);
47
+ }
48
+ }
49
+ }
50
+ /** The reader owning `path`'s extension (lowercased by `fileExtension`), or the fallback. */
51
+ readerFor(path) {
52
+ return this.byExtension.get(fileExtension(path)) ?? this.fallback;
53
+ }
54
+ }
55
+ //# sourceMappingURL=file-reader.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"file-reader.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/file-reader.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AACH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAevD;;;;GAIG;AACH,MAAM,UAAU,WAAW,CAAC,IAAY;IACtC,OAAO,IAAI,CAAC,OAAO,CAAC,SAAS,EAAE,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC;AACtE,CAAC;AAED;;;;GAIG;AACH,MAAM,CAAC,MAAM,OAAO,GAAG,WAAW,CAAC;AAkCnC;;;;GAIG;AACH,MAAM,OAAO,kBAAkB;IAKV;IAJF,WAAW,GAAG,IAAI,GAAG,EAAsB,CAAC;IAE7D,YACE,OAA8B,EACb,QAAoB;QAApB,aAAQ,GAAR,QAAQ,CAAY;QAErC,KAAK,MAAM,MAAM,IAAI,OAAO,EAAE,CAAC;YAC7B,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,UAAU,EAAE,CAAC;gBACpC,wEAAwE;gBACxE,yEAAyE;gBACzE,6DAA6D;gBAC7D,MAAM,GAAG,GAAG,GAAG,CAAC,WAAW,EAAE,CAAC;gBAC9B,mEAAmE;gBACnE,8DAA8D;gBAC9D,IAAI,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC,GAAG,CAAC;oBAAE,MAAM,IAAI,KAAK,CAAC,yCAAyC,GAAG,GAAG,CAAC,CAAC;gBAChG,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC,GAAG,EAAE,MAAM,CAAC,CAAC;YACpC,CAAC;QACH,CAAC;IACH,CAAC;IAED,6FAA6F;IAC7F,SAAS,CAAC,IAAY;QACpB,OAAO,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC,aAAa,CAAC,IAAI,CAAC,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC;IACpE,CAAC;CACF"}
@@ -0,0 +1,13 @@
1
+ import type { DocExtractService } from './doc-extract.service.js';
2
+ import { FileReaderRegistry } from './file-reader.js';
3
+ /**
4
+ * THE registry of file readers — the one place that says which reader owns
5
+ * which extension. Adding a format is one entry here (plus its reader/extract
6
+ * function); read_file, grep and the write-refusal all follow automatically.
7
+ *
8
+ * `docExtract` is the shared content-hash extraction cache the document
9
+ * readers wrap their pure extract functions with (see DocumentReader).
10
+ * Everything not claimed below falls back to the TextReader.
11
+ */
12
+ export declare function createFileReaderRegistry(docExtract: DocExtractService): FileReaderRegistry;
13
+ //# sourceMappingURL=file-reader.registry.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"file-reader.registry.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/file-reader.registry.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,0BAA0B,CAAC;AAYlE,OAAO,EAAE,kBAAkB,EAAE,MAAM,kBAAkB,CAAC;AAItD;;;;;;;;GAQG;AACH,wBAAgB,wBAAwB,CAAC,UAAU,EAAE,iBAAiB,GAAG,kBAAkB,CAmB1F"}
@@ -0,0 +1,41 @@
1
+ import { DocumentReader } from './document-reader.js';
2
+ import { EmailReader } from './email-reader.js';
3
+ import { extractDocx } from './extract-docx.js';
4
+ import { extractEml } from './extract-eml.js';
5
+ import { extractMsg } from './extract-msg.js';
6
+ import { extractOdp } from './extract-odp.js';
7
+ import { extractOds } from './extract-ods.js';
8
+ import { extractOdt } from './extract-odt.js';
9
+ import { extractPdf } from './extract-pdf.js';
10
+ import { extractPptx } from './extract-pptx.js';
11
+ import { extractXlsx } from './extract-xlsx.js';
12
+ import { FileReaderRegistry } from './file-reader.js';
13
+ import { ImageReader } from './image-reader.js';
14
+ import { LegacyOfficeReader, TextReader } from './text-reader.js';
15
+ /**
16
+ * THE registry of file readers — the one place that says which reader owns
17
+ * which extension. Adding a format is one entry here (plus its reader/extract
18
+ * function); read_file, grep and the write-refusal all follow automatically.
19
+ *
20
+ * `docExtract` is the shared content-hash extraction cache the document
21
+ * readers wrap their pure extract functions with (see DocumentReader).
22
+ * Everything not claimed below falls back to the TextReader.
23
+ */
24
+ export function createFileReaderRegistry(docExtract) {
25
+ return new FileReaderRegistry([
26
+ new DocumentReader('.docx', extractDocx, docExtract),
27
+ new DocumentReader('.pptx', extractPptx, docExtract),
28
+ new DocumentReader('.xlsx', extractXlsx, docExtract),
29
+ new DocumentReader('.pdf', extractPdf, docExtract),
30
+ new DocumentReader('.odt', extractOdt, docExtract),
31
+ new DocumentReader('.odp', extractOdp, docExtract),
32
+ new DocumentReader('.ods', extractOds, docExtract),
33
+ // Email files ride the same document machinery (cached extraction,
34
+ // greppable, not text-editable) with an email-honest write refusal.
35
+ new EmailReader('.eml', extractEml, docExtract),
36
+ new EmailReader('.msg', extractMsg, docExtract),
37
+ new ImageReader(),
38
+ new LegacyOfficeReader(),
39
+ ], new TextReader());
40
+ }
41
+ //# sourceMappingURL=file-reader.registry.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"file-reader.registry.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/file-reader.registry.ts"],"names":[],"mappings":"AACA,OAAO,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAC;AACtD,OAAO,EAAE,WAAW,EAAE,MAAM,mBAAmB,CAAC;AAChD,OAAO,EAAE,WAAW,EAAE,MAAM,mBAAmB,CAAC;AAChD,OAAO,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAC9C,OAAO,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAC9C,OAAO,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAC9C,OAAO,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAC9C,OAAO,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAC9C,OAAO,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAC9C,OAAO,EAAE,WAAW,EAAE,MAAM,mBAAmB,CAAC;AAChD,OAAO,EAAE,WAAW,EAAE,MAAM,mBAAmB,CAAC;AAChD,OAAO,EAAE,kBAAkB,EAAE,MAAM,kBAAkB,CAAC;AACtD,OAAO,EAAE,WAAW,EAAE,MAAM,mBAAmB,CAAC;AAChD,OAAO,EAAE,kBAAkB,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAElE;;;;;;;;GAQG;AACH,MAAM,UAAU,wBAAwB,CAAC,UAA6B;IACpE,OAAO,IAAI,kBAAkB,CAC3B;QACE,IAAI,cAAc,CAAC,OAAO,EAAE,WAAW,EAAE,UAAU,CAAC;QACpD,IAAI,cAAc,CAAC,OAAO,EAAE,WAAW,EAAE,UAAU,CAAC;QACpD,IAAI,cAAc,CAAC,OAAO,EAAE,WAAW,EAAE,UAAU,CAAC;QACpD,IAAI,cAAc,CAAC,MAAM,EAAE,UAAU,EAAE,UAAU,CAAC;QAClD,IAAI,cAAc,CAAC,MAAM,EAAE,UAAU,EAAE,UAAU,CAAC;QAClD,IAAI,cAAc,CAAC,MAAM,EAAE,UAAU,EAAE,UAAU,CAAC;QAClD,IAAI,cAAc,CAAC,MAAM,EAAE,UAAU,EAAE,UAAU,CAAC;QAClD,mEAAmE;QACnE,oEAAoE;QACpE,IAAI,WAAW,CAAC,MAAM,EAAE,UAAU,EAAE,UAAU,CAAC;QAC/C,IAAI,WAAW,CAAC,MAAM,EAAE,UAAU,EAAE,UAAU,CAAC;QAC/C,IAAI,WAAW,EAAE;QACjB,IAAI,kBAAkB,EAAE;KACzB,EACD,IAAI,UAAU,EAAE,CACjB,CAAC;AACJ,CAAC"}
@@ -0,0 +1,35 @@
1
+ /** The extensions the `ImageReader` registers for — exactly the map above. */
2
+ export declare const IMAGE_EXTENSIONS: readonly string[];
3
+ /** Is this a file `read_file` should return as an image? */
4
+ export declare function isImageFile(path: string): boolean;
5
+ /** The MCP mime type for an image path — call only after `isImageFile`. */
6
+ export declare function imageMimeType(path: string): string;
7
+ /**
8
+ * Cap on the RAW bytes of an image `read_file` will return.
9
+ *
10
+ * The arithmetic: Claude-family clients reject a base64 image payload around
11
+ * 5 MB encoded (5 * 1024 * 1024 = 5,242,880 chars). Base64 inflates by 4/3,
12
+ * so the raw ceiling for that limit is 5,242,880 * 3/4 = 3,932,160 bytes. We
13
+ * cap at 3.5 MiB raw = 3,670,016 bytes, which encodes to
14
+ * ceil(3,670,016 / 3) * 4 = 4,893,356 base64 chars (~4.67 MiB) — under the
15
+ * reject line with headroom for the JSON envelope the sentinel travels in.
16
+ */
17
+ export declare const IMAGE_MAX_RAW_BYTES: number;
18
+ /** Width×height parsed from the container header, when the format makes that cheap. */
19
+ export interface ImageDimensions {
20
+ width: number;
21
+ height: number;
22
+ }
23
+ /**
24
+ * Best-effort dimensions straight from the header bytes: PNG (IHDR), GIF
25
+ * (logical screen descriptor), JPEG (SOFn scan). WebP is left dimensionless —
26
+ * its three sub-formats (VP8/VP8L/VP8X) each encode size differently, and the
27
+ * note is honest without it. Returns undefined on anything unexpected; never
28
+ * throws (a corrupt image must still be delivered or refused, not 500).
29
+ */
30
+ export declare function imageDimensions(bytes: Buffer): ImageDimensions | undefined;
31
+ /** The one-line note emitted as a text block beside the image, so the transcript names what it shows. */
32
+ export declare function imageNote(path: string, mime: string, sizeBytes: number, dims: ImageDimensions | undefined): string;
33
+ /** The honest refusal for an image over {@link IMAGE_MAX_RAW_BYTES} — returned as the file's text content. */
34
+ export declare function oversizedImageNotice(path: string, mime: string, sizeBytes: number): string;
35
+ //# sourceMappingURL=image-read.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"image-read.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/image-read.ts"],"names":[],"mappings":"AAwBA,8EAA8E;AAC9E,eAAO,MAAM,gBAAgB,EAAE,SAAS,MAAM,EAAmC,CAAC;AAElF,4DAA4D;AAC5D,wBAAgB,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAEjD;AAED,2EAA2E;AAC3E,wBAAgB,aAAa,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAElD;AAED;;;;;;;;;GASG;AACH,eAAO,MAAM,mBAAmB,QAAoB,CAAC;AAErD,uFAAuF;AACvF,MAAM,WAAW,eAAe;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,MAAM,EAAE,MAAM,CAAC;CAChB;AAED;;;;;;GAMG;AACH,wBAAgB,eAAe,CAAC,KAAK,EAAE,MAAM,GAAG,eAAe,GAAG,SAAS,CA2C1E;AAED,yGAAyG;AACzG,wBAAgB,SAAS,CAAC,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,IAAI,EAAE,eAAe,GAAG,SAAS,GAAG,MAAM,CAGlH;AAED,8GAA8G;AAC9G,wBAAgB,oBAAoB,CAAC,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,GAAG,MAAM,CAO1F"}
@@ -0,0 +1,108 @@
1
+ /**
2
+ * `read_file` image support: which files come back as a native MCP image
3
+ * content block instead of text, the size cap that keeps the encoded payload
4
+ * under what Claude-family clients accept, and the self-describing note that
5
+ * rides beside the image so the transcript still says what was read.
6
+ *
7
+ * Deliberately dependency-free: dimensions are parsed straight from the
8
+ * container headers (PNG/GIF/JPEG) rather than through a native image library.
9
+ * No downscaling in this increment — `sharp` is a heavy native dependency, so
10
+ * an oversized image gets an honest refusal telling the caller to downscale
11
+ * locally or upload a smaller export.
12
+ */
13
+ import { fileExtension } from './doc-extract.types.js';
14
+ /** The image types `read_file` returns as a native MCP image content block. `.svg` is TEXT and stays on the text path. */
15
+ const IMAGE_MIME_BY_EXT = {
16
+ '.png': 'image/png',
17
+ '.jpg': 'image/jpeg',
18
+ '.jpeg': 'image/jpeg',
19
+ // GIF passes through whole under the same cap; clients render the first frame.
20
+ '.gif': 'image/gif',
21
+ '.webp': 'image/webp',
22
+ };
23
+ /** The extensions the `ImageReader` registers for — exactly the map above. */
24
+ export const IMAGE_EXTENSIONS = Object.keys(IMAGE_MIME_BY_EXT);
25
+ /** Is this a file `read_file` should return as an image? */
26
+ export function isImageFile(path) {
27
+ return fileExtension(path) in IMAGE_MIME_BY_EXT;
28
+ }
29
+ /** The MCP mime type for an image path — call only after `isImageFile`. */
30
+ export function imageMimeType(path) {
31
+ return IMAGE_MIME_BY_EXT[fileExtension(path)];
32
+ }
33
+ /**
34
+ * Cap on the RAW bytes of an image `read_file` will return.
35
+ *
36
+ * The arithmetic: Claude-family clients reject a base64 image payload around
37
+ * 5 MB encoded (5 * 1024 * 1024 = 5,242,880 chars). Base64 inflates by 4/3,
38
+ * so the raw ceiling for that limit is 5,242,880 * 3/4 = 3,932,160 bytes. We
39
+ * cap at 3.5 MiB raw = 3,670,016 bytes, which encodes to
40
+ * ceil(3,670,016 / 3) * 4 = 4,893,356 base64 chars (~4.67 MiB) — under the
41
+ * reject line with headroom for the JSON envelope the sentinel travels in.
42
+ */
43
+ export const IMAGE_MAX_RAW_BYTES = 3.5 * 1024 * 1024; // = 3,670,016
44
+ /**
45
+ * Best-effort dimensions straight from the header bytes: PNG (IHDR), GIF
46
+ * (logical screen descriptor), JPEG (SOFn scan). WebP is left dimensionless —
47
+ * its three sub-formats (VP8/VP8L/VP8X) each encode size differently, and the
48
+ * note is honest without it. Returns undefined on anything unexpected; never
49
+ * throws (a corrupt image must still be delivered or refused, not 500).
50
+ */
51
+ export function imageDimensions(bytes) {
52
+ try {
53
+ // PNG: 8-byte signature, then the IHDR chunk — width/height at 16/20 (BE).
54
+ if (bytes.length >= 24 && bytes.readUInt32BE(0) === 0x89504e47 && bytes.toString('latin1', 12, 16) === 'IHDR') {
55
+ return { width: bytes.readUInt32BE(16), height: bytes.readUInt32BE(20) };
56
+ }
57
+ // GIF: the FULL "GIF87a"/"GIF89a" signature, then the logical screen
58
+ // size at 6/8 (LE) — a bare "GIF" prefix is not a GIF header.
59
+ if (bytes.length >= 10) {
60
+ const sig = bytes.toString('latin1', 0, 6);
61
+ if (sig === 'GIF87a' || sig === 'GIF89a') {
62
+ return { width: bytes.readUInt16LE(6), height: bytes.readUInt16LE(8) };
63
+ }
64
+ }
65
+ // JPEG: walk the marker segments to the first SOFn frame header.
66
+ if (bytes.length >= 4 && bytes[0] === 0xff && bytes[1] === 0xd8) {
67
+ let i = 2;
68
+ while (i + 9 < bytes.length) {
69
+ if (bytes[i] !== 0xff) {
70
+ i += 1; // stray fill byte — resync
71
+ continue;
72
+ }
73
+ const marker = bytes[i + 1];
74
+ // Standalone markers (no length field): padding, restarts, SOI/EOI.
75
+ if (marker === 0xff) {
76
+ i += 1;
77
+ continue;
78
+ }
79
+ if (marker === 0x01 || (marker >= 0xd0 && marker <= 0xd9)) {
80
+ i += 2;
81
+ continue;
82
+ }
83
+ // SOFn (C0–CF except the non-frame C4/C8/CC): height at +5, width at +7.
84
+ if (marker >= 0xc0 && marker <= 0xcf && marker !== 0xc4 && marker !== 0xc8 && marker !== 0xcc) {
85
+ return { height: bytes.readUInt16BE(i + 5), width: bytes.readUInt16BE(i + 7) };
86
+ }
87
+ i += 2 + bytes.readUInt16BE(i + 2);
88
+ }
89
+ }
90
+ }
91
+ catch {
92
+ // Truncated/corrupt header — the note simply omits dimensions.
93
+ }
94
+ return undefined;
95
+ }
96
+ /** The one-line note emitted as a text block beside the image, so the transcript names what it shows. */
97
+ export function imageNote(path, mime, sizeBytes, dims) {
98
+ const dimsPart = dims ? `, ${dims.width}×${dims.height} px` : '';
99
+ return `[image: ${path} — ${mime}, ${sizeBytes} bytes${dimsPart}]`;
100
+ }
101
+ /** The honest refusal for an image over {@link IMAGE_MAX_RAW_BYTES} — returned as the file's text content. */
102
+ export function oversizedImageNotice(path, mime, sizeBytes) {
103
+ return (`[${path} is a ${mime} image of ${sizeBytes} bytes — too large to return over MCP. The cap is ` +
104
+ `${IMAGE_MAX_RAW_BYTES} bytes (3.5 MiB) of raw image data, because base64 encoding inflates it by 4/3 and ` +
105
+ 'Claude-family clients reject images near 5 MB encoded. Downscale the image locally or upload a smaller ' +
106
+ 'export, then read that file instead.]');
107
+ }
108
+ //# sourceMappingURL=image-read.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"image-read.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/image-read.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;GAWG;AACH,OAAO,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAEvD,0HAA0H;AAC1H,MAAM,iBAAiB,GAA2B;IAChD,MAAM,EAAE,WAAW;IACnB,MAAM,EAAE,YAAY;IACpB,OAAO,EAAE,YAAY;IACrB,+EAA+E;IAC/E,MAAM,EAAE,WAAW;IACnB,OAAO,EAAE,YAAY;CACtB,CAAC;AAEF,8EAA8E;AAC9E,MAAM,CAAC,MAAM,gBAAgB,GAAsB,MAAM,CAAC,IAAI,CAAC,iBAAiB,CAAC,CAAC;AAElF,4DAA4D;AAC5D,MAAM,UAAU,WAAW,CAAC,IAAY;IACtC,OAAO,aAAa,CAAC,IAAI,CAAC,IAAI,iBAAiB,CAAC;AAClD,CAAC;AAED,2EAA2E;AAC3E,MAAM,UAAU,aAAa,CAAC,IAAY;IACxC,OAAO,iBAAiB,CAAC,aAAa,CAAC,IAAI,CAAC,CAAE,CAAC;AACjD,CAAC;AAED;;;;;;;;;GASG;AACH,MAAM,CAAC,MAAM,mBAAmB,GAAG,GAAG,GAAG,IAAI,GAAG,IAAI,CAAC,CAAC,cAAc;AAQpE;;;;;;GAMG;AACH,MAAM,UAAU,eAAe,CAAC,KAAa;IAC3C,IAAI,CAAC;QACH,2EAA2E;QAC3E,IAAI,KAAK,CAAC,MAAM,IAAI,EAAE,IAAI,KAAK,CAAC,YAAY,CAAC,CAAC,CAAC,KAAK,UAAU,IAAI,KAAK,CAAC,QAAQ,CAAC,QAAQ,EAAE,EAAE,EAAE,EAAE,CAAC,KAAK,MAAM,EAAE,CAAC;YAC9G,OAAO,EAAE,KAAK,EAAE,KAAK,CAAC,YAAY,CAAC,EAAE,CAAC,EAAE,MAAM,EAAE,KAAK,CAAC,YAAY,CAAC,EAAE,CAAC,EAAE,CAAC;QAC3E,CAAC;QACD,qEAAqE;QACrE,8DAA8D;QAC9D,IAAI,KAAK,CAAC,MAAM,IAAI,EAAE,EAAE,CAAC;YACvB,MAAM,GAAG,GAAG,KAAK,CAAC,QAAQ,CAAC,QAAQ,EAAE,CAAC,EAAE,CAAC,CAAC,CAAC;YAC3C,IAAI,GAAG,KAAK,QAAQ,IAAI,GAAG,KAAK,QAAQ,EAAE,CAAC;gBACzC,OAAO,EAAE,KAAK,EAAE,KAAK,CAAC,YAAY,CAAC,CAAC,CAAC,EAAE,MAAM,EAAE,KAAK,CAAC,YAAY,CAAC,CAAC,CAAC,EAAE,CAAC;YACzE,CAAC;QACH,CAAC;QACD,iEAAiE;QACjE,IAAI,KAAK,CAAC,MAAM,IAAI,CAAC,IAAI,KAAK,CAAC,CAAC,CAAC,KAAK,IAAI,IAAI,KAAK,CAAC,CAAC,CAAC,KAAK,IAAI,EAAE,CAAC;YAChE,IAAI,CAAC,GAAG,CAAC,CAAC;YACV,OAAO,CAAC,GAAG,CAAC,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC;gBAC5B,IAAI,KAAK,CAAC,CAAC,CAAC,KAAK,IAAI,EAAE,CAAC;oBACtB,CAAC,IAAI,CAAC,CAAC,CAAC,2BAA2B;oBACnC,SAAS;gBACX,CAAC;gBACD,MAAM,MAAM,GAAG,KAAK,CAAC,CAAC,GAAG,CAAC,CAAE,CAAC;gBAC7B,oEAAoE;gBACpE,IAAI,MAAM,KAAK,IAAI,EAAE,CAAC;oBACpB,CAAC,IAAI,CAAC,CAAC;oBACP,SAAS;gBACX,CAAC;gBACD,IAAI,MAAM,KAAK,IAAI,IAAI,CAAC,MAAM,IAAI,IAAI,IAAI,MAAM,IAAI,IAAI,CAAC,EAAE,CAAC;oBAC1D,CAAC,IAAI,CAAC,CAAC;oBACP,SAAS;gBACX,CAAC;gBACD,yEAAyE;gBACzE,IAAI,MAAM,IAAI,IAAI,IAAI,MAAM,IAAI,IAAI,IAAI,MAAM,KAAK,IAAI,IAAI,MAAM,KAAK,IAAI,IAAI,MAAM,KAAK,IAAI,EAAE,CAAC;oBAC9F,OAAO,EAAE,MAAM,EAAE,KAAK,CAAC,YAAY,CAAC,CAAC,GAAG,CAAC,CAAC,EAAE,KAAK,EAAE,KAAK,CAAC,YAAY,CAAC,CAAC,GAAG,CAAC,CAAC,EAAE,CAAC;gBACjF,CAAC;gBACD,CAAC,IAAI,CAAC,GAAG,KAAK,CAAC,YAAY,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC;YACrC,CAAC;QACH,CAAC;IACH,CAAC;IAAC,MAAM,CAAC;QACP,+DAA+D;IACjE,CAAC;IACD,OAAO,SAAS,CAAC;AACnB,CAAC;AAED,yGAAyG;AACzG,MAAM,UAAU,SAAS,CAAC,IAAY,EAAE,IAAY,EAAE,SAAiB,EAAE,IAAiC;IACxG,MAAM,QAAQ,GAAG,IAAI,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,KAAK,IAAI,IAAI,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,EAAE,CAAC;IACjE,OAAO,WAAW,IAAI,MAAM,IAAI,KAAK,SAAS,SAAS,QAAQ,GAAG,CAAC;AACrE,CAAC;AAED,8GAA8G;AAC9G,MAAM,UAAU,oBAAoB,CAAC,IAAY,EAAE,IAAY,EAAE,SAAiB;IAChF,OAAO,CACL,IAAI,IAAI,SAAS,IAAI,aAAa,SAAS,oDAAoD;QAC/F,GAAG,mBAAmB,qFAAqF;QAC3G,yGAAyG;QACzG,uCAAuC,CACxC,CAAC;AACJ,CAAC"}
@@ -0,0 +1,19 @@
1
+ import type { FileReader, ReadResult } from './file-reader.js';
2
+ /**
3
+ * FileReader over the image types `read_file` returns as a native MCP image
4
+ * content block (`.svg` is text and stays on the text path). Within the raw
5
+ * cap the read yields the picture itself (base64 + mime + the self-describing
6
+ * note); over it, the honest downscale refusal — see image-read.ts for the
7
+ * cap arithmetic and header-parsing details.
8
+ *
9
+ * No `greppableText`: a picture is never text-searchable. And images stay
10
+ * `textEditable` — the write tools only refuse formats whose reads are lossy
11
+ * EXTRACTIONS (documents); an image read is the real bytes, and image writes
12
+ * were never gated.
13
+ */
14
+ export declare class ImageReader implements FileReader {
15
+ readonly extensions: readonly string[];
16
+ readonly textEditable = true;
17
+ read(bytes: Buffer, path: string): Promise<ReadResult>;
18
+ }
19
+ //# sourceMappingURL=image-reader.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"image-reader.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/image-reader.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,UAAU,EAAE,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAU/D;;;;;;;;;;;GAWG;AACH,qBAAa,WAAY,YAAW,UAAU;IAC5C,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAoB;IAC1D,QAAQ,CAAC,YAAY,QAAQ;IAEvB,IAAI,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,UAAU,CAAC;CAY7D"}
@@ -0,0 +1,30 @@
1
+ import { IMAGE_EXTENSIONS, IMAGE_MAX_RAW_BYTES, imageDimensions, imageMimeType, imageNote, oversizedImageNotice, } from './image-read.js';
2
+ /**
3
+ * FileReader over the image types `read_file` returns as a native MCP image
4
+ * content block (`.svg` is text and stays on the text path). Within the raw
5
+ * cap the read yields the picture itself (base64 + mime + the self-describing
6
+ * note); over it, the honest downscale refusal — see image-read.ts for the
7
+ * cap arithmetic and header-parsing details.
8
+ *
9
+ * No `greppableText`: a picture is never text-searchable. And images stay
10
+ * `textEditable` — the write tools only refuse formats whose reads are lossy
11
+ * EXTRACTIONS (documents); an image read is the real bytes, and image writes
12
+ * were never gated.
13
+ */
14
+ export class ImageReader {
15
+ extensions = IMAGE_EXTENSIONS;
16
+ textEditable = true;
17
+ async read(bytes, path) {
18
+ const mime = imageMimeType(path);
19
+ if (bytes.length > IMAGE_MAX_RAW_BYTES) {
20
+ return { kind: 'refusal', message: oversizedImageNotice(path, mime, bytes.length) };
21
+ }
22
+ return {
23
+ kind: 'image',
24
+ data: bytes.toString('base64'),
25
+ mimeType: mime,
26
+ note: imageNote(path, mime, bytes.length, imageDimensions(bytes)),
27
+ };
28
+ }
29
+ }
30
+ //# sourceMappingURL=image-reader.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"image-reader.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/image-reader.ts"],"names":[],"mappings":"AACA,OAAO,EACL,gBAAgB,EAChB,mBAAmB,EACnB,eAAe,EACf,aAAa,EACb,SAAS,EACT,oBAAoB,GACrB,MAAM,iBAAiB,CAAC;AAEzB;;;;;;;;;;;GAWG;AACH,MAAM,OAAO,WAAW;IACb,UAAU,GAAsB,gBAAgB,CAAC;IACjD,YAAY,GAAG,IAAI,CAAC;IAE7B,KAAK,CAAC,IAAI,CAAC,KAAa,EAAE,IAAY;QACpC,MAAM,IAAI,GAAG,aAAa,CAAC,IAAI,CAAC,CAAC;QACjC,IAAI,KAAK,CAAC,MAAM,GAAG,mBAAmB,EAAE,CAAC;YACvC,OAAO,EAAE,IAAI,EAAE,SAAS,EAAE,OAAO,EAAE,oBAAoB,CAAC,IAAI,EAAE,IAAI,EAAE,KAAK,CAAC,MAAM,CAAC,EAAE,CAAC;QACtF,CAAC;QACD,OAAO;YACL,IAAI,EAAE,OAAO;YACb,IAAI,EAAE,KAAK,CAAC,QAAQ,CAAC,QAAQ,CAAC;YAC9B,QAAQ,EAAE,IAAI;YACd,IAAI,EAAE,SAAS,CAAC,IAAI,EAAE,IAAI,EAAE,KAAK,CAAC,MAAM,EAAE,eAAe,CAAC,KAAK,CAAC,CAAC;SAClE,CAAC;IACJ,CAAC;CACF"}