@bevel-software/platform-core-backend 0.11.2 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/THIRD-PARTY-NOTICES.md +1165 -427
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +2 -0
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +5 -0
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  13. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  14. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  15. package/dist/modules/kb-fs/clone-config.d.ts +40 -2
  16. package/dist/modules/kb-fs/clone-config.d.ts.map +1 -1
  17. package/dist/modules/kb-fs/clone-config.js +94 -2
  18. package/dist/modules/kb-fs/clone-config.js.map +1 -1
  19. package/dist/modules/workflow/workflow.service.d.ts +38 -0
  20. package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
  21. package/dist/modules/workflow/workflow.service.js +112 -6
  22. package/dist/modules/workflow/workflow.service.js.map +1 -1
  23. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  24. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  25. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  26. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  27. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  28. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  29. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  30. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  31. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  32. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  33. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  34. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  35. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  36. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  37. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  38. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  39. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  40. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  41. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  42. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  43. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  44. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  45. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  46. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  47. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  48. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  49. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  50. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  51. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  52. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  53. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  54. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  55. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  56. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  57. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  58. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  59. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  60. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  61. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  62. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  63. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  64. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  65. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  66. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  67. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  68. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  69. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  70. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  71. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  72. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  73. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  74. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  75. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  76. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  77. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  78. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  79. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  80. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  81. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  82. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  83. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  84. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  85. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  86. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  87. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  88. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  89. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  90. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  91. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  92. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  93. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  94. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  95. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  96. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  97. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  98. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  99. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  100. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  101. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  102. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  103. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  104. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  105. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  106. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  107. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  108. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  109. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  110. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  111. package/dist/modules/workspace/startup/kb-git.d.ts.map +1 -1
  112. package/dist/modules/workspace/startup/kb-git.js +21 -4
  113. package/dist/modules/workspace/startup/kb-git.js.map +1 -1
  114. package/dist/modules/workspace/workspace.service.d.ts +52 -8
  115. package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
  116. package/dist/modules/workspace/workspace.service.js +121 -23
  117. package/dist/modules/workspace/workspace.service.js.map +1 -1
  118. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  119. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  120. package/dist/modules/workspace/workspace.tools.js +158 -15
  121. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  122. package/package.json +11 -6
  123. package/src/core/create-core-server.ts +1 -1
  124. package/src/core/create-core-services.ts +6 -0
  125. package/src/core-config.ts +9 -0
  126. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  127. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  128. package/src/modules/kb-fs/__tests__/clone-config.test.ts +63 -2
  129. package/src/modules/kb-fs/clone-config.ts +97 -2
  130. package/src/modules/secrets-vault/secrets-vault.routes.ts +582 -582
  131. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  132. package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +11 -5
  133. package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +172 -7
  134. package/src/modules/workflow/workflow.service.ts +118 -6
  135. package/src/modules/workspace/__tests__/workspace.service.test.ts +1 -1
  136. package/src/modules/workspace/__tests__/workspace.tools.test.ts +500 -2
  137. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  138. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  139. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  140. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  141. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  142. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  143. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  144. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  145. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  146. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  147. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  148. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  149. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  150. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  151. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  152. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  153. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  154. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  155. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  156. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  157. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  158. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  159. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  160. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  161. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  162. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  163. package/src/modules/workspace/startup/__tests__/kb-startup-runner.test.ts +141 -0
  164. package/src/modules/workspace/startup/kb-git.ts +20 -7
  165. package/src/modules/workspace/workspace.service.ts +132 -25
  166. package/src/modules/workspace/workspace.tools.ts +174 -12
@@ -0,0 +1,90 @@
1
+ import { extractionMarker, fileExtension } from './doc-extract.types.js';
2
+ import { DocExtractionCache, gitBlobSha } from './extraction-cache.js';
3
+ /**
4
+ * The content-hash CACHE wrap around document text extraction. The per-format
5
+ * `DocumentReader`s in the file-reader registry stay pure (each pairs one
6
+ * extension with one extract function); this service adds the caching
7
+ * generically — a reader hands its extract function in, and the service only
8
+ * runs it on a cache miss.
9
+ *
10
+ * Caching: keyed by the git BLOB sha of the bytes (see `gitBlobSha`) PLUS the
11
+ * path's extension, stored as `{ summary, text }` WITHOUT the path — the
12
+ * marker is assembled per call so the same content read under two paths
13
+ * (copies, branches) shares one entry yet each read's marker names the path
14
+ * that was read. The extension is part of the key because identical bytes
15
+ * extract DIFFERENTLY per format: a `.odt` renamed `.ods` must run the ods
16
+ * extractor, not return the odt extraction. Extraction is deterministic
17
+ * (stable slide/sheet/page ordering), which is what makes a content-keyed
18
+ * cache correct. Failures are NOT cached (rare, and fail fast). The key also carries
19
+ * `EXTRACTION_SCHEMA`, so an upgrade that changes what an extractor emits does
20
+ * not keep serving the previous release's text for unchanged bytes.
21
+ */
22
+ /**
23
+ * Bumped whenever extraction OUTPUT changes — a new extractor, a fixed one, a
24
+ * reworded summary, a different marker.
25
+ *
26
+ * The rest of the key is content, and content-addressing is only correct while
27
+ * the same bytes mean the same text. An upgrade that changes what an extractor
28
+ * emits breaks that: every already-cached document would keep serving the OLD
29
+ * extraction forever, because its bytes never changed. Bumping this retires
30
+ * those entries (the cache evicts by age, so they cost nothing for long).
31
+ *
32
+ * v2: extraction moved from hand-rolled scanning to a real XML/HTML parser,
33
+ * which changed self-closing paragraphs, entity edge cases and recovery from
34
+ * malformed parts.
35
+ */
36
+ export const EXTRACTION_SCHEMA = 'v2';
37
+ export class DocExtractService {
38
+ cache;
39
+ /**
40
+ * Cold extractions currently running, by cache key. Two reads of the same
41
+ * document arriving together would otherwise BOTH parse it — the expensive
42
+ * half of this service, run twice for one answer — because neither had
43
+ * written the cache entry yet when the other looked.
44
+ */
45
+ inFlight = new Map();
46
+ constructor(cacheRoot) {
47
+ this.cache = new DocExtractionCache(cacheRoot);
48
+ }
49
+ /**
50
+ * The CACHED extraction for these bytes, or undefined on a cache miss —
51
+ * never extracts. `grep` uses this (via `DocumentReader.greppableText`) to
52
+ * search already-extracted documents for free and apply its per-walk budget
53
+ * only to cold ones.
54
+ */
55
+ async getCached(path, bytes) {
56
+ const hit = await this.cache.get(this.cacheKey(path, bytes));
57
+ return hit && { marker: extractionMarker(path, hit.summary), text: hit.text };
58
+ }
59
+ /** Extract via `extractFn` (cache hit or parse + store). `path` supplies the marker AND the key's format part. */
60
+ async extract(path, bytes, extractFn) {
61
+ const key = this.cacheKey(path, bytes);
62
+ const hit = await this.cache.get(key);
63
+ if (hit)
64
+ return { ok: true, marker: extractionMarker(path, hit.summary), text: hit.text };
65
+ const running = this.inFlight.get(key);
66
+ const result = await (running ??
67
+ (() => {
68
+ // The cache write stays INSIDE the shared promise: were the entry
69
+ // dropped as soon as parsing settled, a read arriving during the
70
+ // write would miss both the cache and the in-flight map — and parse
71
+ // the same document again.
72
+ const started = (async () => {
73
+ const res = await extractFn(bytes);
74
+ if (res.ok)
75
+ await this.cache.put(key, { summary: res.summary, text: res.text });
76
+ return res;
77
+ })().finally(() => this.inFlight.delete(key));
78
+ this.inFlight.set(key, started);
79
+ return started;
80
+ })());
81
+ if (!result.ok)
82
+ return result;
83
+ return { ok: true, marker: extractionMarker(path, result.summary), text: result.text };
84
+ }
85
+ /** Content hash + lowercased extension — e.g. `…sha….odt` (see the class doc). */
86
+ cacheKey(path, bytes) {
87
+ return `${gitBlobSha(bytes)}${fileExtension(path)}.${EXTRACTION_SCHEMA}`;
88
+ }
89
+ }
90
+ //# sourceMappingURL=doc-extract.service.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"doc-extract.service.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/doc-extract.service.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,gBAAgB,EAAE,aAAa,EAAsC,MAAM,wBAAwB,CAAC;AAC7G,OAAO,EAAE,kBAAkB,EAAE,UAAU,EAAE,MAAM,uBAAuB,CAAC;AAcvE;;;;;;;;;;;;;;;;;;GAkBG;AACH;;;;;;;;;;;;;GAaG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,IAAI,CAAC;AAEtC,MAAM,OAAO,iBAAiB;IACX,KAAK,CAAqB;IAC3C;;;;;OAKG;IACc,QAAQ,GAAG,IAAI,GAAG,EAAkC,CAAC;IAEtE,YAAY,SAAiB;QAC3B,IAAI,CAAC,KAAK,GAAG,IAAI,kBAAkB,CAAC,SAAS,CAAC,CAAC;IACjD,CAAC;IAED;;;;;OAKG;IACH,KAAK,CAAC,SAAS,CAAC,IAAY,EAAE,KAAa;QACzC,MAAM,GAAG,GAAG,MAAM,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,IAAI,CAAC,QAAQ,CAAC,IAAI,EAAE,KAAK,CAAC,CAAC,CAAC;QAC7D,OAAO,GAAG,IAAI,EAAE,MAAM,EAAE,gBAAgB,CAAC,IAAI,EAAE,GAAG,CAAC,OAAO,CAAC,EAAE,IAAI,EAAE,GAAG,CAAC,IAAI,EAAE,CAAC;IAChF,CAAC;IAED,kHAAkH;IAClH,KAAK,CAAC,OAAO,CAAC,IAAY,EAAE,KAAa,EAAE,SAAoB;QAC7D,MAAM,GAAG,GAAG,IAAI,CAAC,QAAQ,CAAC,IAAI,EAAE,KAAK,CAAC,CAAC;QACvC,MAAM,GAAG,GAAG,MAAM,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;QACtC,IAAI,GAAG;YAAE,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,MAAM,EAAE,gBAAgB,CAAC,IAAI,EAAE,GAAG,CAAC,OAAO,CAAC,EAAE,IAAI,EAAE,GAAG,CAAC,IAAI,EAAE,CAAC;QAC1F,MAAM,OAAO,GAAG,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC;QACvC,MAAM,MAAM,GAAG,MAAM,CAAC,OAAO;YAC3B,CAAC,GAAG,EAAE;gBACJ,kEAAkE;gBAClE,iEAAiE;gBACjE,oEAAoE;gBACpE,2BAA2B;gBAC3B,MAAM,OAAO,GAAG,CAAC,KAAK,IAA4B,EAAE;oBAClD,MAAM,GAAG,GAAG,MAAM,SAAS,CAAC,KAAK,CAAC,CAAC;oBACnC,IAAI,GAAG,CAAC,EAAE;wBAAE,MAAM,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,GAAG,EAAE,EAAE,OAAO,EAAE,GAAG,CAAC,OAAO,EAAE,IAAI,EAAE,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC;oBAChF,OAAO,GAAG,CAAC;gBACb,CAAC,CAAC,EAAE,CAAC,OAAO,CAAC,GAAG,EAAE,CAAC,IAAI,CAAC,QAAQ,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,CAAC;gBAC9C,IAAI,CAAC,QAAQ,CAAC,GAAG,CAAC,GAAG,EAAE,OAAO,CAAC,CAAC;gBAChC,OAAO,OAAO,CAAC;YACjB,CAAC,CAAC,EAAE,CAAC,CAAC;QACR,IAAI,CAAC,MAAM,CAAC,EAAE;YAAE,OAAO,MAAM,CAAC;QAC9B,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,MAAM,EAAE,gBAAgB,CAAC,IAAI,EAAE,MAAM,CAAC,OAAO,CAAC,EAAE,IAAI,EAAE,MAAM,CAAC,IAAI,EAAE,CAAC;IACzF,CAAC;IAED,kFAAkF;IAC1E,QAAQ,CAAC,IAAY,EAAE,KAAa;QAC1C,OAAO,GAAG,UAAU,CAAC,KAAK,CAAC,GAAG,aAAa,CAAC,IAAI,CAAC,IAAI,iBAAiB,EAAE,CAAC;IAC3E,CAAC;CACF"}
@@ -0,0 +1,55 @@
1
+ /**
2
+ * Server-side document text extraction — shared types.
3
+ *
4
+ * The extractors turn an office document (`.docx` / `.pptx` / `.xlsx`), an
5
+ * OpenDocument file (`.odt` / `.odp` / `.ods`) or a PDF into plain text an
6
+ * agent can read and grep. They are HONEST about being
7
+ * lossy: every successful extraction carries a one-line `summary` the consumer
8
+ * turns into a marker header (`[extracted text of <path> — <summary>]`) so the
9
+ * reader knows it is looking at extracted text, not the file's bytes.
10
+ *
11
+ * The `summary` (not a full marker) is what extractors return and what the
12
+ * cache stores, because the cache is keyed by CONTENT (git blob sha): the same
13
+ * document at two workspace paths shares one cache entry, and baking a path
14
+ * into the cached value would surface the wrong path on the second read.
15
+ */
16
+ /** A successful extraction: the marker-summary line + the extracted text. */
17
+ export interface ExtractedDoc {
18
+ /**
19
+ * One-line description for the marker header, e.g.
20
+ * `14 slides + notes; layout, images and formatting omitted`.
21
+ */
22
+ summary: string;
23
+ /** The extracted text: paragraphs/rows as lines, with `[slide N]` / `[sheet: Name]` / `[page N]` structure markers. */
24
+ text: string;
25
+ }
26
+ /**
27
+ * Extraction outcome. `ok: false` is a TYPED failure (corrupt/unparseable
28
+ * file) the consumers turn into an honest message — extraction never throws
29
+ * for bad file content, so a broken upload cannot 500 a read.
30
+ */
31
+ export type ExtractResult = ({
32
+ ok: true;
33
+ } & ExtractedDoc) | {
34
+ ok: false;
35
+ message: string;
36
+ };
37
+ /**
38
+ * A format's PURE extract function — bytes in, `ExtractResult` out (async for
39
+ * pdf.js). One per supported format (extract-docx.ts and friends); each is
40
+ * paired with its extension by a `DocumentReader` entry in the file-reader
41
+ * registry, and cache-wrapped by `DocExtractService`.
42
+ */
43
+ export type ExtractFn = (bytes: Buffer) => ExtractResult | Promise<ExtractResult>;
44
+ /**
45
+ * Build the honest ONE-LINE header consumers prepend to extracted text.
46
+ *
47
+ * One line is a promise the rest of the read path keeps: grep counts on the
48
+ * marker occupying exactly one, so read_file and grep agree on line numbers.
49
+ * A path may legally carry a CR or LF, which would forge extra lines and shift
50
+ * every number after it, so those are shown as escapes rather than obeyed.
51
+ */
52
+ export declare function extractionMarker(path: string, summary: string): string;
53
+ /** Lowercased extension of `path` including the dot, or '' when there is none. */
54
+ export declare function fileExtension(path: string): string;
55
+ //# sourceMappingURL=doc-extract.types.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"doc-extract.types.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/doc-extract.types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AAEH,6EAA6E;AAC7E,MAAM,WAAW,YAAY;IAC3B;;;OAGG;IACH,OAAO,EAAE,MAAM,CAAC;IAChB,uHAAuH;IACvH,IAAI,EAAE,MAAM,CAAC;CACd;AAED;;;;GAIG;AACH,MAAM,MAAM,aAAa,GACrB,CAAC;IAAE,EAAE,EAAE,IAAI,CAAA;CAAE,GAAG,YAAY,CAAC,GAC7B;IAAE,EAAE,EAAE,KAAK,CAAC;IAAC,OAAO,EAAE,MAAM,CAAA;CAAE,CAAC;AAEnC;;;;;GAKG;AACH,MAAM,MAAM,SAAS,GAAG,CAAC,KAAK,EAAE,MAAM,KAAK,aAAa,GAAG,OAAO,CAAC,aAAa,CAAC,CAAC;AAElF;;;;;;;GAOG;AACH,wBAAgB,gBAAgB,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,EAAE,MAAM,GAAG,MAAM,CAGtE;AAED,kFAAkF;AAClF,wBAAgB,aAAa,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAIlD"}
@@ -0,0 +1,34 @@
1
+ /**
2
+ * Server-side document text extraction — shared types.
3
+ *
4
+ * The extractors turn an office document (`.docx` / `.pptx` / `.xlsx`), an
5
+ * OpenDocument file (`.odt` / `.odp` / `.ods`) or a PDF into plain text an
6
+ * agent can read and grep. They are HONEST about being
7
+ * lossy: every successful extraction carries a one-line `summary` the consumer
8
+ * turns into a marker header (`[extracted text of <path> — <summary>]`) so the
9
+ * reader knows it is looking at extracted text, not the file's bytes.
10
+ *
11
+ * The `summary` (not a full marker) is what extractors return and what the
12
+ * cache stores, because the cache is keyed by CONTENT (git blob sha): the same
13
+ * document at two workspace paths shares one cache entry, and baking a path
14
+ * into the cached value would surface the wrong path on the second read.
15
+ */
16
+ /**
17
+ * Build the honest ONE-LINE header consumers prepend to extracted text.
18
+ *
19
+ * One line is a promise the rest of the read path keeps: grep counts on the
20
+ * marker occupying exactly one, so read_file and grep agree on line numbers.
21
+ * A path may legally carry a CR or LF, which would forge extra lines and shift
22
+ * every number after it, so those are shown as escapes rather than obeyed.
23
+ */
24
+ export function extractionMarker(path, summary) {
25
+ const oneLine = path.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
26
+ return `[extracted text of ${oneLine} — ${summary}]`;
27
+ }
28
+ /** Lowercased extension of `path` including the dot, or '' when there is none. */
29
+ export function fileExtension(path) {
30
+ const name = path.slice(path.lastIndexOf('/') + 1);
31
+ const dot = name.lastIndexOf('.');
32
+ return dot > 0 ? name.slice(dot).toLowerCase() : '';
33
+ }
34
+ //# sourceMappingURL=doc-extract.types.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"doc-extract.types.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/doc-extract.types.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;GAcG;AA8BH;;;;;;;GAOG;AACH,MAAM,UAAU,gBAAgB,CAAC,IAAY,EAAE,OAAe;IAC5D,MAAM,OAAO,GAAG,IAAI,CAAC,OAAO,CAAC,SAAS,EAAE,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC;IAC7E,OAAO,sBAAsB,OAAO,MAAM,OAAO,GAAG,CAAC;AACvD,CAAC;AAED,kFAAkF;AAClF,MAAM,UAAU,aAAa,CAAC,IAAY;IACxC,MAAM,IAAI,GAAG,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,CAAC;IACnD,MAAM,GAAG,GAAG,IAAI,CAAC,WAAW,CAAC,GAAG,CAAC,CAAC;IAClC,OAAO,GAAG,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,WAAW,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;AACtD,CAAC"}
@@ -0,0 +1,32 @@
1
+ import type { DocExtractService } from './doc-extract.service.js';
2
+ import type { ExtractFn } from './doc-extract.types.js';
3
+ import { type FileReader, type ReadResult } from './file-reader.js';
4
+ /**
5
+ * FileReader over one document format: a thin wrapper pairing the format's
6
+ * PURE extract function (extract-docx.ts and friends) with the shared
7
+ * content-hash extraction cache (`DocExtractService`). Reads return the
8
+ * extraction under its honest `[extracted text of …]` marker; a parse failure
9
+ * becomes a refusal message, never a 500.
10
+ *
11
+ * `grep` semantics: `greppableText` serves only the CACHED extraction (a hit
12
+ * costs one small JSON read), returning null for a cold document — whether to
13
+ * spend grep's per-walk extraction budget on a cold one is the walk's call,
14
+ * which then extracts through `read` (see grepWalk in workspace.tools.ts).
15
+ * grep also branches on `instanceof DocumentReader` for exactly that decision.
16
+ */
17
+ export declare class DocumentReader implements FileReader {
18
+ private readonly extract;
19
+ private readonly service;
20
+ readonly extensions: readonly string[];
21
+ /**
22
+ * `read_file` returns an EXTRACTION for these types, so text written back
23
+ * could not round-trip — the write tools refuse (documents are replaced by
24
+ * uploading a new version).
25
+ */
26
+ readonly textEditable = false;
27
+ constructor(extension: string, extract: ExtractFn, service: DocExtractService);
28
+ read(bytes: Buffer, path: string): Promise<ReadResult>;
29
+ /** The cached extraction (marker line included, so grep's line numbers match read_file's), or null when cold. */
30
+ greppableText(bytes: Buffer, path: string): Promise<string | null>;
31
+ }
32
+ //# sourceMappingURL=document-reader.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"document-reader.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/document-reader.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,iBAAiB,EAAE,MAAM,0BAA0B,CAAC;AAClE,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,wBAAwB,CAAC;AACxD,OAAO,EAAwB,KAAK,UAAU,EAAE,KAAK,UAAU,EAAE,MAAM,kBAAkB,CAAC;AAE1F;;;;;;;;;;;;GAYG;AACH,qBAAa,cAAe,YAAW,UAAU;IAW7C,OAAO,CAAC,QAAQ,CAAC,OAAO;IACxB,OAAO,CAAC,QAAQ,CAAC,OAAO;IAX1B,QAAQ,CAAC,UAAU,EAAE,SAAS,MAAM,EAAE,CAAC;IACvC;;;;OAIG;IACH,QAAQ,CAAC,YAAY,SAAS;gBAG5B,SAAS,EAAE,MAAM,EACA,OAAO,EAAE,SAAS,EAClB,OAAO,EAAE,iBAAiB;IAKvC,IAAI,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,UAAU,CAAC;IAwB5D,iHAAiH;IAC3G,aAAa,CAAC,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC,MAAM,GAAG,IAAI,CAAC;CAIzE"}
@@ -0,0 +1,59 @@
1
+ import { displayPath, oneLine } from './file-reader.js';
2
+ /**
3
+ * FileReader over one document format: a thin wrapper pairing the format's
4
+ * PURE extract function (extract-docx.ts and friends) with the shared
5
+ * content-hash extraction cache (`DocExtractService`). Reads return the
6
+ * extraction under its honest `[extracted text of …]` marker; a parse failure
7
+ * becomes a refusal message, never a 500.
8
+ *
9
+ * `grep` semantics: `greppableText` serves only the CACHED extraction (a hit
10
+ * costs one small JSON read), returning null for a cold document — whether to
11
+ * spend grep's per-walk extraction budget on a cold one is the walk's call,
12
+ * which then extracts through `read` (see grepWalk in workspace.tools.ts).
13
+ * grep also branches on `instanceof DocumentReader` for exactly that decision.
14
+ */
15
+ export class DocumentReader {
16
+ extract;
17
+ service;
18
+ extensions;
19
+ /**
20
+ * `read_file` returns an EXTRACTION for these types, so text written back
21
+ * could not round-trip — the write tools refuse (documents are replaced by
22
+ * uploading a new version).
23
+ */
24
+ textEditable = false;
25
+ constructor(extension, extract, service) {
26
+ this.extract = extract;
27
+ this.service = service;
28
+ this.extensions = [extension];
29
+ }
30
+ async read(bytes, path) {
31
+ // An extractor is contracted to ANSWER for bad content rather than throw,
32
+ // but it wraps third-party parsers, and one of those throwing past its own
33
+ // guard is a corrupt file — not a server fault. read_file says so instead
34
+ // of failing the whole tool call with a 500.
35
+ let res;
36
+ try {
37
+ res = await this.service.extract(path, bytes, this.extract);
38
+ }
39
+ catch (err) {
40
+ const reason = err instanceof Error ? err.message : String(err);
41
+ return {
42
+ kind: 'refusal',
43
+ message: `[${displayPath(path)} could not be extracted (${oneLine(reason)}) — the file may be corrupt or mislabeled. To fix it, replace the document by uploading a new version.]`,
44
+ };
45
+ }
46
+ return res.ok
47
+ ? { kind: 'text', text: `${res.marker}\n${res.text}` }
48
+ : {
49
+ kind: 'refusal',
50
+ message: `[${displayPath(path)} ${oneLine(res.message)} — the file may be corrupt or mislabeled. To fix it, replace the document by uploading a new version.]`,
51
+ };
52
+ }
53
+ /** The cached extraction (marker line included, so grep's line numbers match read_file's), or null when cold. */
54
+ async greppableText(bytes, path) {
55
+ const hit = await this.service.getCached(path, bytes);
56
+ return hit ? `${hit.marker}\n${hit.text}` : null;
57
+ }
58
+ }
59
+ //# sourceMappingURL=document-reader.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"document-reader.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/document-reader.ts"],"names":[],"mappings":"AAEA,OAAO,EAAE,WAAW,EAAE,OAAO,EAAoC,MAAM,kBAAkB,CAAC;AAE1F;;;;;;;;;;;;GAYG;AACH,MAAM,OAAO,cAAc;IAWN;IACA;IAXV,UAAU,CAAoB;IACvC;;;;OAIG;IACM,YAAY,GAAG,KAAK,CAAC;IAE9B,YACE,SAAiB,EACA,OAAkB,EAClB,OAA0B;QAD1B,YAAO,GAAP,OAAO,CAAW;QAClB,YAAO,GAAP,OAAO,CAAmB;QAE3C,IAAI,CAAC,UAAU,GAAG,CAAC,SAAS,CAAC,CAAC;IAChC,CAAC;IAED,KAAK,CAAC,IAAI,CAAC,KAAa,EAAE,IAAY;QACpC,0EAA0E;QAC1E,2EAA2E;QAC3E,0EAA0E;QAC1E,6CAA6C;QAC7C,IAAI,GAAqD,CAAC;QAC1D,IAAI,CAAC;YACH,GAAG,GAAG,MAAM,IAAI,CAAC,OAAO,CAAC,OAAO,CAAC,IAAI,EAAE,KAAK,EAAE,IAAI,CAAC,OAAO,CAAC,CAAC;QAC9D,CAAC;QAAC,OAAO,GAAG,EAAE,CAAC;YACb,MAAM,MAAM,GAAG,GAAG,YAAY,KAAK,CAAC,CAAC,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC;YAChE,OAAO;gBACL,IAAI,EAAE,SAAS;gBACf,OAAO,EAAE,IAAI,WAAW,CAAC,IAAI,CAAC,4BAA4B,OAAO,CAAC,MAAM,CAAC,yGAAyG;aACnL,CAAC;QACJ,CAAC;QAED,OAAO,GAAG,CAAC,EAAE;YACX,CAAC,CAAC,EAAE,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,GAAG,GAAG,CAAC,MAAM,KAAK,GAAG,CAAC,IAAI,EAAE,EAAE;YACtD,CAAC,CAAC;gBACE,IAAI,EAAE,SAAS;gBACf,OAAO,EAAE,IAAI,WAAW,CAAC,IAAI,CAAC,IAAI,OAAO,CAAC,GAAG,CAAC,OAAO,CAAC,wGAAwG;aAC/J,CAAC;IACR,CAAC;IAED,iHAAiH;IACjH,KAAK,CAAC,aAAa,CAAC,KAAa,EAAE,IAAY;QAC7C,MAAM,GAAG,GAAG,MAAM,IAAI,CAAC,OAAO,CAAC,SAAS,CAAC,IAAI,EAAE,KAAK,CAAC,CAAC;QACtD,OAAO,GAAG,CAAC,CAAC,CAAC,GAAG,GAAG,CAAC,MAAM,KAAK,GAAG,CAAC,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,CAAC;IACnD,CAAC;CACF"}
@@ -0,0 +1,15 @@
1
+ import { DocumentReader } from './document-reader.js';
2
+ /**
3
+ * FileReader over one email format (`.eml` / `.msg`): a `DocumentReader` in
4
+ * every mechanical respect — cache-wrapped extraction, greppable when cached,
5
+ * corrupt files answered with the typed could-not-be-parsed refusal — with
6
+ * only the WRITE-refusal copy specialized. The generic document refusal talks
7
+ * about round-tripping an office document; an email deserves the honest
8
+ * version of the same "no": the file is a snapshot of a message, and editing
9
+ * a snapshot's text is not a thing.
10
+ */
11
+ export declare class EmailReader extends DocumentReader {
12
+ /** The write-refusal for the agent text-editing tools (see `assertNotDocumentEdit`). */
13
+ editRefusal(path: string): string;
14
+ }
15
+ //# sourceMappingURL=email-reader.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"email-reader.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/email-reader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAEtD;;;;;;;;GAQG;AACH,qBAAa,WAAY,SAAQ,cAAc;IAC7C,wFAAwF;IACxF,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;CAOlC"}
@@ -0,0 +1,19 @@
1
+ import { DocumentReader } from './document-reader.js';
2
+ /**
3
+ * FileReader over one email format (`.eml` / `.msg`): a `DocumentReader` in
4
+ * every mechanical respect — cache-wrapped extraction, greppable when cached,
5
+ * corrupt files answered with the typed could-not-be-parsed refusal — with
6
+ * only the WRITE-refusal copy specialized. The generic document refusal talks
7
+ * about round-tripping an office document; an email deserves the honest
8
+ * version of the same "no": the file is a snapshot of a message, and editing
9
+ * a snapshot's text is not a thing.
10
+ */
11
+ export class EmailReader extends DocumentReader {
12
+ /** The write-refusal for the agent text-editing tools (see `assertNotDocumentEdit`). */
13
+ editRefusal(path) {
14
+ return (`"${path}" is an email file — a snapshot of a message. read_file returns EXTRACTED text for it, ` +
15
+ 'and a snapshot cannot be text-edited; to change what is stored, replace the file by uploading ' +
16
+ 'a new version.');
17
+ }
18
+ }
19
+ //# sourceMappingURL=email-reader.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"email-reader.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/email-reader.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAEtD;;;;;;;;GAQG;AACH,MAAM,OAAO,WAAY,SAAQ,cAAc;IAC7C,wFAAwF;IACxF,WAAW,CAAC,IAAY;QACtB,OAAO,CACL,IAAI,IAAI,yFAAyF;YACjG,gGAAgG;YAChG,gBAAgB,CACjB,CAAC;IACJ,CAAC;CACF"}
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Shared shaping for the email extractors (`extract-eml.ts` / `extract-msg.ts`).
3
+ *
4
+ * Both formats extract to the SAME text shape, so an agent greps a mailbox
5
+ * without caring which client saved the file:
6
+ *
7
+ * [from] Ada Lovelace <ada@example.com>
8
+ * [to] Bob <bob@example.com>, carol@example.com
9
+ * [subject] Quarterly numbers
10
+ * [date] 2026-01-05T10:00:00.000Z
11
+ *
12
+ * the body…
13
+ *
14
+ * [attachments]
15
+ * report.pdf (application/pdf, 48211 bytes)
16
+ *
17
+ * Header lines are omitted when the message lacks the field (never printed
18
+ * empty). The body prefers the plain-text part; an HTML-only body is stripped
19
+ * to text (block tags become newlines so paragraphs survive) and the marker
20
+ * summary says so. Attachments are LISTED by name only — v1 does not extract
21
+ * inside them, and the summary says that too.
22
+ */
23
+ import type { ExtractedDoc } from './doc-extract.types.js';
24
+ /** One listed attachment. Size/type are printed only when known. */
25
+ export interface EmailAttachment {
26
+ name: string;
27
+ mimeType?: string;
28
+ sizeBytes?: number;
29
+ }
30
+ /** Where the body text came from — drives the honest summary + body notes. */
31
+ export type EmailBodySource = 'text' | 'html' | 'rtf-only' | 'none';
32
+ /** The format-independent email, as far as the extraction cares. */
33
+ export interface EmailModel {
34
+ from?: string;
35
+ to?: string;
36
+ cc?: string;
37
+ bcc?: string;
38
+ subject?: string;
39
+ /** ISO timestamp when the date parsed, the raw header value otherwise. */
40
+ date?: string;
41
+ /** Body as plain text ('' when there is none or it is RTF-only). */
42
+ body: string;
43
+ bodySource: EmailBodySource;
44
+ attachments: EmailAttachment[];
45
+ }
46
+ /** The line a body that exists only as RTF gets INSTEAD of body text. */
47
+ export declare const RTF_ONLY_BODY_LINE = "[body is RTF; no plain-text part]";
48
+ export declare function htmlToEmailText(html: string): string;
49
+ /** Render the model into the marker summary + extraction text (see module doc). */
50
+ export declare function emailExtraction(model: EmailModel): ExtractedDoc;
51
+ //# sourceMappingURL=email-text.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"email-text.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/email-text.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;;GAqBG;AACH,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,wBAAwB,CAAC;AAI3D,oEAAoE;AACpE,MAAM,WAAW,eAAe;IAC9B,IAAI,EAAE,MAAM,CAAC;IACb,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,SAAS,CAAC,EAAE,MAAM,CAAC;CACpB;AAED,8EAA8E;AAC9E,MAAM,MAAM,eAAe,GAAG,MAAM,GAAG,MAAM,GAAG,UAAU,GAAG,MAAM,CAAC;AAEpE,oEAAoE;AACpE,MAAM,WAAW,UAAU;IACzB,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,EAAE,CAAC,EAAE,MAAM,CAAC;IACZ,EAAE,CAAC,EAAE,MAAM,CAAC;IACZ,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,0EAA0E;IAC1E,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,oEAAoE;IACpE,IAAI,EAAE,MAAM,CAAC;IACb,UAAU,EAAE,eAAe,CAAC;IAC5B,WAAW,EAAE,eAAe,EAAE,CAAC;CAChC;AAED,yEAAyE;AACzE,eAAO,MAAM,kBAAkB,sCAAsC,CAAC;AAsCtE,wBAAgB,eAAe,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CA8CpD;AAsBD,mFAAmF;AACnF,wBAAgB,eAAe,CAAC,KAAK,EAAE,UAAU,GAAG,YAAY,CAiC/D"}
@@ -0,0 +1,151 @@
1
+ import { Parser } from 'htmlparser2';
2
+ import { MAX_ELEMENT_DEPTH, TOO_DEEP } from './ooxml-text.js';
3
+ /** The line a body that exists only as RTF gets INSTEAD of body text. */
4
+ export const RTF_ONLY_BODY_LINE = '[body is RTF; no plain-text part]';
5
+ /**
6
+ * Strip an HTML email body to plain text. Deliberately simple (the same
7
+ * stance as `ooxml-text.ts`): comments and `<style>`/`<script>`/`<head>`/
8
+ * `<title>` containers are dropped whole, `<br>` and block-level tag
9
+ * boundaries become newlines so paragraphs survive, every other tag is
10
+ * removed, and entities are decoded through the module's shared
11
+ * `decodeXmlEntities` (plus `&nbsp;`, which HTML has and XML does not). Runs
12
+ * of blank lines collapse to one.
13
+ *
14
+ * Implemented as a SINGLE-PASS linear scanner, not regexes: the earlier
15
+ * quote-aware tag regexes re-scanned the remaining body from every `<` when a
16
+ * quoted attribute never closed — malformed input with many `<` characters
17
+ * plus one unterminated quote pinned the server quadratically. The scanner is
18
+ * quote-aware the same way (a `>` INSIDE a quoted attribute value, as in
19
+ * `<a title="a > b">`, never ends the tag early) but amortizes the failures:
20
+ * a scan that reaches end-of-input marks every `<` it passed OUTSIDE quotes
21
+ * as known-literal (their scans would be identical tails), so no position is
22
+ * rescanned from more than the three possible quote states. An unterminated
23
+ * tag is literal text, not a tag — and so is a `<…>` span that names no
24
+ * element at all, which is how `1 < 2 > 0` survives into the body.
25
+ */
26
+ /** Elements whose CONTENT is not body text and is dropped with the element. */
27
+ const CONTAINER_TAGS = new Set(['script', 'style', 'head', 'title']);
28
+ /** Elements whose boundaries end a line, so paragraphs survive as paragraphs. */
29
+ const BLOCK_TAGS = new Set([
30
+ 'p', 'div', 'section', 'article', 'header', 'footer', 'main',
31
+ 'table', 'thead', 'tbody', 'tfoot', 'tr', 'td', 'th',
32
+ 'li', 'ul', 'ol', 'dl', 'dt', 'dd', 'blockquote', 'pre', 'hr',
33
+ 'h1', 'h2', 'h3', 'h4', 'h5', 'h6',
34
+ ]);
35
+ // Depth bound shared with the OOXML/ODF extractors (`ooxml-text.ts`): mail
36
+ // nests a few dozen levels even at its most table-happy; a crafted body can
37
+ // nest as deep as it has bytes. What was read before the bound is kept.
38
+ export function htmlToEmailText(html) {
39
+ let s = '';
40
+ let depth = 0;
41
+ let skipping = 0; // inside a container whose content is not body text
42
+ const parser = new Parser({
43
+ onopentag(name) {
44
+ depth++;
45
+ if (CONTAINER_TAGS.has(name))
46
+ skipping++;
47
+ else if (name === 'br' || BLOCK_TAGS.has(name))
48
+ s += '\n';
49
+ if (depth > MAX_ELEMENT_DEPTH)
50
+ throw TOO_DEEP;
51
+ },
52
+ ontext(text) {
53
+ if (skipping === 0)
54
+ s += text;
55
+ },
56
+ onclosetag(name) {
57
+ if (CONTAINER_TAGS.has(name)) {
58
+ if (skipping > 0)
59
+ skipping--;
60
+ }
61
+ else if (BLOCK_TAGS.has(name))
62
+ s += '\n';
63
+ depth--;
64
+ },
65
+ },
66
+ // HTML mode, not XML: an email body is HTML, with its void elements, its
67
+ // implied closes, its raw-text `<script>`, and its named entities.
68
+ { decodeEntities: true });
69
+ try {
70
+ parser.write(html);
71
+ parser.end();
72
+ }
73
+ catch (err) {
74
+ if (err !== TOO_DEEP)
75
+ throw err;
76
+ parser.reset();
77
+ }
78
+ // `&nbsp;` decodes to U+00A0, which LOOKS like a space and is not one: an
79
+ // agent grepping the extraction for "Para one" would miss a line that reads
80
+ // exactly that. Extracted text is for reading and searching, so the
81
+ // non-breaking space becomes an ordinary one.
82
+ const out = [];
83
+ for (const raw of s.replace(/ /g, ' ').split('\n')) {
84
+ const line = raw.trim();
85
+ if (line !== '')
86
+ out.push(line);
87
+ else if (out.length > 0 && out[out.length - 1] !== '')
88
+ out.push('');
89
+ }
90
+ while (out.length > 0 && out[out.length - 1] === '')
91
+ out.pop();
92
+ return out.join('\n');
93
+ }
94
+ /** `name (mimeType, N bytes)` with the parenthesis dropped when nothing is known.
95
+ * CR/LF in the metadata are shown as escapes rather than obeyed, so each
96
+ * attachment stays on ONE line and grep line numbers hold. */
97
+ function attachmentLine(a) {
98
+ // The extractors default a MISSING name, but an EMPTY (or whitespace-only)
99
+ // one reaches here as ''; without the same fallback the section printed a
100
+ // blank line — or one starting with the parenthesis — for that attachment.
101
+ const name = a.name.trim() === '' ? 'unnamed attachment' : a.name;
102
+ const details = [a.mimeType, a.sizeBytes !== undefined ? `${a.sizeBytes} bytes` : undefined]
103
+ .filter((d) => d !== undefined && d !== '')
104
+ .join(', ');
105
+ const line = details === '' ? name : `${name} (${details})`;
106
+ return line.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
107
+ }
108
+ /** A header value with its line breaks made visible instead of obeyed. */
109
+ function oneLine(value) {
110
+ return value.replace(/[\r\n]/g, (c) => (c === '\r' ? '\\r' : '\\n'));
111
+ }
112
+ /** Render the model into the marker summary + extraction text (see module doc). */
113
+ export function emailExtraction(model) {
114
+ const header = [];
115
+ // Absent OR blank fields are omitted — a header marker is never printed empty.
116
+ const pushHeader = (label, value) => {
117
+ if (value === undefined || value.trim() === '')
118
+ return;
119
+ // A header marker is ONE line. A From or Subject carrying CR/LF would
120
+ // otherwise forge further lines into the extraction — including lines that
121
+ // read like other markers — so the breaks are shown as escapes.
122
+ header.push(`[${label}] ${oneLine(value)}`);
123
+ };
124
+ pushHeader('from', model.from);
125
+ pushHeader('to', model.to);
126
+ pushHeader('cc', model.cc);
127
+ pushHeader('bcc', model.bcc);
128
+ pushHeader('subject', model.subject);
129
+ pushHeader('date', model.date);
130
+ const sections = [];
131
+ if (header.length > 0)
132
+ sections.push(header.join('\n'));
133
+ if (model.bodySource === 'rtf-only')
134
+ sections.push(RTF_ONLY_BODY_LINE);
135
+ else if (model.body !== '')
136
+ sections.push(model.body);
137
+ if (model.attachments.length > 0) {
138
+ sections.push(`[attachments]\n${model.attachments.map(attachmentLine).join('\n')}`);
139
+ }
140
+ const n = model.attachments.length;
141
+ const parts = ['email message'];
142
+ if (n > 0)
143
+ parts.push(`${n} attachment${n === 1 ? '' : 's'} listed (names only; not extracted)`);
144
+ if (model.bodySource === 'html')
145
+ parts.push('HTML body rendered as plain text');
146
+ if (model.bodySource === 'rtf-only')
147
+ parts.push('body is RTF; no plain-text part');
148
+ parts.push('formatting and full headers omitted');
149
+ return { summary: parts.join('; '), text: sections.join('\n\n') };
150
+ }
151
+ //# sourceMappingURL=email-text.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"email-text.js","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/email-text.ts"],"names":[],"mappings":"AAuBA,OAAO,EAAE,MAAM,EAAE,MAAM,aAAa,CAAC;AACrC,OAAO,EAAE,iBAAiB,EAAE,QAAQ,EAAE,MAAM,iBAAiB,CAAC;AA2B9D,yEAAyE;AACzE,MAAM,CAAC,MAAM,kBAAkB,GAAG,mCAAmC,CAAC;AAEtE;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,+EAA+E;AAC/E,MAAM,cAAc,GAAG,IAAI,GAAG,CAAC,CAAC,QAAQ,EAAE,OAAO,EAAE,MAAM,EAAE,OAAO,CAAC,CAAC,CAAC;AAErE,iFAAiF;AACjF,MAAM,UAAU,GAAG,IAAI,GAAG,CAAC;IACzB,GAAG,EAAE,KAAK,EAAE,SAAS,EAAE,SAAS,EAAE,QAAQ,EAAE,QAAQ,EAAE,MAAM;IAC5D,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,OAAO,EAAE,IAAI,EAAE,IAAI,EAAE,IAAI;IACpD,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,YAAY,EAAE,KAAK,EAAE,IAAI;IAC7D,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,IAAI,EAAE,IAAI;CACnC,CAAC,CAAC;AAEH,2EAA2E;AAC3E,4EAA4E;AAC5E,wEAAwE;AAExE,MAAM,UAAU,eAAe,CAAC,IAAY;IAC1C,IAAI,CAAC,GAAG,EAAE,CAAC;IACX,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,IAAI,QAAQ,GAAG,CAAC,CAAC,CAAC,oDAAoD;IACtE,MAAM,MAAM,GAAG,IAAI,MAAM,CACvB;QACE,SAAS,CAAC,IAAI;YACZ,KAAK,EAAE,CAAC;YACR,IAAI,cAAc,CAAC,GAAG,CAAC,IAAI,CAAC;gBAAE,QAAQ,EAAE,CAAC;iBACpC,IAAI,IAAI,KAAK,IAAI,IAAI,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC;gBAAE,CAAC,IAAI,IAAI,CAAC;YAC1D,IAAI,KAAK,GAAG,iBAAiB;gBAAE,MAAM,QAAQ,CAAC;QAChD,CAAC;QACD,MAAM,CAAC,IAAI;YACT,IAAI,QAAQ,KAAK,CAAC;gBAAE,CAAC,IAAI,IAAI,CAAC;QAChC,CAAC;QACD,UAAU,CAAC,IAAI;YACb,IAAI,cAAc,CAAC,GAAG,CAAC,IAAI,CAAC,EAAE,CAAC;gBAC7B,IAAI,QAAQ,GAAG,CAAC;oBAAE,QAAQ,EAAE,CAAC;YAC/B,CAAC;iBAAM,IAAI,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC;gBAAE,CAAC,IAAI,IAAI,CAAC;YAC3C,KAAK,EAAE,CAAC;QACV,CAAC;KACF;IACD,yEAAyE;IACzE,mEAAmE;IACnE,EAAE,cAAc,EAAE,IAAI,EAAE,CACzB,CAAC;IACF,IAAI,CAAC;QACH,MAAM,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC;QACnB,MAAM,CAAC,GAAG,EAAE,CAAC;IACf,CAAC;IAAC,OAAO,GAAG,EAAE,CAAC;QACb,IAAI,GAAG,KAAK,QAAQ;YAAE,MAAM,GAAG,CAAC;QAChC,MAAM,CAAC,KAAK,EAAE,CAAC;IACjB,CAAC;IAED,0EAA0E;IAC1E,4EAA4E;IAC5E,oEAAoE;IACpE,8CAA8C;IAC9C,MAAM,GAAG,GAAa,EAAE,CAAC;IACzB,KAAK,MAAM,GAAG,IAAI,CAAC,CAAC,OAAO,CAAC,IAAI,EAAE,GAAG,CAAC,CAAC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC;QACnD,MAAM,IAAI,GAAG,GAAG,CAAC,IAAI,EAAE,CAAC;QACxB,IAAI,IAAI,KAAK,EAAE;YAAE,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;aAC3B,IAAI,GAAG,CAAC,MAAM,GAAG,CAAC,IAAI,GAAG,CAAC,GAAG,CAAC,MAAM,GAAG,CAAC,CAAC,KAAK,EAAE;YAAE,GAAG,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACtE,CAAC;IACD,OAAO,GAAG,CAAC,MAAM,GAAG,CAAC,IAAI,GAAG,CAAC,GAAG,CAAC,MAAM,GAAG,CAAC,CAAC,KAAK,EAAE;QAAE,GAAG,CAAC,GAAG,EAAE,CAAC;IAC/D,OAAO,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AACxB,CAAC;AAED;;8DAE8D;AAC9D,SAAS,cAAc,CAAC,CAAkB;IACxC,2EAA2E;IAC3E,0EAA0E;IAC1E,2EAA2E;IAC3E,MAAM,IAAI,GAAG,CAAC,CAAC,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE,CAAC,CAAC,CAAC,oBAAoB,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC;IAClE,MAAM,OAAO,GAAG,CAAC,CAAC,CAAC,QAAQ,EAAE,CAAC,CAAC,SAAS,KAAK,SAAS,CAAC,CAAC,CAAC,GAAG,CAAC,CAAC,SAAS,QAAQ,CAAC,CAAC,CAAC,SAAS,CAAC;SACzF,MAAM,CAAC,CAAC,CAAC,EAAe,EAAE,CAAC,CAAC,KAAK,SAAS,IAAI,CAAC,KAAK,EAAE,CAAC;SACvD,IAAI,CAAC,IAAI,CAAC,CAAC;IACd,MAAM,IAAI,GAAG,OAAO,KAAK,EAAE,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,GAAG,IAAI,KAAK,OAAO,GAAG,CAAC;IAC5D,OAAO,IAAI,CAAC,OAAO,CAAC,SAAS,EAAE,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC;AACtE,CAAC;AAED,0EAA0E;AAC1E,SAAS,OAAO,CAAC,KAAa;IAC5B,OAAO,KAAK,CAAC,OAAO,CAAC,SAAS,EAAE,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,CAAC,KAAK,IAAI,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,CAAC,CAAC;AACvE,CAAC;AAED,mFAAmF;AACnF,MAAM,UAAU,eAAe,CAAC,KAAiB;IAC/C,MAAM,MAAM,GAAa,EAAE,CAAC;IAC5B,+EAA+E;IAC/E,MAAM,UAAU,GAAG,CAAC,KAAa,EAAE,KAAyB,EAAQ,EAAE;QACpE,IAAI,KAAK,KAAK,SAAS,IAAI,KAAK,CAAC,IAAI,EAAE,KAAK,EAAE;YAAE,OAAO;QACvD,sEAAsE;QACtE,2EAA2E;QAC3E,gEAAgE;QAChE,MAAM,CAAC,IAAI,CAAC,IAAI,KAAK,KAAK,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC;IAC9C,CAAC,CAAC;IACF,UAAU,CAAC,MAAM,EAAE,KAAK,CAAC,IAAI,CAAC,CAAC;IAC/B,UAAU,CAAC,IAAI,EAAE,KAAK,CAAC,EAAE,CAAC,CAAC;IAC3B,UAAU,CAAC,IAAI,EAAE,KAAK,CAAC,EAAE,CAAC,CAAC;IAC3B,UAAU,CAAC,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;IAC7B,UAAU,CAAC,SAAS,EAAE,KAAK,CAAC,OAAO,CAAC,CAAC;IACrC,UAAU,CAAC,MAAM,EAAE,KAAK,CAAC,IAAI,CAAC,CAAC;IAE/B,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,IAAI,MAAM,CAAC,MAAM,GAAG,CAAC;QAAE,QAAQ,CAAC,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC;IACxD,IAAI,KAAK,CAAC,UAAU,KAAK,UAAU;QAAE,QAAQ,CAAC,IAAI,CAAC,kBAAkB,CAAC,CAAC;SAClE,IAAI,KAAK,CAAC,IAAI,KAAK,EAAE;QAAE,QAAQ,CAAC,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC;IACtD,IAAI,KAAK,CAAC,WAAW,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACjC,QAAQ,CAAC,IAAI,CAAC,kBAAkB,KAAK,CAAC,WAAW,CAAC,GAAG,CAAC,cAAc,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACtF,CAAC;IAED,MAAM,CAAC,GAAG,KAAK,CAAC,WAAW,CAAC,MAAM,CAAC;IACnC,MAAM,KAAK,GAAG,CAAC,eAAe,CAAC,CAAC;IAChC,IAAI,CAAC,GAAG,CAAC;QAAE,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,cAAc,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,GAAG,qCAAqC,CAAC,CAAC;IACjG,IAAI,KAAK,CAAC,UAAU,KAAK,MAAM;QAAE,KAAK,CAAC,IAAI,CAAC,kCAAkC,CAAC,CAAC;IAChF,IAAI,KAAK,CAAC,UAAU,KAAK,UAAU;QAAE,KAAK,CAAC,IAAI,CAAC,iCAAiC,CAAC,CAAC;IACnF,KAAK,CAAC,IAAI,CAAC,qCAAqC,CAAC,CAAC;IAElD,OAAO,EAAE,OAAO,EAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,QAAQ,CAAC,IAAI,CAAC,MAAM,CAAC,EAAE,CAAC;AACpE,CAAC"}
@@ -0,0 +1,13 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ /**
3
+ * Extract the BODY text of a `.docx` (Word) document.
4
+ *
5
+ * A docx is a zip whose main part is `word/document.xml`. v1 extracts the body
6
+ * only — headers/footers are skipped, and the marker summary says so.
7
+ *
8
+ * - Paragraphs become lines. `<w:t>` runs are concatenated with NO separator
9
+ * (Word splits runs mid-word on formatting boundaries).
10
+ * - Tables become lines with cell text tab-separated, one line per row.
11
+ */
12
+ export declare function extractDocx(bytes: Buffer): ExtractResult;
13
+ //# sourceMappingURL=extract-docx.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"extract-docx.d.ts","sourceRoot":"","sources":["../../../../src/modules/workspace/file-readers/extract-docx.ts"],"names":[],"mappings":"AACA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,wBAAwB,CAAC;AAG5D;;;;;;;;;GASG;AACH,wBAAgB,WAAW,CAAC,KAAK,EAAE,MAAM,GAAG,aAAa,CAmDxD"}