@bevel-software/platform-core-backend 0.11.1 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (443) hide show
  1. package/THIRD-PARTY-NOTICES.md +1163 -425
  2. package/dist/core/create-core-server.js +1 -1
  3. package/dist/core/create-core-server.js.map +1 -1
  4. package/dist/core/create-core-services.d.ts +3 -1
  5. package/dist/core/create-core-services.d.ts.map +1 -1
  6. package/dist/core/create-core-services.js +7 -2
  7. package/dist/core/create-core-services.js.map +1 -1
  8. package/dist/core-config.d.ts +7 -0
  9. package/dist/core-config.d.ts.map +1 -1
  10. package/dist/core-config.js +9 -0
  11. package/dist/core-config.js.map +1 -1
  12. package/dist/modules/access/access-control.service.d.ts +1 -253
  13. package/dist/modules/access/access-control.service.d.ts.map +1 -1
  14. package/dist/modules/access/access-control.service.js +3 -632
  15. package/dist/modules/access/access-control.service.js.map +1 -1
  16. package/dist/modules/access/access-declarations.d.ts +2 -2
  17. package/dist/modules/access/access-declarations.d.ts.map +1 -1
  18. package/dist/modules/access/access-declarations.js +2 -2
  19. package/dist/modules/access/access-declarations.js.map +1 -1
  20. package/dist/modules/access/access-mutation.service.d.ts +3 -3
  21. package/dist/modules/access/access-mutation.service.d.ts.map +1 -1
  22. package/dist/modules/access/access-mutation.service.js +3 -3
  23. package/dist/modules/access/access-mutation.service.js.map +1 -1
  24. package/dist/modules/access/access.routes.js +4 -4
  25. package/dist/modules/access/access.routes.js.map +1 -1
  26. package/dist/modules/access/admin-locked-commit.d.ts +2 -2
  27. package/dist/modules/access/admin-locked-commit.d.ts.map +1 -1
  28. package/dist/modules/access/admin-locked-commit.js +2 -2
  29. package/dist/modules/access/admin-locked-commit.js.map +1 -1
  30. package/dist/modules/access/admin-route-helpers.js +1 -1
  31. package/dist/modules/access/admin-route-helpers.js.map +1 -1
  32. package/dist/modules/access/capability-registry.js +1 -1
  33. package/dist/modules/access/capability-registry.js.map +1 -1
  34. package/dist/modules/access/creator-access.d.ts +1 -45
  35. package/dist/modules/access/creator-access.d.ts.map +1 -1
  36. package/dist/modules/access/creator-access.js +4 -24
  37. package/dist/modules/access/creator-access.js.map +1 -1
  38. package/dist/modules/access/groups-admin.routes.js +1 -1
  39. package/dist/modules/access/groups-admin.routes.js.map +1 -1
  40. package/dist/modules/access/groups-admin.service.d.ts +1 -1
  41. package/dist/modules/access/groups-admin.service.d.ts.map +1 -1
  42. package/dist/modules/access/groups-admin.service.js +6 -5
  43. package/dist/modules/access/groups-admin.service.js.map +1 -1
  44. package/dist/modules/access/groups-edit.js +2 -2
  45. package/dist/modules/access/groups-edit.js.map +1 -1
  46. package/dist/modules/access/reference-scan.js +1 -1
  47. package/dist/modules/access/reference-scan.js.map +1 -1
  48. package/dist/modules/access/roles-admin.service.d.ts +1 -1
  49. package/dist/modules/access/roles-admin.service.d.ts.map +1 -1
  50. package/dist/modules/access/roles-admin.service.js +8 -7
  51. package/dist/modules/access/roles-admin.service.js.map +1 -1
  52. package/dist/modules/access/roles-edit.js +1 -1
  53. package/dist/modules/access/roles-edit.js.map +1 -1
  54. package/dist/modules/access/synced-groups-committer.js +4 -4
  55. package/dist/modules/access/synced-groups-committer.js.map +1 -1
  56. package/dist/modules/access/synced-groups-writer.js +2 -2
  57. package/dist/modules/access/synced-groups-writer.js.map +1 -1
  58. package/dist/modules/access-model/access-errors.d.ts +34 -0
  59. package/dist/modules/access-model/access-errors.d.ts.map +1 -0
  60. package/dist/modules/access-model/access-errors.js +40 -0
  61. package/dist/modules/access-model/access-errors.js.map +1 -0
  62. package/dist/modules/access-model/access-grammar.d.ts +266 -0
  63. package/dist/modules/access-model/access-grammar.d.ts.map +1 -0
  64. package/dist/modules/access-model/access-grammar.js +644 -0
  65. package/dist/modules/access-model/access-grammar.js.map +1 -0
  66. package/dist/modules/access-model/access-splice.d.ts +140 -0
  67. package/dist/modules/access-model/access-splice.d.ts.map +1 -0
  68. package/dist/modules/access-model/access-splice.js +389 -0
  69. package/dist/modules/access-model/access-splice.js.map +1 -0
  70. package/dist/modules/access-model/creator.d.ts +54 -0
  71. package/dist/modules/access-model/creator.d.ts.map +1 -0
  72. package/dist/modules/access-model/creator.js +30 -0
  73. package/dist/modules/access-model/creator.js.map +1 -0
  74. package/dist/modules/access-model/group-files.d.ts +83 -0
  75. package/dist/modules/access-model/group-files.d.ts.map +1 -0
  76. package/dist/modules/access-model/group-files.js +167 -0
  77. package/dist/modules/access-model/group-files.js.map +1 -0
  78. package/dist/modules/access-model/kb-read-filter.d.ts +41 -0
  79. package/dist/modules/access-model/kb-read-filter.d.ts.map +1 -0
  80. package/dist/modules/access-model/kb-read-filter.js +60 -0
  81. package/dist/modules/access-model/kb-read-filter.js.map +1 -0
  82. package/dist/modules/access-model/render-roles-yaml.d.ts +22 -0
  83. package/dist/modules/access-model/render-roles-yaml.d.ts.map +1 -0
  84. package/dist/modules/access-model/render-roles-yaml.js +56 -0
  85. package/dist/modules/access-model/render-roles-yaml.js.map +1 -0
  86. package/dist/modules/access-model/roles-yaml-guard.d.ts +56 -0
  87. package/dist/modules/access-model/roles-yaml-guard.d.ts.map +1 -0
  88. package/dist/modules/access-model/roles-yaml-guard.js +79 -0
  89. package/dist/modules/access-model/roles-yaml-guard.js.map +1 -0
  90. package/dist/modules/admin/admin-access.service.d.ts.map +1 -1
  91. package/dist/modules/admin/admin-access.service.js +1 -1
  92. package/dist/modules/admin/admin-access.service.js.map +1 -1
  93. package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
  94. package/dist/modules/code-mode/code-mode.tool.js +7 -1
  95. package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
  96. package/dist/modules/diff/diff.routes.js +4 -4
  97. package/dist/modules/diff/diff.routes.js.map +1 -1
  98. package/dist/modules/diff/diff.service.d.ts +1 -1
  99. package/dist/modules/diff/diff.service.d.ts.map +1 -1
  100. package/dist/modules/kb-fs/branch-name.d.ts +10 -0
  101. package/dist/modules/kb-fs/branch-name.d.ts.map +1 -0
  102. package/dist/modules/kb-fs/branch-name.js +76 -0
  103. package/dist/modules/kb-fs/branch-name.js.map +1 -0
  104. package/dist/modules/kb-fs/clone-config.d.ts +61 -0
  105. package/dist/modules/kb-fs/clone-config.d.ts.map +1 -0
  106. package/dist/modules/kb-fs/clone-config.js +69 -0
  107. package/dist/modules/kb-fs/clone-config.js.map +1 -0
  108. package/dist/modules/kb-fs/file-change-notifier.d.ts +38 -0
  109. package/dist/modules/kb-fs/file-change-notifier.d.ts.map +1 -0
  110. package/dist/modules/kb-fs/file-change-notifier.js +22 -0
  111. package/dist/modules/kb-fs/file-change-notifier.js.map +1 -0
  112. package/dist/modules/kb-fs/locking-filesystem.d.ts +137 -0
  113. package/dist/modules/kb-fs/locking-filesystem.d.ts.map +1 -0
  114. package/dist/modules/kb-fs/locking-filesystem.js +553 -0
  115. package/dist/modules/kb-fs/locking-filesystem.js.map +1 -0
  116. package/dist/modules/kb-fs/mutex.d.ts +11 -0
  117. package/dist/modules/kb-fs/mutex.d.ts.map +1 -0
  118. package/dist/modules/kb-fs/mutex.js +23 -0
  119. package/dist/modules/kb-fs/mutex.js.map +1 -0
  120. package/dist/modules/kb-fs/read-only-filesystem.d.ts +24 -0
  121. package/dist/modules/kb-fs/read-only-filesystem.d.ts.map +1 -0
  122. package/dist/modules/kb-fs/read-only-filesystem.js +39 -0
  123. package/dist/modules/kb-fs/read-only-filesystem.js.map +1 -0
  124. package/dist/modules/plugins/join-proposals.d.ts +1 -1
  125. package/dist/modules/plugins/join-proposals.d.ts.map +1 -1
  126. package/dist/modules/plugins/join-proposals.js +1 -1
  127. package/dist/modules/plugins/join-proposals.js.map +1 -1
  128. package/dist/modules/plugins/join-requests.service.d.ts.map +1 -1
  129. package/dist/modules/plugins/join-requests.service.js +1 -1
  130. package/dist/modules/plugins/join-requests.service.js.map +1 -1
  131. package/dist/modules/plugins/plugin-provision.service.js +3 -3
  132. package/dist/modules/plugins/plugin-provision.service.js.map +1 -1
  133. package/dist/modules/plugins/plugins.routes.js +2 -2
  134. package/dist/modules/plugins/plugins.routes.js.map +1 -1
  135. package/dist/modules/plugins/plugins.service.js +1 -1
  136. package/dist/modules/plugins/plugins.service.js.map +1 -1
  137. package/dist/modules/secrets-vault/secrets-vault.routes.js +1 -1
  138. package/dist/modules/secrets-vault/secrets-vault.routes.js.map +1 -1
  139. package/dist/modules/skills/pending-skills.service.d.ts.map +1 -1
  140. package/dist/modules/skills/pending-skills.service.js +1 -1
  141. package/dist/modules/skills/pending-skills.service.js.map +1 -1
  142. package/dist/modules/skills/skills.service.js +1 -1
  143. package/dist/modules/skills/skills.service.js.map +1 -1
  144. package/dist/modules/tool-helpers/tool-context.d.ts +2 -2
  145. package/dist/modules/tool-helpers/tool-context.d.ts.map +1 -1
  146. package/dist/modules/tool-helpers/tool-context.js +4 -4
  147. package/dist/modules/tool-helpers/tool-context.js.map +1 -1
  148. package/dist/modules/tool-manuals/mcp-server-edit.service.d.ts +1 -1
  149. package/dist/modules/tool-manuals/mcp-server-edit.service.d.ts.map +1 -1
  150. package/dist/modules/tool-manuals/mcp-server-edit.service.js +1 -1
  151. package/dist/modules/tool-manuals/mcp-server-edit.service.js.map +1 -1
  152. package/dist/modules/tool-manuals/tool-manuals.routes.d.ts +1 -1
  153. package/dist/modules/tool-manuals/tool-manuals.routes.d.ts.map +1 -1
  154. package/dist/modules/tool-manuals/tool-manuals.routes.js +1 -1
  155. package/dist/modules/tool-manuals/tool-manuals.routes.js.map +1 -1
  156. package/dist/modules/tool-manuals/tool-manuals.service.js +1 -1
  157. package/dist/modules/tool-manuals/tool-manuals.service.js.map +1 -1
  158. package/dist/modules/tool-manuals/tool-manuals.tools.js +1 -1
  159. package/dist/modules/tool-manuals/tool-manuals.tools.js.map +1 -1
  160. package/dist/modules/workflow/agent-tools/workflow.tools.js +1 -1
  161. package/dist/modules/workflow/agent-tools/workflow.tools.js.map +1 -1
  162. package/dist/modules/workflow/file-lock.service.js +1 -1
  163. package/dist/modules/workflow/file-lock.service.js.map +1 -1
  164. package/dist/modules/workflow/git/git.service.d.ts +2 -2
  165. package/dist/modules/workflow/git/git.service.d.ts.map +1 -1
  166. package/dist/modules/workflow/git/git.service.js +5 -5
  167. package/dist/modules/workflow/git/git.service.js.map +1 -1
  168. package/dist/modules/workflow/git/pull-request.service.js +1 -1
  169. package/dist/modules/workflow/git/pull-request.service.js.map +1 -1
  170. package/dist/modules/workflow/review-workflow/review-workflow.service.js +2 -2
  171. package/dist/modules/workflow/review-workflow/review-workflow.service.js.map +1 -1
  172. package/dist/modules/workflow/session-ontology.service.js +1 -1
  173. package/dist/modules/workflow/session-ontology.service.js.map +1 -1
  174. package/dist/modules/workflow/workflow.routes.js +2 -2
  175. package/dist/modules/workflow/workflow.routes.js.map +1 -1
  176. package/dist/modules/workflow/workflow.service.d.ts +1 -1
  177. package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
  178. package/dist/modules/workflow/workflow.service.js +4 -4
  179. package/dist/modules/workflow/workflow.service.js.map +1 -1
  180. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
  181. package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
  182. package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
  183. package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
  184. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
  185. package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
  186. package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
  187. package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
  188. package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
  189. package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
  190. package/dist/modules/workspace/file-readers/document-reader.js +59 -0
  191. package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
  192. package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
  193. package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
  194. package/dist/modules/workspace/file-readers/email-reader.js +19 -0
  195. package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
  196. package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
  197. package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
  198. package/dist/modules/workspace/file-readers/email-text.js +151 -0
  199. package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
  200. package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
  201. package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
  202. package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
  203. package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
  204. package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
  205. package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
  206. package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
  207. package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
  208. package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
  209. package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
  210. package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
  211. package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
  212. package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
  213. package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
  214. package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
  215. package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
  216. package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
  217. package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
  218. package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
  219. package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
  220. package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
  221. package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
  222. package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
  223. package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
  224. package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
  225. package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
  226. package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
  227. package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
  228. package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
  229. package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
  230. package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
  231. package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
  232. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
  233. package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
  234. package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
  235. package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
  236. package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
  237. package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
  238. package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
  239. package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
  240. package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
  241. package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
  242. package/dist/modules/workspace/file-readers/file-reader.js +55 -0
  243. package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
  244. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
  245. package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
  246. package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
  247. package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
  248. package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
  249. package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
  250. package/dist/modules/workspace/file-readers/image-read.js +108 -0
  251. package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
  252. package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
  253. package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
  254. package/dist/modules/workspace/file-readers/image-reader.js +30 -0
  255. package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
  256. package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
  257. package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
  258. package/dist/modules/workspace/file-readers/odf-text.js +116 -0
  259. package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
  260. package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
  261. package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
  262. package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
  263. package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
  264. package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
  265. package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
  266. package/dist/modules/workspace/file-readers/text-reader.js +117 -0
  267. package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
  268. package/dist/modules/workspace/startup/kb-startup-runner.js +1 -1
  269. package/dist/modules/workspace/startup/kb-startup-runner.js.map +1 -1
  270. package/dist/modules/workspace/startup/steps/groups-to-plugins.step.js +1 -1
  271. package/dist/modules/workspace/startup/steps/groups-to-plugins.step.js.map +1 -1
  272. package/dist/modules/workspace/startup/steps/roles-yaml.step.d.ts +1 -1
  273. package/dist/modules/workspace/startup/steps/roles-yaml.step.js +2 -2
  274. package/dist/modules/workspace/startup/steps/roles-yaml.step.js.map +1 -1
  275. package/dist/modules/workspace/startup/steps/seed-tree.js +1 -1
  276. package/dist/modules/workspace/startup/steps/seed-tree.js.map +1 -1
  277. package/dist/modules/workspace/workspace.routes.d.ts +1 -1
  278. package/dist/modules/workspace/workspace.routes.d.ts.map +1 -1
  279. package/dist/modules/workspace/workspace.routes.js +5 -4
  280. package/dist/modules/workspace/workspace.routes.js.map +1 -1
  281. package/dist/modules/workspace/workspace.service.d.ts +2 -9
  282. package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
  283. package/dist/modules/workspace/workspace.service.js +14 -15
  284. package/dist/modules/workspace/workspace.service.js.map +1 -1
  285. package/dist/modules/workspace/workspace.tools.d.ts +2 -1
  286. package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
  287. package/dist/modules/workspace/workspace.tools.js +161 -18
  288. package/dist/modules/workspace/workspace.tools.js.map +1 -1
  289. package/dist/shared/domain-errors.d.ts +202 -0
  290. package/dist/shared/domain-errors.d.ts.map +1 -0
  291. package/dist/shared/domain-errors.js +303 -0
  292. package/dist/shared/domain-errors.js.map +1 -0
  293. package/dist/shared/workspace-id.d.ts +27 -0
  294. package/dist/shared/workspace-id.d.ts.map +1 -0
  295. package/dist/shared/workspace-id.js +36 -0
  296. package/dist/shared/workspace-id.js.map +1 -0
  297. package/package.json +9 -4
  298. package/src/core/create-core-server.ts +1 -1
  299. package/src/core/create-core-services.ts +8 -2
  300. package/src/core-config.ts +9 -0
  301. package/src/modules/access/__tests__/access-control.service.test.ts +1 -1
  302. package/src/modules/access/__tests__/access-groups.test.ts +6 -5
  303. package/src/modules/access/__tests__/access-md-format.test.ts +3 -3
  304. package/src/modules/access/__tests__/access-mutation.service.test.ts +1 -1
  305. package/src/modules/access/__tests__/admin-locked-commit.test.ts +1 -1
  306. package/src/modules/access/__tests__/admin-route-helpers.test.ts +1 -1
  307. package/src/modules/access/__tests__/roles-admin.service.test.ts +2 -2
  308. package/src/modules/access/__tests__/roles-edit.test.ts +1 -1
  309. package/src/modules/access/__tests__/synced-groups-committer.test.ts +1 -1
  310. package/src/modules/access/__tests__/synced-groups-writer.test.ts +1 -1
  311. package/src/modules/access/access-control.service.ts +23 -768
  312. package/src/modules/access/access-declarations.ts +2 -2
  313. package/src/modules/access/access-mutation.service.ts +3 -3
  314. package/src/modules/access/access.routes.ts +5 -5
  315. package/src/modules/access/admin-locked-commit.ts +2 -2
  316. package/src/modules/access/admin-route-helpers.ts +1 -1
  317. package/src/modules/access/capability-registry.ts +1 -1
  318. package/src/modules/access/creator-access.ts +9 -72
  319. package/src/modules/access/groups-admin.routes.ts +1 -1
  320. package/src/modules/access/groups-admin.service.ts +6 -7
  321. package/src/modules/access/groups-edit.ts +2 -2
  322. package/src/modules/access/reference-scan.ts +1 -1
  323. package/src/modules/access/roles-admin.service.ts +8 -8
  324. package/src/modules/access/roles-edit.ts +1 -1
  325. package/src/modules/access/synced-groups-committer.ts +4 -4
  326. package/src/modules/access/synced-groups-writer.ts +2 -2
  327. package/src/modules/access-model/__tests__/access-grammar.test.ts +31 -0
  328. package/src/modules/{access → access-model}/__tests__/access-splice.test.ts +268 -268
  329. package/src/modules/{access → access-model}/access-errors.ts +1 -1
  330. package/src/modules/access-model/access-grammar.ts +778 -0
  331. package/src/modules/{access → access-model}/access-splice.ts +1 -1
  332. package/src/modules/access-model/creator.ts +79 -0
  333. package/src/modules/{access → access-model}/group-files.ts +1 -1
  334. package/src/modules/{access → access-model}/render-roles-yaml.ts +1 -1
  335. package/src/modules/{access → access-model}/roles-yaml-guard.ts +2 -2
  336. package/src/modules/admin/admin-access.service.ts +2 -1
  337. package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
  338. package/src/modules/code-mode/code-mode.tool.ts +7 -1
  339. package/src/modules/diff/__tests__/diff.routes.rejectPathsLocked.test.ts +2 -2
  340. package/src/modules/diff/__tests__/diff.service.seed-atomicity.test.ts +3 -2
  341. package/src/modules/diff/__tests__/diff.service.test.ts +3 -2
  342. package/src/modules/diff/diff.routes.ts +4 -4
  343. package/src/modules/diff/diff.service.ts +1 -1
  344. package/src/modules/{workflow/git → kb-fs}/__tests__/branch-name.test.ts +1 -1
  345. package/src/modules/{workflow → kb-fs}/__tests__/locking-filesystem.test.ts +1 -1
  346. package/src/modules/{workflow/git → kb-fs}/branch-name.ts +1 -1
  347. package/src/modules/{workflow → kb-fs}/locking-filesystem.ts +2 -2
  348. package/src/modules/plugins/__tests__/join-requests.service.test.ts +2 -4
  349. package/src/modules/plugins/__tests__/plugin-index.service.test.ts +1 -1
  350. package/src/modules/plugins/__tests__/plugins.routes.test.ts +1 -1
  351. package/src/modules/plugins/join-proposals.ts +1 -1
  352. package/src/modules/plugins/join-requests.service.ts +2 -1
  353. package/src/modules/plugins/plugin-provision.service.ts +3 -3
  354. package/src/modules/plugins/plugins.routes.ts +2 -2
  355. package/src/modules/plugins/plugins.service.ts +1 -1
  356. package/src/modules/secrets-vault/secrets-vault.routes.ts +1 -1
  357. package/src/modules/skills/__tests__/skills.service.test.ts +1 -1
  358. package/src/modules/skills/pending-skills.service.ts +2 -1
  359. package/src/modules/skills/skills.service.ts +1 -1
  360. package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
  361. package/src/modules/tool-helpers/tool-context.ts +6 -5
  362. package/src/modules/tool-manuals/__tests__/mcp-server-edit.service.test.ts +1 -1
  363. package/src/modules/tool-manuals/__tests__/tool-manuals.archive.route.test.ts +1 -1
  364. package/src/modules/tool-manuals/__tests__/tool-manuals.detail.route.test.ts +1 -1
  365. package/src/modules/tool-manuals/__tests__/tool-manuals.mcp-oauth.test.ts +1 -1
  366. package/src/modules/tool-manuals/__tests__/tool-manuals.service.test.ts +1 -1
  367. package/src/modules/tool-manuals/mcp-server-edit.service.ts +2 -1
  368. package/src/modules/tool-manuals/tool-manuals.routes.ts +2 -1
  369. package/src/modules/tool-manuals/tool-manuals.service.ts +1 -1
  370. package/src/modules/tool-manuals/tool-manuals.tools.ts +1 -1
  371. package/src/modules/workflow/__tests__/preserve-roles-yaml.test.ts +1 -1
  372. package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +1 -1
  373. package/src/modules/workflow/__tests__/workflow.service.facade.test.ts +1 -1
  374. package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +1 -1
  375. package/src/modules/workflow/agent-tools/workflow.tools.ts +1 -1
  376. package/src/modules/workflow/file-lock.service.ts +1 -1
  377. package/src/modules/workflow/git/__tests__/git.service.accessGating.test.ts +1 -1
  378. package/src/modules/workflow/git/__tests__/git.service.deleteBranch.test.ts +1 -1
  379. package/src/modules/workflow/git/__tests__/git.service.diffFileAtCommit.test.ts +1 -1
  380. package/src/modules/workflow/git/__tests__/git.service.diffFileBetweenBranches.test.ts +1 -1
  381. package/src/modules/workflow/git/__tests__/git.service.pull.test.ts +1 -1
  382. package/src/modules/workflow/git/git.service.ts +5 -5
  383. package/src/modules/workflow/git/pull-request.service.ts +1 -1
  384. package/src/modules/workflow/review-workflow/__tests__/cancel-pr.test.ts +1 -1
  385. package/src/modules/workflow/review-workflow/review-workflow.service.ts +2 -2
  386. package/src/modules/workflow/session-ontology.service.ts +1 -1
  387. package/src/modules/workflow/workflow.routes.ts +2 -2
  388. package/src/modules/workflow/workflow.service.ts +5 -5
  389. package/src/modules/workspace/__tests__/workspace.routes.create-grant.test.ts +1 -1
  390. package/src/modules/workspace/__tests__/workspace.routes.delete.test.ts +1 -1
  391. package/src/modules/workspace/__tests__/workspace.routes.download.test.ts +2 -2
  392. package/src/modules/workspace/__tests__/workspace.routes.read-gate.test.ts +1 -1
  393. package/src/modules/workspace/__tests__/workspace.service.read-filter.test.ts +2 -5
  394. package/src/modules/workspace/__tests__/workspace.service.test.ts +13 -1
  395. package/src/modules/workspace/__tests__/workspace.tools.test.ts +501 -3
  396. package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
  397. package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
  398. package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
  399. package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
  400. package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
  401. package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
  402. package/src/modules/workspace/file-readers/document-reader.ts +64 -0
  403. package/src/modules/workspace/file-readers/email-reader.ts +21 -0
  404. package/src/modules/workspace/file-readers/email-text.ts +193 -0
  405. package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
  406. package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
  407. package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
  408. package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
  409. package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
  410. package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
  411. package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
  412. package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
  413. package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
  414. package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
  415. package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
  416. package/src/modules/workspace/file-readers/file-reader.ts +104 -0
  417. package/src/modules/workspace/file-readers/image-read.ts +122 -0
  418. package/src/modules/workspace/file-readers/image-reader.ts +39 -0
  419. package/src/modules/workspace/file-readers/odf-text.ts +123 -0
  420. package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
  421. package/src/modules/workspace/file-readers/text-reader.ts +131 -0
  422. package/src/modules/workspace/startup/kb-startup-runner.ts +1 -1
  423. package/src/modules/workspace/startup/steps/__tests__/steps.test.ts +1 -1
  424. package/src/modules/workspace/startup/steps/groups-to-plugins.step.ts +1 -1
  425. package/src/modules/workspace/startup/steps/roles-yaml.step.ts +2 -2
  426. package/src/modules/workspace/startup/steps/seed-tree.ts +1 -1
  427. package/src/modules/workspace/workspace.routes.ts +6 -5
  428. package/src/modules/workspace/workspace.service.ts +14 -18
  429. package/src/modules/workspace/workspace.tools.ts +177 -15
  430. package/src/shared/__tests__/join-request.test.ts +1 -1
  431. package/src/shared/__tests__/workspace-id.test.ts +17 -0
  432. package/src/{modules/workflow/workflow.errors.ts → shared/domain-errors.ts} +6 -1
  433. package/src/shared/workspace-id.ts +36 -0
  434. /package/src/modules/{access → access-model}/__tests__/kb-read-filter.test.ts +0 -0
  435. /package/src/modules/{access → access-model}/__tests__/roles-yaml-guard.test.ts +0 -0
  436. /package/src/modules/{access → access-model}/kb-read-filter.ts +0 -0
  437. /package/src/modules/{workflow/git → kb-fs}/__tests__/clone-config.test.ts +0 -0
  438. /package/src/modules/{workflow → kb-fs}/__tests__/file-change-notifier.test.ts +0 -0
  439. /package/src/modules/{workflow/git → kb-fs}/__tests__/mutex.test.ts +0 -0
  440. /package/src/modules/{workflow/git → kb-fs}/clone-config.ts +0 -0
  441. /package/src/modules/{workflow → kb-fs}/file-change-notifier.ts +0 -0
  442. /package/src/modules/{workflow/git → kb-fs}/mutex.ts +0 -0
  443. /package/src/modules/{workflow → kb-fs}/read-only-filesystem.ts +0 -0
@@ -0,0 +1,134 @@
1
+ import MsgReaderImport from '@kenjiuno/msgreader';
2
+ import type { ExtractResult } from './doc-extract.types.js';
3
+ import { emailExtraction, htmlToEmailText, type EmailAttachment, type EmailModel } from './email-text.js';
4
+ import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
5
+
6
+ /**
7
+ * Extract a `.msg` (Outlook item, CFB container) email into the shared email
8
+ * text shape (see `email-text.ts`).
9
+ *
10
+ * Parsing is `@kenjiuno/msgreader` (HiraokaHyperTools, Apache-2.0) — the
11
+ * maintained MAPI/CFB reader. It never throws for bad content of its own
12
+ * accord: unparseable bytes come back as `{ dataType: null, error }`, which
13
+ * maps onto the typed could-not-be-parsed failure here.
14
+ *
15
+ * Body preference mirrors `.eml`: the plain-text `PidTagBody` first, an HTML
16
+ * body stripped to text second. An Outlook item whose body exists ONLY as
17
+ * compressed RTF is degraded honestly — the extraction says
18
+ * "[body is RTF; no plain-text part]" instead of pretending to decode RTF.
19
+ */
20
+ export function extractMsg(bytes: Buffer): ExtractResult {
21
+ if (bytes.length > MAX_DOC_PART_BYTES) {
22
+ return {
23
+ ok: false,
24
+ message: `could not be extracted as a .msg (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
25
+ };
26
+ }
27
+ let fields: FieldsData;
28
+ try {
29
+ // DataView over the Buffer's exact region — no copy, and msgreader never
30
+ // sees bytes outside the file.
31
+ fields = new MsgReader(new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength)).getFileData();
32
+ } catch (err) {
33
+ return { ok: false, message: `could not be parsed as a .msg (${(err as Error).message})` };
34
+ }
35
+ if (fields.dataType !== 'msg') {
36
+ return {
37
+ ok: false,
38
+ message: `could not be parsed as a .msg (${fields.error ?? 'not an Outlook message file'})`,
39
+ };
40
+ }
41
+ return { ok: true, ...emailExtraction(msgModel(fields)) };
42
+ }
43
+
44
+ /** msgreader's field data, shaped into the format-independent email model. */
45
+ function msgModel(fields: FieldsData): EmailModel {
46
+ const recipients = fields.recipients ?? [];
47
+ const text = fields.body !== undefined && fields.body.trim() !== '' ? fields.body : undefined;
48
+ const html = htmlBody(fields);
49
+ const body = text ?? (html !== undefined ? htmlToEmailText(html) : '');
50
+ const bodySource: EmailModel['bodySource'] =
51
+ text !== undefined ? 'text' : html !== undefined ? 'html' : fields.compressedRtf !== undefined ? 'rtf-only' : 'none';
52
+ const date = fields.clientSubmitTime ?? fields.messageDeliveryTime;
53
+ return {
54
+ from: mailboxText(fields.senderName, fields.senderSmtpAddress ?? fields.senderEmail),
55
+ to: recipientList(recipients, 'to'),
56
+ cc: recipientList(recipients, 'cc'),
57
+ bcc: recipientList(recipients, 'bcc'),
58
+ subject: fields.subject,
59
+ date: date !== undefined ? isoDate(date) : undefined,
60
+ body: body.replace(/\s+$/, ''),
61
+ bodySource,
62
+ attachments: (fields.attachments ?? []).map(
63
+ (a): EmailAttachment => ({
64
+ name: a.fileName ?? a.fileNameShort ?? a.name ?? 'unnamed attachment',
65
+ mimeType: a.attachMimeTag,
66
+ sizeBytes: a.contentLength,
67
+ }),
68
+ ),
69
+ };
70
+ }
71
+
72
+ /** The HTML body, whichever MAPI property carries it (string, or utf-8 bytes). */
73
+ function htmlBody(fields: FieldsData): string | undefined {
74
+ if (fields.bodyHtml !== undefined && fields.bodyHtml.trim() !== '') return fields.bodyHtml;
75
+ if (fields.html instanceof Uint8Array && fields.html.length > 0) {
76
+ return Buffer.from(fields.html).toString('utf8');
77
+ }
78
+ return undefined;
79
+ }
80
+
81
+ /** `Name <addr>` / `Name` / `addr` — whatever the message carries. */
82
+ function mailboxText(name: string | undefined, address: string | undefined): string | undefined {
83
+ const n = name?.trim() ?? '';
84
+ const a = address?.trim() ?? '';
85
+ if (n !== '' && a !== '' && n !== a) return `${n} <${a}>`;
86
+ if (a !== '') return a;
87
+ return n !== '' ? n : undefined;
88
+ }
89
+
90
+ /**
91
+ * The bucket a recipient belongs to. msgreader maps the MAPI `PidTagRecipientType`
92
+ * values 1/2/3 to these strings itself (lib/MsgReader.js, the `recipType` case)
93
+ * — but ONLY those three: any other raw PT_LONG value, e.g. `MAPI_TO | MAPI_P1`
94
+ * (0x10000001) on a resubmitted message, leaks through as a NUMBER despite the
95
+ * `'to' | 'cc' | 'bcc'` typing. Mask the resubmit/submitted flag bits and remap
96
+ * so such a recipient keeps its line instead of vanishing; anything else
97
+ * (including untyped) counts as `to`, matching the frontend's `msgMessage.ts`.
98
+ */
99
+ function recipientBucket(recipType: unknown): 'to' | 'cc' | 'bcc' {
100
+ if (recipType === 'to' || recipType === 'cc' || recipType === 'bcc') return recipType;
101
+ if (typeof recipType === 'number') {
102
+ const base = recipType & 0x0fffffff; // strip MAPI_SUBMITTED (0x80000000) / MAPI_P1 (0x10000000)
103
+ if (base === 2) return 'cc';
104
+ if (base === 3) return 'bcc';
105
+ }
106
+ return 'to';
107
+ }
108
+
109
+ /** The comma-joined mailboxes of one recipient type. Untyped recipients count as `to`. */
110
+ function recipientList(recipients: readonly FieldsData[], type: 'to' | 'cc' | 'bcc'): string | undefined {
111
+ const s = recipients
112
+ .filter((r) => recipientBucket(r.recipType) === type)
113
+ .map((r) => mailboxText(r.name, r.smtpAddress ?? r.email))
114
+ .filter((t): t is string => t !== undefined)
115
+ .join(', ');
116
+ return s === '' ? undefined : s;
117
+ }
118
+
119
+ /** msgreader emits RFC-1123 GMT strings; normalize to ISO, keep raw when unparseable. */
120
+ function isoDate(value: string): string {
121
+ const d = new Date(value);
122
+ return Number.isNaN(d.getTime()) ? value : d.toISOString();
123
+ }
124
+
125
+ /**
126
+ * CJS/ESM interop: msgreader is CJS with a transpiled `exports.default`.
127
+ * Vitest's transform hands the class straight through the default import, but
128
+ * NATIVE Node ESM (the built `dist/`) hands the exports OBJECT — so unwrap
129
+ * `.default` when it is there.
130
+ */
131
+ type MsgReaderClass = typeof MsgReaderImport;
132
+ const MsgReader: MsgReaderClass =
133
+ (MsgReaderImport as unknown as { default?: MsgReaderClass }).default ?? MsgReaderImport;
134
+ type FieldsData = ReturnType<InstanceType<MsgReaderClass>['getFileData']>;
@@ -0,0 +1,63 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ import { odfParagraphLines, readOdfContentXml } from './odf-text.js';
3
+ import { localBlocks, localElementBlocks, removeLocalElements } from './ooxml-text.js';
4
+
5
+ /**
6
+ * Extract the text of a `.odp` (OpenDocument Presentation) deck.
7
+ *
8
+ * Slides are the `<draw:page>` elements of `content.xml`, in DOCUMENT order —
9
+ * ODF orders slides in the file itself, so unlike pptx there is no numeric
10
+ * filename sort. Each slide is emitted under a `[slide N]` marker (N = 1-based
11
+ * position); speaker notes (`<presentation:notes>` inside the page) follow
12
+ * under `[slide N notes]` when non-empty. Within a slide, each `<text:p>` in
13
+ * its frames is a line; spans concatenate with no separator.
14
+ */
15
+ export function extractOdp(bytes: Buffer): ExtractResult {
16
+ const content = readOdfContentXml(bytes, '.odp');
17
+ if (!content.ok) return content;
18
+
19
+ const pages = drawPageBlocks(content.xml);
20
+ if (pages.length === 0) {
21
+ return { ok: false, message: 'could not be parsed as a .odp (no draw:page elements in content.xml)' };
22
+ }
23
+
24
+ const lines: string[] = [];
25
+ let anyNotes = false;
26
+ pages.forEach((page, i) => {
27
+ // Split the notes part out FIRST so its paragraphs don't render as slide text.
28
+ // Notes read by the parser and matched on their LOCAL name: a comment that
29
+ // resembled `<presentation:notes>` used to be emitted as real speaker notes,
30
+ // and a deck binding the presentation namespace to another prefix had none.
31
+ const notesXml = localBlocks(page, 'notes')[0] ?? '';
32
+ // The slide's own text is the page with the notes ELEMENTS removed by
33
+ // their parsed boundaries — global string replacement of the notes BODY
34
+ // also deleted slide text that happened to serialize identically to it.
35
+ const slideXml = removeLocalElements(page, ['notes']);
36
+ lines.push(`[slide ${i + 1}]`);
37
+ lines.push(...odfParagraphLines(slideXml));
38
+ const noteLines = odfParagraphLines(notesXml);
39
+ if (noteLines.length > 0) {
40
+ anyNotes = true;
41
+ lines.push(`[slide ${i + 1} notes]`);
42
+ lines.push(...noteLines);
43
+ }
44
+ });
45
+ return {
46
+ ok: true,
47
+ summary: `${pages.length} slide${pages.length === 1 ? '' : 's'}${anyNotes ? ' + notes' : ''}; layout, images and formatting omitted`,
48
+ text: lines.join('\n'),
49
+ };
50
+ }
51
+
52
+ /**
53
+ * The `<draw:page>…</draw:page>` bodies in document order (pages never nest).
54
+ * A SELF-CLOSING `<draw:page/>` is a legal, fully blank slide — it yields ''
55
+ * so the deck's numbering (and a deliberately blank deck) stays correct.
56
+ */
57
+ function drawPageBlocks(xml: string): string[] {
58
+ // The shared quote-aware scanner: a `/>` INSIDE a quoted attribute value (a
59
+ // page name like `a/>b`) is part of the value, never the self-closing
60
+ // delimiter, and a page whose close tag is missing costs one scan of the
61
+ // document rather than one per opener (see `xmlElementBlocks`).
62
+ return localElementBlocks(xml, ['page']).map((e) => e.body ?? '');
63
+ }
@@ -0,0 +1,182 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ import {
3
+ attrByLocalName,
4
+ decodeXmlEntities,
5
+ localElementBlocks,
6
+ localName,
7
+ walkLocalElementBlocks,
8
+ } from './ooxml-text.js';
9
+ import {
10
+ odfParagraphBlocks,
11
+ odfParagraphText,
12
+ readOdfContentXml,
13
+ } from './odf-text.js';
14
+
15
+ /**
16
+ * Per-sheet extraction caps — the SAME bounds as the xlsx extractor. ODF is
17
+ * fond of `table:number-columns-repeated="16384"` (or a million empty trailing
18
+ * rows) to pad a sheet to the grid, so repeats are expanded BOUNDED and the
19
+ * extraction says when it truncated (a `[sheet truncated …]` line right under
20
+ * the sheet marker). Trailing EMPTY cells/rows are trimmed before their
21
+ * repeats are applied at all, so grid padding never counts as truncation.
22
+ */
23
+ const MAX_ROWS_PER_SHEET = 10_000;
24
+ const MAX_COLS_PER_SHEET = 200;
25
+
26
+ /**
27
+ * Extract a `.ods` (OpenDocument Spreadsheet) workbook: per `<table:table>`
28
+ * (sheet) a `[sheet: Name]` marker (the `table:name` attribute), then the rows
29
+ * as tab-separated cell text. A cell's text is its `<text:p>` content
30
+ * (multiple paragraphs join with a space — a newline would break the row
31
+ * line); covered cells (under a merge) render empty.
32
+ */
33
+ export function extractOds(bytes: Buffer): ExtractResult {
34
+ const content = readOdfContentXml(bytes, '.ods');
35
+ if (!content.ok) return content;
36
+
37
+ const tables = tableBlocks(content.xml);
38
+ if (tables.length === 0) {
39
+ return { ok: false, message: 'could not be parsed as a .ods (no table:table elements in content.xml)' };
40
+ }
41
+
42
+ const lines: string[] = [];
43
+ for (const table of tables) {
44
+ lines.push(`[sheet: ${table.name}]`);
45
+ const { rows, truncated } = expandRows(table.xml);
46
+ if (truncated.length > 0) lines.push(`[sheet truncated to the ${truncated.join(' and ')}]`);
47
+ lines.push(...rows.map((cells) => cells.join('\t')));
48
+ }
49
+ return {
50
+ ok: true,
51
+ summary: `${tables.length} sheet${tables.length === 1 ? '' : 's'}, rows as tab-separated values; formulas, formatting and charts omitted`,
52
+ text: lines.join('\n'),
53
+ };
54
+ }
55
+
56
+ /**
57
+ * The `<table:table>` blocks with their decoded `table:name`, in document
58
+ * order. Non-greedy close — a nested table (legal in ODF text documents, not
59
+ * produced by spreadsheets) would end the outer block early, degrading
60
+ * grouping but never crashing.
61
+ */
62
+ function tableBlocks(xml: string): Array<{ name: string; xml: string }> {
63
+ // Read by the parser and matched on the LOCAL name: a comment or CDATA
64
+ // section holding a table-looking fragment used to answer as a real sheet,
65
+ // and a document binding the table namespace to another prefix had none.
66
+ const out: Array<{ name: string; xml: string }> = [];
67
+ for (const table of localElementBlocks(xml, ['table'])) {
68
+ const raw = attrByLocalName(table.attributes, 'name');
69
+ out.push({
70
+ // Control separators become spaces: a name holding an encoded newline
71
+ // or tab (`&#10;`) would corrupt the `[sheet: …]` marker's own line and
72
+ // the TSV structure under it.
73
+ name: raw ? decodeXmlEntities(raw).replace(/[\t\n\r]+/g, ' ') : `Sheet${out.length + 1}`,
74
+ xml: table.body ?? '',
75
+ });
76
+ }
77
+ return out;
78
+ }
79
+
80
+ /**
81
+ * Expand a sheet's rows with BOUNDED repeat handling, INCREMENTALLY — a row's
82
+ * expansion lands in the capped output as it parses, so the caps bound memory
83
+ * as well as output (materializing every row's cells before consulting the
84
+ * cap let an accepted ODS allocate its whole expansion first):
85
+ *
86
+ * - all-empty rows are buffered as a COUNT (with their
87
+ * `table:number-rows-repeated` applied) and flushed only when a non-empty
88
+ * row follows, so a million-row empty tail simply disappears,
89
+ * - once the row cap is hit, the remaining rows are never parsed at all.
90
+ *
91
+ * `truncated` lists what the caps cut (mirrors the xlsx extractor's note).
92
+ */
93
+ function expandRows(tableXml: string): { rows: string[][]; truncated: string[] } {
94
+ // The shared quote-aware scanner as a WALK, not an array: materializing
95
+ // every row block before consulting the cap let a sheet of >10k explicit
96
+ // rows allocate them all first. Each row lands here as it parses, and the
97
+ // visitor's `true` stops the scan at the cap — a self-closing row WITH
98
+ // attributes is still a row, a `/>` inside a quoted attribute value is not
99
+ // a delimiter, and an UNCLOSED row costs one scan of the sheet rather than
100
+ // one per opener (see `xmlElementBlocks`).
101
+ const rows: string[][] = [];
102
+ let pendingEmpty = 0;
103
+ let rowsTruncated = false;
104
+ let colsTruncated = false;
105
+ walkLocalElementBlocks(tableXml, ['table-row'], (row) => {
106
+ const repeat = repeatCount(attrByLocalName(row.attributes, 'number-rows-repeated'));
107
+ const cells = expandCells(row.body ?? '');
108
+ if (cells.cells.length === 0) {
109
+ // Empty rows are interior padding until a non-empty row proves it —
110
+ // trailing ones are dropped with their repeats (grid padding, not data).
111
+ pendingEmpty += repeat;
112
+ return false;
113
+ }
114
+ if (cells.truncated) colsTruncated = true;
115
+ for (; pendingEmpty > 0 && rows.length < MAX_ROWS_PER_SHEET; pendingEmpty--) rows.push([]);
116
+ let i = 0;
117
+ for (; i < repeat && rows.length < MAX_ROWS_PER_SHEET; i++) rows.push(cells.cells);
118
+ if (pendingEmpty > 0 || i < repeat) {
119
+ // The cap cut real content (a sheet that merely FILLS it is not truncated).
120
+ rowsTruncated = true;
121
+ return true;
122
+ }
123
+ return false;
124
+ });
125
+ const truncated: string[] = [];
126
+ if (rowsTruncated) truncated.push(`first ${MAX_ROWS_PER_SHEET} rows`);
127
+ if (colsTruncated) truncated.push(`first ${MAX_COLS_PER_SHEET} columns`);
128
+ return { rows, truncated };
129
+ }
130
+
131
+ /**
132
+ * One row's cell texts: `<table:table-cell>` / `<table:covered-table-cell>`
133
+ * in order, expanded INCREMENTALLY like the rows above — trailing EMPTY cells
134
+ * are buffered as a count (their `table:number-columns-repeated` never
135
+ * expands) and the parse stops at the column cap.
136
+ */
137
+ function expandCells(rowXml: string): { cells: string[]; truncated: boolean } {
138
+ // Same walking scanner as the rows above — a row spelling a million
139
+ // explicit cells stops parsing at the column cap too.
140
+ const cells: string[] = [];
141
+ let pendingEmpty = 0;
142
+ let truncated = false;
143
+ walkLocalElementBlocks(rowXml, ['table-cell', 'covered-table-cell'], (cell) => {
144
+ const repeat = repeatCount(attrByLocalName(cell.attributes, 'number-columns-repeated'));
145
+ // Covered cells carry no own text anyway. Element-produced newlines/tabs
146
+ // INSIDE a cell (<text:line-break/>, <text:tab/>) become single spaces:
147
+ // the extraction's contract is one row per line with tab-separated cells,
148
+ // and a literal \n or \t inside a cell's text would silently break both.
149
+ // A COVERED cell is the hidden half of a merge: the visible cell carries
150
+ // the text. Such a cell may still hold stale content, and emitting it put
151
+ // a value in the grid where the sheet shows none.
152
+ const text = localName(cell.name) === 'covered-table-cell'
153
+ ? ''
154
+ : odfParagraphBlocks(cell.body ?? '')
155
+ .map(odfParagraphText)
156
+ .join(' ')
157
+ .replace(/[\t\n\r]+/g, ' ');
158
+ if (text === '') {
159
+ pendingEmpty += repeat;
160
+ return false;
161
+ }
162
+ for (; pendingEmpty > 0 && cells.length < MAX_COLS_PER_SHEET; pendingEmpty--) cells.push('');
163
+ let i = 0;
164
+ for (; i < repeat && cells.length < MAX_COLS_PER_SHEET; i++) cells.push(text);
165
+ if (pendingEmpty > 0 || i < repeat) {
166
+ // The cap cut real content (a row that merely FILLS it is not truncated).
167
+ truncated = true;
168
+ return true;
169
+ }
170
+ return false;
171
+ });
172
+ return { cells, truncated };
173
+ }
174
+
175
+ /** A `…-repeated="N"` attribute value, clamped to a sane positive integer. */
176
+ function repeatCount(raw: string | undefined): number {
177
+ // Decoded first: the block scanner hands attribute values RAW, and a repeat
178
+ // legally written with character references (`&#49;&#48;`) must count as
179
+ // 10, not silently fall back to 1.
180
+ const n = raw !== undefined ? parseInt(decodeXmlEntities(raw), 10) : 1;
181
+ return Number.isFinite(n) && n >= 1 ? n : 1;
182
+ }
@@ -0,0 +1,48 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ import { localBlocks, removeLocalElements } from './ooxml-text.js';
3
+ import { odfParagraphBlocks, odfParagraphText, readOdfContentXml } from './odf-text.js';
4
+
5
+ /**
6
+ * Extract the BODY text of a `.odt` (OpenDocument Text) document.
7
+ *
8
+ * An odt is a zip whose main part is `content.xml`; the body lives under
9
+ * `<office:text>`. Headers/footers are skipped like docx — in ODF they live in
10
+ * `styles.xml`, which is never opened, so reading `content.xml` alone IS the
11
+ * body-only extraction.
12
+ *
13
+ * Paragraphs (`<text:p>`) and headings (`<text:h>`, heading text as its own
14
+ * line) become lines in document order; `<text:span>` runs inside concatenate
15
+ * with NO separator, and the ODF whitespace elements (`<text:tab/>`,
16
+ * `<text:line-break/>`, `<text:s text:c="N"/>`) render as real characters —
17
+ * see `odfParagraphText`.
18
+ */
19
+ export function extractOdt(bytes: Buffer): ExtractResult {
20
+ const content = readOdfContentXml(bytes, '.odt');
21
+ if (!content.ok) return content;
22
+
23
+ // Table cells contain their own <text:p>, so the flat paragraph scan renders
24
+ // table text too (one line per cell paragraph, like the raw document order).
25
+ // The text BODY, read by the parser and matched on its LOCAL name: a
26
+ // comment mentioning `</office:text>` used to terminate the body early and
27
+ // drop every paragraph after it, and an ODT binding the office namespace to
28
+ // another prefix had no body at all.
29
+ const body = localBlocks(content.xml, 'text')[0] ?? content.xml;
30
+
31
+ // Tracked-change bookkeeping is not body text: `<text:tracked-changes>`
32
+ // stores every DELETION's content as ordinary paragraphs, so the flat scan
33
+ // below would read deleted text back in as document lines. Removed by its
34
+ // parsed element boundaries before the paragraph walk.
35
+ // `<office:annotation>` is a COMMENT on the document, stored as ordinary
36
+ // paragraphs: read flat, a reviewer's note came back as a document line.
37
+ const visible = removeLocalElements(body, ['tracked-changes', 'annotation']);
38
+
39
+ const lines = odfParagraphBlocks(visible).map(odfParagraphText);
40
+ const paragraphs = lines.length;
41
+ while (lines.length > 0 && lines[lines.length - 1].trim() === '') lines.pop();
42
+
43
+ return {
44
+ ok: true,
45
+ summary: `${paragraphs} paragraph${paragraphs === 1 ? '' : 's'}; layout, images and formatting omitted`,
46
+ text: lines.join('\n'),
47
+ };
48
+ }
@@ -0,0 +1,178 @@
1
+ import type { ExtractResult } from './doc-extract.types.js';
2
+ import { MAX_DOC_PART_BYTES } from './ooxml-text.js';
3
+
4
+ /**
5
+ * Extract a PDF's TEXT LAYER page by page with pdf.js (`pdfjs-dist`,
6
+ * Mozilla's maintained renderer — chosen over the unmaintained thin wrappers
7
+ * around it). The LEGACY build is the one supported under Node; it is loaded
8
+ * lazily (and once) because it is a heavyweight module most deployments only
9
+ * need after the first PDF read.
10
+ *
11
+ * Layout heuristic: text items on one line are joined with single spaces; a
12
+ * new line starts when pdf.js flags an EOL or the item's Y position jumps.
13
+ * A PDF with NO text layer (a scan) extracts to just the `[page N]` markers,
14
+ * and the summary says "no text layer (scanned document?)" — no OCR in v1.
15
+ */
16
+ /**
17
+ * How much DECODED text one PDF may yield before extraction gives up.
18
+ *
19
+ * The raw-size cap below bounds what arrives; it does not bound what comes
20
+ * out. PDF text lives in compressed streams, so a file comfortably under
21
+ * 50 MB can decode to far more than that, and every character of it is held
22
+ * in `lines` until the extraction returns. This bound is the decoded
23
+ * counterpart, checked as the text accumulates rather than after.
24
+ */
25
+ const MAX_PDF_TEXT_CHARS = 20 * 1024 * 1024; // 20M chars of extracted text
26
+
27
+ /**
28
+ * How many pages one PDF may have before extraction gives up.
29
+ *
30
+ * The decoded-text bound does not cover a document whose cost is its PAGE
31
+ * COUNT rather than its prose: every page costs a `getPage`, a
32
+ * `getTextContent` and a retained `[page N]` marker even when it holds no
33
+ * text at all, so a file declaring hundreds of thousands of empty pages spends
34
+ * minutes and megabytes without ever tripping a character budget. Real
35
+ * documents do not come close — a 2,000-page manual is an outlier.
36
+ */
37
+ const MAX_PDF_PAGES = 10_000;
38
+
39
+ /**
40
+ * How many text items one PAGE may hold. Items arrive through
41
+ * `streamTextContent` in small chunks (~100 items each), so this bound — like
42
+ * the character budget — fires while the page is still streaming, not after
43
+ * it has materialized. It exists because item COUNT is its own cost: each
44
+ * item is a retained heap object, and a page of empty-string items would
45
+ * never trip the character budget.
46
+ */
47
+ const MAX_PDF_ITEMS_PER_PAGE = 200_000;
48
+
49
+ /** The typed failure both decoded-text bounds return. */
50
+ function overBudget(): ExtractResult {
51
+ return {
52
+ ok: false,
53
+ message: `could not be extracted as a PDF (its text decodes to over ${MAX_PDF_TEXT_CHARS} characters — over the extraction limit)`,
54
+ };
55
+ }
56
+
57
+ export async function extractPdf(bytes: Buffer): Promise<ExtractResult> {
58
+ // The same bounded-read guard the zip-based extractors apply per part: a
59
+ // PDF has no compressed container to pre-scan, so the bound is simply the
60
+ // file's raw size, checked before pdf.js parses anything.
61
+ if (bytes.length > MAX_DOC_PART_BYTES) {
62
+ return {
63
+ ok: false,
64
+ message: `could not be extracted as a PDF (the file is ${bytes.length} bytes — over the ${MAX_DOC_PART_BYTES}-byte (50 MB) extraction limit)`,
65
+ };
66
+ }
67
+ let doc: Awaited<ReturnType<typeof openPdf>>;
68
+ try {
69
+ doc = await openPdf(bytes);
70
+ } catch (err) {
71
+ return { ok: false, message: `could not be parsed as a PDF (${(err as Error).message})` };
72
+ }
73
+ try {
74
+ if (doc.numPages > MAX_PDF_PAGES) {
75
+ return {
76
+ ok: false,
77
+ message: `could not be extracted as a PDF (it declares ${doc.numPages} pages — over the ${MAX_PDF_PAGES}-page extraction limit)`,
78
+ };
79
+ }
80
+ const lines: string[] = [];
81
+ let textChars = 0;
82
+ let anyText = false;
83
+ for (let n = 1; n <= doc.numPages; n++) {
84
+ lines.push(`[page ${n}]`);
85
+ // The marker is retained text like any other line: a document whose cost
86
+ // is its page count must reach the same bound as one whose cost is prose.
87
+ textChars += n.toString().length + 8;
88
+ if (textChars > MAX_PDF_TEXT_CHARS) return overBudget();
89
+ const page = await doc.getPage(n);
90
+ // `streamTextContent` delivers the page's items in small chunks (~100
91
+ // items each, `getTextContent` is just this stream materialized), so
92
+ // both budgets fire WHILE the page streams: a crafted single page can
93
+ // no longer build its whole item array before a bound trips. Once one
94
+ // does, the reader is cancelled and pdf.js stops producing.
95
+ const reader = (
96
+ page.streamTextContent() as ReadableStream<Awaited<ReturnType<typeof page.getTextContent>>>
97
+ ).getReader();
98
+ let pageItems = 0;
99
+ let line = '';
100
+ let lastY: number | undefined;
101
+ const flush = (): void => {
102
+ if (line.trim() !== '') {
103
+ lines.push(line);
104
+ anyText = true;
105
+ // The '\n' the final join emits for this line is retained text too.
106
+ textChars += 1;
107
+ }
108
+ line = '';
109
+ };
110
+ for (;;) {
111
+ const { done, value: chunk } = await reader.read();
112
+ if (done) break;
113
+ pageItems += chunk.items.length;
114
+ if (pageItems > MAX_PDF_ITEMS_PER_PAGE) {
115
+ await reader.cancel().catch(() => undefined);
116
+ page.cleanup();
117
+ return {
118
+ ok: false,
119
+ message: `could not be extracted as a PDF (page ${n} holds more than ${MAX_PDF_ITEMS_PER_PAGE} text items — over the extraction limit)`,
120
+ };
121
+ }
122
+ for (const item of chunk.items) {
123
+ if (!('str' in item)) continue; // marked-content item — no text
124
+ const y = item.transform?.[5];
125
+ // Y-position jump = new visual line (1pt tolerance for kerning wobble).
126
+ if (typeof y === 'number') {
127
+ if (lastY !== undefined && Math.abs(y - lastY) > 1) flush();
128
+ lastY = y;
129
+ }
130
+ if (item.str !== '') {
131
+ // COUNTED before it is kept — the join space included: the bound
132
+ // exists to stop the decoded text from accumulating, so it must
133
+ // fire mid-page and cover every character the result will hold.
134
+ textChars += item.str.length + (line === '' ? 0 : 1);
135
+ if (textChars > MAX_PDF_TEXT_CHARS) {
136
+ await reader.cancel().catch(() => undefined);
137
+ return overBudget();
138
+ }
139
+ line += (line === '' ? '' : ' ') + item.str;
140
+ }
141
+ if (item.hasEOL) flush();
142
+ }
143
+ }
144
+ flush();
145
+ page.cleanup();
146
+ }
147
+ const pages = `${doc.numPages} page${doc.numPages === 1 ? '' : 's'}`;
148
+ return anyText
149
+ ? { ok: true, summary: `${pages}; layout, images and formatting omitted`, text: lines.join('\n') }
150
+ : { ok: true, summary: `${pages}; no text layer (scanned document?)`, text: lines.join('\n') };
151
+ } catch (err) {
152
+ return { ok: false, message: `could not extract the PDF's text (${(err as Error).message})` };
153
+ } finally {
154
+ await doc.destroy();
155
+ }
156
+ }
157
+
158
+ type PdfJs = typeof import('pdfjs-dist/legacy/build/pdf.mjs');
159
+ let pdfjsPromise: Promise<PdfJs> | undefined;
160
+
161
+ async function openPdf(bytes: Buffer) {
162
+ pdfjsPromise ??= import('pdfjs-dist/legacy/build/pdf.mjs').catch((err: unknown) => {
163
+ // A FAILED load must not be memoized: left in place, the rejected promise
164
+ // would answer every later read and disable PDF extraction for the whole
165
+ // process. Reset so the next read retries the import.
166
+ pdfjsPromise = undefined;
167
+ throw err;
168
+ });
169
+ const { getDocument } = await pdfjsPromise;
170
+ return getDocument({
171
+ // Copy into a fresh Uint8Array: pdf.js TRANSFERS the buffer it is given
172
+ // (detaching it), and the caller's Buffer must stay usable for hashing.
173
+ data: new Uint8Array(bytes),
174
+ // Server side: no font rendering — text content is all we consume.
175
+ disableFontFace: true,
176
+ useSystemFonts: true,
177
+ }).promise;
178
+ }