@bevel-software/platform-core-backend 0.11.1 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THIRD-PARTY-NOTICES.md +1163 -425
- package/dist/core/create-core-server.js +1 -1
- package/dist/core/create-core-server.js.map +1 -1
- package/dist/core/create-core-services.d.ts +3 -1
- package/dist/core/create-core-services.d.ts.map +1 -1
- package/dist/core/create-core-services.js +7 -2
- package/dist/core/create-core-services.js.map +1 -1
- package/dist/core-config.d.ts +7 -0
- package/dist/core-config.d.ts.map +1 -1
- package/dist/core-config.js +9 -0
- package/dist/core-config.js.map +1 -1
- package/dist/modules/access/access-control.service.d.ts +1 -253
- package/dist/modules/access/access-control.service.d.ts.map +1 -1
- package/dist/modules/access/access-control.service.js +3 -632
- package/dist/modules/access/access-control.service.js.map +1 -1
- package/dist/modules/access/access-declarations.d.ts +2 -2
- package/dist/modules/access/access-declarations.d.ts.map +1 -1
- package/dist/modules/access/access-declarations.js +2 -2
- package/dist/modules/access/access-declarations.js.map +1 -1
- package/dist/modules/access/access-mutation.service.d.ts +3 -3
- package/dist/modules/access/access-mutation.service.d.ts.map +1 -1
- package/dist/modules/access/access-mutation.service.js +3 -3
- package/dist/modules/access/access-mutation.service.js.map +1 -1
- package/dist/modules/access/access.routes.js +4 -4
- package/dist/modules/access/access.routes.js.map +1 -1
- package/dist/modules/access/admin-locked-commit.d.ts +2 -2
- package/dist/modules/access/admin-locked-commit.d.ts.map +1 -1
- package/dist/modules/access/admin-locked-commit.js +2 -2
- package/dist/modules/access/admin-locked-commit.js.map +1 -1
- package/dist/modules/access/admin-route-helpers.js +1 -1
- package/dist/modules/access/admin-route-helpers.js.map +1 -1
- package/dist/modules/access/capability-registry.js +1 -1
- package/dist/modules/access/capability-registry.js.map +1 -1
- package/dist/modules/access/creator-access.d.ts +1 -45
- package/dist/modules/access/creator-access.d.ts.map +1 -1
- package/dist/modules/access/creator-access.js +4 -24
- package/dist/modules/access/creator-access.js.map +1 -1
- package/dist/modules/access/groups-admin.routes.js +1 -1
- package/dist/modules/access/groups-admin.routes.js.map +1 -1
- package/dist/modules/access/groups-admin.service.d.ts +1 -1
- package/dist/modules/access/groups-admin.service.d.ts.map +1 -1
- package/dist/modules/access/groups-admin.service.js +6 -5
- package/dist/modules/access/groups-admin.service.js.map +1 -1
- package/dist/modules/access/groups-edit.js +2 -2
- package/dist/modules/access/groups-edit.js.map +1 -1
- package/dist/modules/access/reference-scan.js +1 -1
- package/dist/modules/access/reference-scan.js.map +1 -1
- package/dist/modules/access/roles-admin.service.d.ts +1 -1
- package/dist/modules/access/roles-admin.service.d.ts.map +1 -1
- package/dist/modules/access/roles-admin.service.js +8 -7
- package/dist/modules/access/roles-admin.service.js.map +1 -1
- package/dist/modules/access/roles-edit.js +1 -1
- package/dist/modules/access/roles-edit.js.map +1 -1
- package/dist/modules/access/synced-groups-committer.js +4 -4
- package/dist/modules/access/synced-groups-committer.js.map +1 -1
- package/dist/modules/access/synced-groups-writer.js +2 -2
- package/dist/modules/access/synced-groups-writer.js.map +1 -1
- package/dist/modules/access-model/access-errors.d.ts +34 -0
- package/dist/modules/access-model/access-errors.d.ts.map +1 -0
- package/dist/modules/access-model/access-errors.js +40 -0
- package/dist/modules/access-model/access-errors.js.map +1 -0
- package/dist/modules/access-model/access-grammar.d.ts +266 -0
- package/dist/modules/access-model/access-grammar.d.ts.map +1 -0
- package/dist/modules/access-model/access-grammar.js +644 -0
- package/dist/modules/access-model/access-grammar.js.map +1 -0
- package/dist/modules/access-model/access-splice.d.ts +140 -0
- package/dist/modules/access-model/access-splice.d.ts.map +1 -0
- package/dist/modules/access-model/access-splice.js +389 -0
- package/dist/modules/access-model/access-splice.js.map +1 -0
- package/dist/modules/access-model/creator.d.ts +54 -0
- package/dist/modules/access-model/creator.d.ts.map +1 -0
- package/dist/modules/access-model/creator.js +30 -0
- package/dist/modules/access-model/creator.js.map +1 -0
- package/dist/modules/access-model/group-files.d.ts +83 -0
- package/dist/modules/access-model/group-files.d.ts.map +1 -0
- package/dist/modules/access-model/group-files.js +167 -0
- package/dist/modules/access-model/group-files.js.map +1 -0
- package/dist/modules/access-model/kb-read-filter.d.ts +41 -0
- package/dist/modules/access-model/kb-read-filter.d.ts.map +1 -0
- package/dist/modules/access-model/kb-read-filter.js +60 -0
- package/dist/modules/access-model/kb-read-filter.js.map +1 -0
- package/dist/modules/access-model/render-roles-yaml.d.ts +22 -0
- package/dist/modules/access-model/render-roles-yaml.d.ts.map +1 -0
- package/dist/modules/access-model/render-roles-yaml.js +56 -0
- package/dist/modules/access-model/render-roles-yaml.js.map +1 -0
- package/dist/modules/access-model/roles-yaml-guard.d.ts +56 -0
- package/dist/modules/access-model/roles-yaml-guard.d.ts.map +1 -0
- package/dist/modules/access-model/roles-yaml-guard.js +79 -0
- package/dist/modules/access-model/roles-yaml-guard.js.map +1 -0
- package/dist/modules/admin/admin-access.service.d.ts.map +1 -1
- package/dist/modules/admin/admin-access.service.js +1 -1
- package/dist/modules/admin/admin-access.service.js.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.d.ts.map +1 -1
- package/dist/modules/code-mode/code-mode.tool.js +7 -1
- package/dist/modules/code-mode/code-mode.tool.js.map +1 -1
- package/dist/modules/diff/diff.routes.js +4 -4
- package/dist/modules/diff/diff.routes.js.map +1 -1
- package/dist/modules/diff/diff.service.d.ts +1 -1
- package/dist/modules/diff/diff.service.d.ts.map +1 -1
- package/dist/modules/kb-fs/branch-name.d.ts +10 -0
- package/dist/modules/kb-fs/branch-name.d.ts.map +1 -0
- package/dist/modules/kb-fs/branch-name.js +76 -0
- package/dist/modules/kb-fs/branch-name.js.map +1 -0
- package/dist/modules/kb-fs/clone-config.d.ts +61 -0
- package/dist/modules/kb-fs/clone-config.d.ts.map +1 -0
- package/dist/modules/kb-fs/clone-config.js +69 -0
- package/dist/modules/kb-fs/clone-config.js.map +1 -0
- package/dist/modules/kb-fs/file-change-notifier.d.ts +38 -0
- package/dist/modules/kb-fs/file-change-notifier.d.ts.map +1 -0
- package/dist/modules/kb-fs/file-change-notifier.js +22 -0
- package/dist/modules/kb-fs/file-change-notifier.js.map +1 -0
- package/dist/modules/kb-fs/locking-filesystem.d.ts +137 -0
- package/dist/modules/kb-fs/locking-filesystem.d.ts.map +1 -0
- package/dist/modules/kb-fs/locking-filesystem.js +553 -0
- package/dist/modules/kb-fs/locking-filesystem.js.map +1 -0
- package/dist/modules/kb-fs/mutex.d.ts +11 -0
- package/dist/modules/kb-fs/mutex.d.ts.map +1 -0
- package/dist/modules/kb-fs/mutex.js +23 -0
- package/dist/modules/kb-fs/mutex.js.map +1 -0
- package/dist/modules/kb-fs/read-only-filesystem.d.ts +24 -0
- package/dist/modules/kb-fs/read-only-filesystem.d.ts.map +1 -0
- package/dist/modules/kb-fs/read-only-filesystem.js +39 -0
- package/dist/modules/kb-fs/read-only-filesystem.js.map +1 -0
- package/dist/modules/plugins/join-proposals.d.ts +1 -1
- package/dist/modules/plugins/join-proposals.d.ts.map +1 -1
- package/dist/modules/plugins/join-proposals.js +1 -1
- package/dist/modules/plugins/join-proposals.js.map +1 -1
- package/dist/modules/plugins/join-requests.service.d.ts.map +1 -1
- package/dist/modules/plugins/join-requests.service.js +1 -1
- package/dist/modules/plugins/join-requests.service.js.map +1 -1
- package/dist/modules/plugins/plugin-provision.service.js +3 -3
- package/dist/modules/plugins/plugin-provision.service.js.map +1 -1
- package/dist/modules/plugins/plugins.routes.js +2 -2
- package/dist/modules/plugins/plugins.routes.js.map +1 -1
- package/dist/modules/plugins/plugins.service.js +1 -1
- package/dist/modules/plugins/plugins.service.js.map +1 -1
- package/dist/modules/secrets-vault/secrets-vault.routes.js +1 -1
- package/dist/modules/secrets-vault/secrets-vault.routes.js.map +1 -1
- package/dist/modules/skills/pending-skills.service.d.ts.map +1 -1
- package/dist/modules/skills/pending-skills.service.js +1 -1
- package/dist/modules/skills/pending-skills.service.js.map +1 -1
- package/dist/modules/skills/skills.service.js +1 -1
- package/dist/modules/skills/skills.service.js.map +1 -1
- package/dist/modules/tool-helpers/tool-context.d.ts +2 -2
- package/dist/modules/tool-helpers/tool-context.d.ts.map +1 -1
- package/dist/modules/tool-helpers/tool-context.js +4 -4
- package/dist/modules/tool-helpers/tool-context.js.map +1 -1
- package/dist/modules/tool-manuals/mcp-server-edit.service.d.ts +1 -1
- package/dist/modules/tool-manuals/mcp-server-edit.service.d.ts.map +1 -1
- package/dist/modules/tool-manuals/mcp-server-edit.service.js +1 -1
- package/dist/modules/tool-manuals/mcp-server-edit.service.js.map +1 -1
- package/dist/modules/tool-manuals/tool-manuals.routes.d.ts +1 -1
- package/dist/modules/tool-manuals/tool-manuals.routes.d.ts.map +1 -1
- package/dist/modules/tool-manuals/tool-manuals.routes.js +1 -1
- package/dist/modules/tool-manuals/tool-manuals.routes.js.map +1 -1
- package/dist/modules/tool-manuals/tool-manuals.service.js +1 -1
- package/dist/modules/tool-manuals/tool-manuals.service.js.map +1 -1
- package/dist/modules/tool-manuals/tool-manuals.tools.js +1 -1
- package/dist/modules/tool-manuals/tool-manuals.tools.js.map +1 -1
- package/dist/modules/workflow/agent-tools/workflow.tools.js +1 -1
- package/dist/modules/workflow/agent-tools/workflow.tools.js.map +1 -1
- package/dist/modules/workflow/file-lock.service.js +1 -1
- package/dist/modules/workflow/file-lock.service.js.map +1 -1
- package/dist/modules/workflow/git/git.service.d.ts +2 -2
- package/dist/modules/workflow/git/git.service.d.ts.map +1 -1
- package/dist/modules/workflow/git/git.service.js +5 -5
- package/dist/modules/workflow/git/git.service.js.map +1 -1
- package/dist/modules/workflow/git/pull-request.service.js +1 -1
- package/dist/modules/workflow/git/pull-request.service.js.map +1 -1
- package/dist/modules/workflow/review-workflow/review-workflow.service.js +2 -2
- package/dist/modules/workflow/review-workflow/review-workflow.service.js.map +1 -1
- package/dist/modules/workflow/session-ontology.service.js +1 -1
- package/dist/modules/workflow/session-ontology.service.js.map +1 -1
- package/dist/modules/workflow/workflow.routes.js +2 -2
- package/dist/modules/workflow/workflow.routes.js.map +1 -1
- package/dist/modules/workflow/workflow.service.d.ts +1 -1
- package/dist/modules/workflow/workflow.service.d.ts.map +1 -1
- package/dist/modules/workflow/workflow.service.js +4 -4
- package/dist/modules/workflow/workflow.service.js.map +1 -1
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts +71 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js +90 -0
- package/dist/modules/workspace/file-readers/doc-extract.service.js.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts +55 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js +34 -0
- package/dist/modules/workspace/file-readers/doc-extract.types.js.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts +32 -0
- package/dist/modules/workspace/file-readers/document-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/document-reader.js +59 -0
- package/dist/modules/workspace/file-readers/document-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts +15 -0
- package/dist/modules/workspace/file-readers/email-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-reader.js +19 -0
- package/dist/modules/workspace/file-readers/email-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts +51 -0
- package/dist/modules/workspace/file-readers/email-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/email-text.js +151 -0
- package/dist/modules/workspace/file-readers/email-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-docx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-docx.js +67 -0
- package/dist/modules/workspace/file-readers/extract-docx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts +18 -0
- package/dist/modules/workspace/file-readers/extract-eml.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-eml.js +87 -0
- package/dist/modules/workspace/file-readers/extract-eml.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-msg.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-msg.js +121 -0
- package/dist/modules/workspace/file-readers/extract-msg.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts +13 -0
- package/dist/modules/workspace/file-readers/extract-odp.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odp.js +60 -0
- package/dist/modules/workspace/file-readers/extract-odp.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-ods.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-ods.js +173 -0
- package/dist/modules/workspace/file-readers/extract-ods.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts +17 -0
- package/dist/modules/workspace/file-readers/extract-odt.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-odt.js +45 -0
- package/dist/modules/workspace/file-readers/extract-odt.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts +3 -0
- package/dist/modules/workspace/file-readers/extract-pdf.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js +176 -0
- package/dist/modules/workspace/file-readers/extract-pdf.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts +37 -0
- package/dist/modules/workspace/file-readers/extract-pptx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js +288 -0
- package/dist/modules/workspace/file-readers/extract-pptx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts +10 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js +98 -0
- package/dist/modules/workspace/file-readers/extract-xlsx.js.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts +61 -0
- package/dist/modules/workspace/file-readers/extraction-cache.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js +135 -0
- package/dist/modules/workspace/file-readers/extraction-cache.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts +76 -0
- package/dist/modules/workspace/file-readers/file-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.js +55 -0
- package/dist/modules/workspace/file-readers/file-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts +13 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js +41 -0
- package/dist/modules/workspace/file-readers/file-reader.registry.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts +35 -0
- package/dist/modules/workspace/file-readers/image-read.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-read.js +108 -0
- package/dist/modules/workspace/file-readers/image-read.js.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts +19 -0
- package/dist/modules/workspace/file-readers/image-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/image-reader.js +30 -0
- package/dist/modules/workspace/file-readers/image-reader.js.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts +26 -0
- package/dist/modules/workspace/file-readers/odf-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/odf-text.js +116 -0
- package/dist/modules/workspace/file-readers/odf-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts +172 -0
- package/dist/modules/workspace/file-readers/ooxml-text.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js +439 -0
- package/dist/modules/workspace/file-readers/ooxml-text.js.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts +47 -0
- package/dist/modules/workspace/file-readers/text-reader.d.ts.map +1 -0
- package/dist/modules/workspace/file-readers/text-reader.js +117 -0
- package/dist/modules/workspace/file-readers/text-reader.js.map +1 -0
- package/dist/modules/workspace/startup/kb-startup-runner.js +1 -1
- package/dist/modules/workspace/startup/kb-startup-runner.js.map +1 -1
- package/dist/modules/workspace/startup/steps/groups-to-plugins.step.js +1 -1
- package/dist/modules/workspace/startup/steps/groups-to-plugins.step.js.map +1 -1
- package/dist/modules/workspace/startup/steps/roles-yaml.step.d.ts +1 -1
- package/dist/modules/workspace/startup/steps/roles-yaml.step.js +2 -2
- package/dist/modules/workspace/startup/steps/roles-yaml.step.js.map +1 -1
- package/dist/modules/workspace/startup/steps/seed-tree.js +1 -1
- package/dist/modules/workspace/startup/steps/seed-tree.js.map +1 -1
- package/dist/modules/workspace/workspace.routes.d.ts +1 -1
- package/dist/modules/workspace/workspace.routes.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.routes.js +5 -4
- package/dist/modules/workspace/workspace.routes.js.map +1 -1
- package/dist/modules/workspace/workspace.service.d.ts +2 -9
- package/dist/modules/workspace/workspace.service.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.service.js +14 -15
- package/dist/modules/workspace/workspace.service.js.map +1 -1
- package/dist/modules/workspace/workspace.tools.d.ts +2 -1
- package/dist/modules/workspace/workspace.tools.d.ts.map +1 -1
- package/dist/modules/workspace/workspace.tools.js +161 -18
- package/dist/modules/workspace/workspace.tools.js.map +1 -1
- package/dist/shared/domain-errors.d.ts +202 -0
- package/dist/shared/domain-errors.d.ts.map +1 -0
- package/dist/shared/domain-errors.js +303 -0
- package/dist/shared/domain-errors.js.map +1 -0
- package/dist/shared/workspace-id.d.ts +27 -0
- package/dist/shared/workspace-id.d.ts.map +1 -0
- package/dist/shared/workspace-id.js +36 -0
- package/dist/shared/workspace-id.js.map +1 -0
- package/package.json +9 -4
- package/src/core/create-core-server.ts +1 -1
- package/src/core/create-core-services.ts +8 -2
- package/src/core-config.ts +9 -0
- package/src/modules/access/__tests__/access-control.service.test.ts +1 -1
- package/src/modules/access/__tests__/access-groups.test.ts +6 -5
- package/src/modules/access/__tests__/access-md-format.test.ts +3 -3
- package/src/modules/access/__tests__/access-mutation.service.test.ts +1 -1
- package/src/modules/access/__tests__/admin-locked-commit.test.ts +1 -1
- package/src/modules/access/__tests__/admin-route-helpers.test.ts +1 -1
- package/src/modules/access/__tests__/roles-admin.service.test.ts +2 -2
- package/src/modules/access/__tests__/roles-edit.test.ts +1 -1
- package/src/modules/access/__tests__/synced-groups-committer.test.ts +1 -1
- package/src/modules/access/__tests__/synced-groups-writer.test.ts +1 -1
- package/src/modules/access/access-control.service.ts +23 -768
- package/src/modules/access/access-declarations.ts +2 -2
- package/src/modules/access/access-mutation.service.ts +3 -3
- package/src/modules/access/access.routes.ts +5 -5
- package/src/modules/access/admin-locked-commit.ts +2 -2
- package/src/modules/access/admin-route-helpers.ts +1 -1
- package/src/modules/access/capability-registry.ts +1 -1
- package/src/modules/access/creator-access.ts +9 -72
- package/src/modules/access/groups-admin.routes.ts +1 -1
- package/src/modules/access/groups-admin.service.ts +6 -7
- package/src/modules/access/groups-edit.ts +2 -2
- package/src/modules/access/reference-scan.ts +1 -1
- package/src/modules/access/roles-admin.service.ts +8 -8
- package/src/modules/access/roles-edit.ts +1 -1
- package/src/modules/access/synced-groups-committer.ts +4 -4
- package/src/modules/access/synced-groups-writer.ts +2 -2
- package/src/modules/access-model/__tests__/access-grammar.test.ts +31 -0
- package/src/modules/{access → access-model}/__tests__/access-splice.test.ts +268 -268
- package/src/modules/{access → access-model}/access-errors.ts +1 -1
- package/src/modules/access-model/access-grammar.ts +778 -0
- package/src/modules/{access → access-model}/access-splice.ts +1 -1
- package/src/modules/access-model/creator.ts +79 -0
- package/src/modules/{access → access-model}/group-files.ts +1 -1
- package/src/modules/{access → access-model}/render-roles-yaml.ts +1 -1
- package/src/modules/{access → access-model}/roles-yaml-guard.ts +2 -2
- package/src/modules/admin/admin-access.service.ts +2 -1
- package/src/modules/code-mode/__tests__/code-mode.tool.test.ts +30 -0
- package/src/modules/code-mode/code-mode.tool.ts +7 -1
- package/src/modules/diff/__tests__/diff.routes.rejectPathsLocked.test.ts +2 -2
- package/src/modules/diff/__tests__/diff.service.seed-atomicity.test.ts +3 -2
- package/src/modules/diff/__tests__/diff.service.test.ts +3 -2
- package/src/modules/diff/diff.routes.ts +4 -4
- package/src/modules/diff/diff.service.ts +1 -1
- package/src/modules/{workflow/git → kb-fs}/__tests__/branch-name.test.ts +1 -1
- package/src/modules/{workflow → kb-fs}/__tests__/locking-filesystem.test.ts +1 -1
- package/src/modules/{workflow/git → kb-fs}/branch-name.ts +1 -1
- package/src/modules/{workflow → kb-fs}/locking-filesystem.ts +2 -2
- package/src/modules/plugins/__tests__/join-requests.service.test.ts +2 -4
- package/src/modules/plugins/__tests__/plugin-index.service.test.ts +1 -1
- package/src/modules/plugins/__tests__/plugins.routes.test.ts +1 -1
- package/src/modules/plugins/join-proposals.ts +1 -1
- package/src/modules/plugins/join-requests.service.ts +2 -1
- package/src/modules/plugins/plugin-provision.service.ts +3 -3
- package/src/modules/plugins/plugins.routes.ts +2 -2
- package/src/modules/plugins/plugins.service.ts +1 -1
- package/src/modules/secrets-vault/secrets-vault.routes.ts +1 -1
- package/src/modules/skills/__tests__/skills.service.test.ts +1 -1
- package/src/modules/skills/pending-skills.service.ts +2 -1
- package/src/modules/skills/skills.service.ts +1 -1
- package/src/modules/tool-helpers/__tests__/phase4-tools.test.ts +2 -1
- package/src/modules/tool-helpers/tool-context.ts +6 -5
- package/src/modules/tool-manuals/__tests__/mcp-server-edit.service.test.ts +1 -1
- package/src/modules/tool-manuals/__tests__/tool-manuals.archive.route.test.ts +1 -1
- package/src/modules/tool-manuals/__tests__/tool-manuals.detail.route.test.ts +1 -1
- package/src/modules/tool-manuals/__tests__/tool-manuals.mcp-oauth.test.ts +1 -1
- package/src/modules/tool-manuals/__tests__/tool-manuals.service.test.ts +1 -1
- package/src/modules/tool-manuals/mcp-server-edit.service.ts +2 -1
- package/src/modules/tool-manuals/tool-manuals.routes.ts +2 -1
- package/src/modules/tool-manuals/tool-manuals.service.ts +1 -1
- package/src/modules/tool-manuals/tool-manuals.tools.ts +1 -1
- package/src/modules/workflow/__tests__/preserve-roles-yaml.test.ts +1 -1
- package/src/modules/workflow/__tests__/workflow.service.commitFileWhileLocked.test.ts +1 -1
- package/src/modules/workflow/__tests__/workflow.service.facade.test.ts +1 -1
- package/src/modules/workflow/__tests__/workflow.service.releaseLock.test.ts +1 -1
- package/src/modules/workflow/agent-tools/workflow.tools.ts +1 -1
- package/src/modules/workflow/file-lock.service.ts +1 -1
- package/src/modules/workflow/git/__tests__/git.service.accessGating.test.ts +1 -1
- package/src/modules/workflow/git/__tests__/git.service.deleteBranch.test.ts +1 -1
- package/src/modules/workflow/git/__tests__/git.service.diffFileAtCommit.test.ts +1 -1
- package/src/modules/workflow/git/__tests__/git.service.diffFileBetweenBranches.test.ts +1 -1
- package/src/modules/workflow/git/__tests__/git.service.pull.test.ts +1 -1
- package/src/modules/workflow/git/git.service.ts +5 -5
- package/src/modules/workflow/git/pull-request.service.ts +1 -1
- package/src/modules/workflow/review-workflow/__tests__/cancel-pr.test.ts +1 -1
- package/src/modules/workflow/review-workflow/review-workflow.service.ts +2 -2
- package/src/modules/workflow/session-ontology.service.ts +1 -1
- package/src/modules/workflow/workflow.routes.ts +2 -2
- package/src/modules/workflow/workflow.service.ts +5 -5
- package/src/modules/workspace/__tests__/workspace.routes.create-grant.test.ts +1 -1
- package/src/modules/workspace/__tests__/workspace.routes.delete.test.ts +1 -1
- package/src/modules/workspace/__tests__/workspace.routes.download.test.ts +2 -2
- package/src/modules/workspace/__tests__/workspace.routes.read-gate.test.ts +1 -1
- package/src/modules/workspace/__tests__/workspace.service.read-filter.test.ts +2 -5
- package/src/modules/workspace/__tests__/workspace.service.test.ts +13 -1
- package/src/modules/workspace/__tests__/workspace.tools.test.ts +501 -3
- package/src/modules/workspace/file-readers/__tests__/doc-extract.test.ts +1658 -0
- package/src/modules/workspace/file-readers/__tests__/email-extract.test.ts +485 -0
- package/src/modules/workspace/file-readers/__tests__/file-reader.registry.test.ts +97 -0
- package/src/modules/workspace/file-readers/__tests__/image-read.test.ts +100 -0
- package/src/modules/workspace/file-readers/doc-extract.service.ts +104 -0
- package/src/modules/workspace/file-readers/doc-extract.types.ts +63 -0
- package/src/modules/workspace/file-readers/document-reader.ts +64 -0
- package/src/modules/workspace/file-readers/email-reader.ts +21 -0
- package/src/modules/workspace/file-readers/email-text.ts +193 -0
- package/src/modules/workspace/file-readers/extract-docx.ts +67 -0
- package/src/modules/workspace/file-readers/extract-eml.ts +92 -0
- package/src/modules/workspace/file-readers/extract-msg.ts +134 -0
- package/src/modules/workspace/file-readers/extract-odp.ts +63 -0
- package/src/modules/workspace/file-readers/extract-ods.ts +182 -0
- package/src/modules/workspace/file-readers/extract-odt.ts +48 -0
- package/src/modules/workspace/file-readers/extract-pdf.ts +178 -0
- package/src/modules/workspace/file-readers/extract-pptx.ts +302 -0
- package/src/modules/workspace/file-readers/extract-xlsx.ts +96 -0
- package/src/modules/workspace/file-readers/extraction-cache.ts +142 -0
- package/src/modules/workspace/file-readers/file-reader.registry.ts +45 -0
- package/src/modules/workspace/file-readers/file-reader.ts +104 -0
- package/src/modules/workspace/file-readers/image-read.ts +122 -0
- package/src/modules/workspace/file-readers/image-reader.ts +39 -0
- package/src/modules/workspace/file-readers/odf-text.ts +123 -0
- package/src/modules/workspace/file-readers/ooxml-text.ts +477 -0
- package/src/modules/workspace/file-readers/text-reader.ts +131 -0
- package/src/modules/workspace/startup/kb-startup-runner.ts +1 -1
- package/src/modules/workspace/startup/steps/__tests__/steps.test.ts +1 -1
- package/src/modules/workspace/startup/steps/groups-to-plugins.step.ts +1 -1
- package/src/modules/workspace/startup/steps/roles-yaml.step.ts +2 -2
- package/src/modules/workspace/startup/steps/seed-tree.ts +1 -1
- package/src/modules/workspace/workspace.routes.ts +6 -5
- package/src/modules/workspace/workspace.service.ts +14 -18
- package/src/modules/workspace/workspace.tools.ts +177 -15
- package/src/shared/__tests__/join-request.test.ts +1 -1
- package/src/shared/__tests__/workspace-id.test.ts +17 -0
- package/src/{modules/workflow/workflow.errors.ts → shared/domain-errors.ts} +6 -1
- package/src/shared/workspace-id.ts +36 -0
- /package/src/modules/{access → access-model}/__tests__/kb-read-filter.test.ts +0 -0
- /package/src/modules/{access → access-model}/__tests__/roles-yaml-guard.test.ts +0 -0
- /package/src/modules/{access → access-model}/kb-read-filter.ts +0 -0
- /package/src/modules/{workflow/git → kb-fs}/__tests__/clone-config.test.ts +0 -0
- /package/src/modules/{workflow → kb-fs}/__tests__/file-change-notifier.test.ts +0 -0
- /package/src/modules/{workflow/git → kb-fs}/__tests__/mutex.test.ts +0 -0
- /package/src/modules/{workflow/git → kb-fs}/clone-config.ts +0 -0
- /package/src/modules/{workflow → kb-fs}/file-change-notifier.ts +0 -0
- /package/src/modules/{workflow/git → kb-fs}/mutex.ts +0 -0
- /package/src/modules/{workflow → kb-fs}/read-only-filesystem.ts +0 -0
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
import AdmZip from 'adm-zip';
|
|
2
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
3
|
+
import {
|
|
4
|
+
MAX_DOC_TOTAL_BYTES,
|
|
5
|
+
attrByLocalName,
|
|
6
|
+
decodeXmlEntities,
|
|
7
|
+
localBlocks,
|
|
8
|
+
localElementBlocks,
|
|
9
|
+
localName,
|
|
10
|
+
paragraphRunText,
|
|
11
|
+
zipEntryOversize,
|
|
12
|
+
} from './ooxml-text.js';
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Extract the text of a `.pptx` (PowerPoint) deck.
|
|
16
|
+
*
|
|
17
|
+
* Slides live at `ppt/slides/slideN.xml`; each is emitted under a `[slide N]`
|
|
18
|
+
* marker line, in the PRESENTATION's slide order — `ppt/presentation.xml`'s
|
|
19
|
+
* `<p:sldIdLst>`, resolved through its rels part (see
|
|
20
|
+
* `slideOrderFromPresentation`) — with numeric filename order as the fallback
|
|
21
|
+
* when the package has no readable list. Speaker notes
|
|
22
|
+
* follow their slide under `[slide N notes]` when non-empty. A slide's notes
|
|
23
|
+
* part is resolved through the slide's RELATIONSHIPS part (the `_rels` twin of
|
|
24
|
+
* the slide part's own NAME, relationship type ending `notesSlide`) — the
|
|
25
|
+
* package is free to number notes parts differently from slides — with the
|
|
26
|
+
* `notesSlideN.xml` convention as the fallback when the slide has no rels
|
|
27
|
+
* part at all. Within a slide, each `<a:p>` paragraph is a line;
|
|
28
|
+
* `<a:t>` runs concatenate with no separator (runs split mid-word).
|
|
29
|
+
*
|
|
30
|
+
* Bounded: every entry's DECLARED uncompressed size is checked before
|
|
31
|
+
* inflation (see `zipEntryOversize`), and the parts read for one deck may not
|
|
32
|
+
* exceed `MAX_DOC_TOTAL_BYTES` in total — over either bound is a typed
|
|
33
|
+
* failure, never an allocation.
|
|
34
|
+
*/
|
|
35
|
+
export function extractPptx(bytes: Buffer): ExtractResult {
|
|
36
|
+
let slides: Map<number, SlidePart>;
|
|
37
|
+
let notesBySlide: Map<number, string>;
|
|
38
|
+
let presOrder: string[] | undefined;
|
|
39
|
+
try {
|
|
40
|
+
const zip = new AdmZip(bytes);
|
|
41
|
+
const budget = { remaining: MAX_DOC_TOTAL_BYTES };
|
|
42
|
+
slides = collectNumbered(zip, /^ppt\/slides\/slide(\d+)\.xml$/, budget);
|
|
43
|
+
notesBySlide = collectNotes(zip, slides, budget);
|
|
44
|
+
presOrder = slideOrderFromPresentation(zip, budget);
|
|
45
|
+
} catch (err) {
|
|
46
|
+
return { ok: false, message: `could not be parsed as a .pptx (${(err as Error).message})` };
|
|
47
|
+
}
|
|
48
|
+
if (slides.size === 0) {
|
|
49
|
+
return { ok: false, message: 'could not be parsed as a .pptx (no ppt/slides/slideN.xml inside the archive)' };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// Emission order: the presentation's own slide list when it resolves to
|
|
53
|
+
// selected parts, numeric filename order otherwise (parts the list does not
|
|
54
|
+
// name follow it, in filename order). When the LIST orders the deck, the
|
|
55
|
+
// markers number POSITIONS in it — what a viewer calls slide 1 — because a
|
|
56
|
+
// reordered deck's part filenames no longer mean anything positional.
|
|
57
|
+
const byFilename = [...slides.keys()].sort((a, b) => a - b);
|
|
58
|
+
let order = byFilename;
|
|
59
|
+
let positional = false;
|
|
60
|
+
if (presOrder !== undefined) {
|
|
61
|
+
const numByName = new Map<string, number>();
|
|
62
|
+
for (const [n, slide] of slides) numByName.set(slide.name, n);
|
|
63
|
+
const seen = new Set<number>();
|
|
64
|
+
const fromList: number[] = [];
|
|
65
|
+
for (const name of presOrder) {
|
|
66
|
+
const n = numByName.get(name);
|
|
67
|
+
if (n !== undefined && !seen.has(n)) {
|
|
68
|
+
seen.add(n);
|
|
69
|
+
fromList.push(n);
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
if (fromList.length > 0) {
|
|
73
|
+
order = [...fromList, ...byFilename.filter((n) => !seen.has(n))];
|
|
74
|
+
positional = true;
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
const lines: string[] = [];
|
|
79
|
+
let anyNotes = false;
|
|
80
|
+
order.forEach((n, i) => {
|
|
81
|
+
const label = positional ? i + 1 : n;
|
|
82
|
+
lines.push(`[slide ${label}]`);
|
|
83
|
+
lines.push(...paragraphLines(slides.get(n)!.xml));
|
|
84
|
+
const notesXml = notesBySlide.get(n);
|
|
85
|
+
const noteLines = notesXml !== undefined ? paragraphLines(notesXml) : [];
|
|
86
|
+
if (noteLines.length > 0) {
|
|
87
|
+
anyNotes = true;
|
|
88
|
+
lines.push(`[slide ${label} notes]`);
|
|
89
|
+
lines.push(...noteLines);
|
|
90
|
+
}
|
|
91
|
+
});
|
|
92
|
+
return {
|
|
93
|
+
ok: true,
|
|
94
|
+
summary: `${slides.size} slide${slides.size === 1 ? '' : 's'}${anyNotes ? ' + notes' : ''}; layout, images and formatting omitted`,
|
|
95
|
+
text: lines.join('\n'),
|
|
96
|
+
};
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Non-empty paragraph texts of one slide/notes part, in document order. */
|
|
100
|
+
function paragraphLines(xml: string): string[] {
|
|
101
|
+
const out: string[] = [];
|
|
102
|
+
for (const p of localBlocks(xml, 'p')) {
|
|
103
|
+
const text = paragraphRunText(p, 't');
|
|
104
|
+
if (text.trim() !== '') out.push(text);
|
|
105
|
+
}
|
|
106
|
+
return out;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** Running total of uncompressed bytes one extraction may still read. */
|
|
110
|
+
interface ReadBudget {
|
|
111
|
+
remaining: number;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
/** `entry`'s bytes as UTF-8, after the per-part and aggregate bounds. Throws over either. */
|
|
115
|
+
function readEntryBounded(entry: AdmZip.IZipEntry, budget: ReadBudget): string {
|
|
116
|
+
const oversize = zipEntryOversize(entry);
|
|
117
|
+
if (oversize) throw new Error(oversize);
|
|
118
|
+
budget.remaining -= entry.header.size;
|
|
119
|
+
if (budget.remaining < 0) {
|
|
120
|
+
throw new Error(`the archive's parts exceed the ${MAX_DOC_TOTAL_BYTES}-byte (200 MB) total extraction limit`);
|
|
121
|
+
}
|
|
122
|
+
return entry.getData().toString('utf8');
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/** The part CHOSEN for a slide number: its name as well as its bytes. */
|
|
126
|
+
interface SlidePart {
|
|
127
|
+
/** Full part name, e.g. `ppt/slides/slide01.xml`. */
|
|
128
|
+
name: string;
|
|
129
|
+
xml: string;
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Entries matching `re` (capture 1 = number), decoded as UTF-8, keyed by
|
|
134
|
+
* number. Two part names can parse to the SAME number (`slide1.xml` and
|
|
135
|
+
* `slide01.xml`); the winner is the FIRST in ascending part-name order —
|
|
136
|
+
* deterministic regardless of zip entry order, and the same policy as the
|
|
137
|
+
* browser twin (`pptxOutline.ts`), so viewer and `read_file` agree. Losing
|
|
138
|
+
* duplicates are never inflated (no budget charge).
|
|
139
|
+
*
|
|
140
|
+
* The winner's NAME rides along with its bytes because everything else about
|
|
141
|
+
* a slide hangs off the part name, not the number: see `collectNotes`.
|
|
142
|
+
*/
|
|
143
|
+
function collectNumbered(zip: AdmZip, re: RegExp, budget: ReadBudget): Map<number, SlidePart> {
|
|
144
|
+
const matched: Array<[number, string, AdmZip.IZipEntry]> = [];
|
|
145
|
+
for (const entry of zip.getEntries()) {
|
|
146
|
+
const m = re.exec(entry.entryName);
|
|
147
|
+
if (!m) continue;
|
|
148
|
+
// A crafted name can spell a number past 2^53 (or Infinity): distinct
|
|
149
|
+
// parts would collide in the map and one would silently vanish.
|
|
150
|
+
const n = parseInt(m[1], 10);
|
|
151
|
+
if (Number.isSafeInteger(n)) matched.push([n, entry.entryName, entry]);
|
|
152
|
+
}
|
|
153
|
+
// A zip may list the SAME part name twice. Sorting by name leaves those two
|
|
154
|
+
// in archive order, so which one wins depends on how the file was written —
|
|
155
|
+
// and two archives with identical parts would extract differently. A part
|
|
156
|
+
// claimed twice is not a part this reader can resolve, so it is dropped.
|
|
157
|
+
const claims = new Map<string, number>();
|
|
158
|
+
for (const [, name] of matched) claims.set(name, (claims.get(name) ?? 0) + 1);
|
|
159
|
+
const unique = matched.filter(([, name]) => claims.get(name) === 1);
|
|
160
|
+
unique.sort((a, b) => (a[1] < b[1] ? -1 : a[1] > b[1] ? 1 : 0));
|
|
161
|
+
const out = new Map<number, SlidePart>();
|
|
162
|
+
for (const [n, name, entry] of unique) {
|
|
163
|
+
if (!out.has(n)) out.set(n, { name, xml: readEntryBounded(entry, budget) });
|
|
164
|
+
}
|
|
165
|
+
return out;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
const NOTES_REL_TYPE_SUFFIX = '/notesSlide';
|
|
169
|
+
const SLIDE_REL_TYPE_SUFFIX = '/slide';
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* The value of the attribute whose LOCAL name is `want` AND that carries a
|
|
173
|
+
* namespace prefix. `<p:sldId>` holds both its own `id` and the relationship
|
|
174
|
+
* reference `r:id`; plain local-name matching answers with whichever is
|
|
175
|
+
* written first, so the relationship id must be the PREFIXED one.
|
|
176
|
+
*/
|
|
177
|
+
function prefixedAttrByLocalName(attributes: Record<string, string>, want: string): string | undefined {
|
|
178
|
+
for (const [key, value] of Object.entries(attributes)) {
|
|
179
|
+
if (key === 'xmlns' || key.startsWith('xmlns:')) continue;
|
|
180
|
+
if (key.includes(':') && localName(key) === want) return value;
|
|
181
|
+
}
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Slide part names in PRESENTATION order: `ppt/presentation.xml`'s
|
|
187
|
+
* `<p:sldIdLst>` entries, each `r:id` resolved through the presentation's own
|
|
188
|
+
* rels part — or undefined when the package has no readable list. Reordering
|
|
189
|
+
* slides in PowerPoint rewrites the sldIdLst and leaves the part names alone,
|
|
190
|
+
* so `slide1.xml` need not be the deck's first slide; the numeric filename
|
|
191
|
+
* sort is only the fallback for packages without the list.
|
|
192
|
+
*/
|
|
193
|
+
function slideOrderFromPresentation(zip: AdmZip, budget: ReadBudget): string[] | undefined {
|
|
194
|
+
const rels = zip.getEntry('ppt/_rels/presentation.xml.rels');
|
|
195
|
+
const pres = zip.getEntry('ppt/presentation.xml');
|
|
196
|
+
if (!rels || !pres) return undefined;
|
|
197
|
+
const targetById = new Map<string, string>();
|
|
198
|
+
for (const rel of localElementBlocks(readEntryBounded(rels, budget), ['Relationship'])) {
|
|
199
|
+
const type = attrByLocalName(rel.attributes, 'Type');
|
|
200
|
+
if (type === undefined || !type.endsWith(SLIDE_REL_TYPE_SUFFIX)) continue;
|
|
201
|
+
const id = attrByLocalName(rel.attributes, 'Id');
|
|
202
|
+
// Targets are RAW in the rels (see `notesTargetFromRels`) and relative to
|
|
203
|
+
// the presentation part's directory.
|
|
204
|
+
const target = attrByLocalName(rel.attributes, 'Target');
|
|
205
|
+
if (id !== undefined && target !== undefined) {
|
|
206
|
+
targetById.set(id, resolveRelTarget('ppt', decodeXmlEntities(target)));
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
if (targetById.size === 0) return undefined;
|
|
210
|
+
const list = localBlocks(readEntryBounded(pres, budget), 'sldIdLst')[0];
|
|
211
|
+
if (list === undefined) return undefined;
|
|
212
|
+
const order: string[] = [];
|
|
213
|
+
for (const sld of localElementBlocks(list, ['sldId'])) {
|
|
214
|
+
const rid = prefixedAttrByLocalName(sld.attributes, 'id');
|
|
215
|
+
const target = rid !== undefined ? targetById.get(rid) : undefined;
|
|
216
|
+
if (target !== undefined) order.push(target);
|
|
217
|
+
}
|
|
218
|
+
return order.length > 0 ? order : undefined;
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* The OPC relationships part of `partName` — `dir/_rels/base.rels`. Derived
|
|
223
|
+
* from the part NAME, never from the slide number: when a deck ships both
|
|
224
|
+
* `slide1.xml` and `slide01.xml`, the name-ordering rule picks `slide01.xml`,
|
|
225
|
+
* whose rels part is `slide01.xml.rels`. Rebuilding the path from the number
|
|
226
|
+
* asked for `slide1.xml.rels` — a part belonging to the OTHER file — and so
|
|
227
|
+
* either lost that slide's speaker notes or attached the losing part's.
|
|
228
|
+
*/
|
|
229
|
+
function relsPartName(partName: string): string {
|
|
230
|
+
const cut = partName.lastIndexOf('/');
|
|
231
|
+
return `${partName.slice(0, cut)}/_rels/${partName.slice(cut + 1)}.rels`;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
/**
|
|
235
|
+
* The conventional notes part for a slide part — `slide01.xml` →
|
|
236
|
+
* `notesSlide01.xml`. Derived from the name for the same reason as the rels
|
|
237
|
+
* path, so a zero-padded deck's fallback lands on the matching notes part.
|
|
238
|
+
*/
|
|
239
|
+
function conventionalNotesPart(partName: string): string {
|
|
240
|
+
const base = partName.slice(partName.lastIndexOf('/') + 1);
|
|
241
|
+
return `ppt/notesSlides/notes${base[0].toUpperCase()}${base.slice(1)}`;
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* Slide number → its notes part's XML, resolved through each slide's `.rels`
|
|
246
|
+
* part; the conventional-name fallback ONLY for a slide without a rels part.
|
|
247
|
+
* Both paths come from the SELECTED part's name (see `relsPartName`).
|
|
248
|
+
*/
|
|
249
|
+
function collectNotes(zip: AdmZip, slides: Map<number, SlidePart>, budget: ReadBudget): Map<number, string> {
|
|
250
|
+
const out = new Map<number, string>();
|
|
251
|
+
for (const [n, slide] of slides) {
|
|
252
|
+
const rels = zip.getEntry(relsPartName(slide.name));
|
|
253
|
+
let notesPart: string | undefined;
|
|
254
|
+
if (rels) {
|
|
255
|
+
const target = notesTargetFromRels(readEntryBounded(rels, budget));
|
|
256
|
+
notesPart = target !== undefined ? resolveRelTarget('ppt/slides', target) : undefined;
|
|
257
|
+
} else {
|
|
258
|
+
notesPart = conventionalNotesPart(slide.name);
|
|
259
|
+
}
|
|
260
|
+
if (notesPart === undefined) continue;
|
|
261
|
+
const entry = zip.getEntry(notesPart);
|
|
262
|
+
if (entry) out.set(n, readEntryBounded(entry, budget));
|
|
263
|
+
}
|
|
264
|
+
return out;
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* The Target of the first `notesSlide`-typed Relationship in a rels part, or
|
|
269
|
+
* undefined.
|
|
270
|
+
*
|
|
271
|
+
* Read by the parser: matched on the element's LOCAL name, so a producer that
|
|
272
|
+
* binds the relationships namespace to a prefix (`<r:Relationship r:Type=…>`)
|
|
273
|
+
* is read like any other — and a `<Relationship>`-looking fragment written
|
|
274
|
+
* inside a COMMENT or a CDATA section is text, not live metadata pointing the
|
|
275
|
+
* notes lookup at a part of its author's choosing.
|
|
276
|
+
*/
|
|
277
|
+
export function notesTargetFromRels(relsXml: string): string | undefined {
|
|
278
|
+
for (const rel of localElementBlocks(relsXml, ['Relationship'])) {
|
|
279
|
+
const type = attrByLocalName(rel.attributes, 'Type');
|
|
280
|
+
if (type !== undefined && type.endsWith(NOTES_REL_TYPE_SUFFIX)) {
|
|
281
|
+
// Attribute values are RAW here (the block reader does not decode), and a
|
|
282
|
+
// part name may legally contain `&`, written `&` in the rels.
|
|
283
|
+
const target = attrByLocalName(rel.attributes, 'Target');
|
|
284
|
+
return target !== undefined ? decodeXmlEntities(target) : undefined;
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
return undefined;
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
/** Resolve an OPC relationship Target against the part's base directory. */
|
|
291
|
+
export function resolveRelTarget(baseDir: string, target: string): string {
|
|
292
|
+
const parts = target.startsWith('/')
|
|
293
|
+
? target.slice(1).split('/')
|
|
294
|
+
: [...baseDir.split('/'), ...target.split('/')];
|
|
295
|
+
const out: string[] = [];
|
|
296
|
+
for (const p of parts) {
|
|
297
|
+
if (p === '' || p === '.') continue;
|
|
298
|
+
if (p === '..') out.pop();
|
|
299
|
+
else out.push(p);
|
|
300
|
+
}
|
|
301
|
+
return out.join('/');
|
|
302
|
+
}
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
import AdmZip from 'adm-zip';
|
|
2
|
+
import * as XLSX from 'xlsx';
|
|
3
|
+
import type { ExtractResult } from './doc-extract.types.js';
|
|
4
|
+
import { MAX_DOC_TOTAL_BYTES, zipEntryOversize } from './ooxml-text.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Per-sheet extraction caps. A worksheet's declared range can be enormous
|
|
8
|
+
* (a stray cell at XFD1048576 makes the range 16k x 1M); the caps bound the
|
|
9
|
+
* text an agent gets to something readable, and the extraction SAYS when it
|
|
10
|
+
* truncated (a `[sheet truncated …]` line right under the sheet marker).
|
|
11
|
+
*/
|
|
12
|
+
const MAX_ROWS_PER_SHEET = 10_000;
|
|
13
|
+
const MAX_COLS_PER_SHEET = 200;
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* Extract a `.xlsx` (Excel) workbook via SheetJS: per sheet a `[sheet: Name]`
|
|
17
|
+
* marker, then the rows as tab-separated values (each cell's FORMATTED value,
|
|
18
|
+
* e.g. dates as dates; tabs/newlines INSIDE a cell become single spaces so a
|
|
19
|
+
* cell's line break never reads as a row boundary), with trailing empty rows
|
|
20
|
+
* trimmed.
|
|
21
|
+
*/
|
|
22
|
+
export function extractXlsx(bytes: Buffer): ExtractResult {
|
|
23
|
+
// A real .xlsx is a zip. Check the signature OURSELVES because SheetJS
|
|
24
|
+
// helpfully falls back to parsing arbitrary bytes as CSV/HTML — which would
|
|
25
|
+
// turn a corrupt upload into confident nonsense instead of an honest error.
|
|
26
|
+
if (bytes.length < 4 || bytes[0] !== 0x50 || bytes[1] !== 0x4b) {
|
|
27
|
+
return { ok: false, message: 'could not be parsed as a .xlsx (not a zip archive)' };
|
|
28
|
+
}
|
|
29
|
+
// Zip-bomb bound BEFORE SheetJS inflates anything: the central directory
|
|
30
|
+
// declares every entry's uncompressed size, so per-part (50 MB) and
|
|
31
|
+
// aggregate (200 MB) limits cost one directory scan. A zip AdmZip cannot
|
|
32
|
+
// read falls through — SheetJS then reports its own parse failure.
|
|
33
|
+
try {
|
|
34
|
+
const zip = new AdmZip(bytes);
|
|
35
|
+
let total = 0;
|
|
36
|
+
for (const entry of zip.getEntries()) {
|
|
37
|
+
const oversize = zipEntryOversize(entry);
|
|
38
|
+
if (oversize) return { ok: false, message: `could not be extracted as a .xlsx (${oversize})` };
|
|
39
|
+
total += entry.header.size;
|
|
40
|
+
if (total > MAX_DOC_TOTAL_BYTES) {
|
|
41
|
+
return {
|
|
42
|
+
ok: false,
|
|
43
|
+
message: `could not be extracted as a .xlsx (the archive's parts exceed the ${MAX_DOC_TOTAL_BYTES}-byte (200 MB) total extraction limit)`,
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
} catch {
|
|
48
|
+
// not AdmZip-readable — let SheetJS produce the typed parse failure below
|
|
49
|
+
}
|
|
50
|
+
let wb: XLSX.WorkBook;
|
|
51
|
+
try {
|
|
52
|
+
wb = XLSX.read(bytes, { type: 'buffer' });
|
|
53
|
+
} catch (err) {
|
|
54
|
+
return { ok: false, message: `could not be parsed as a .xlsx (${(err as Error).message})` };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const lines: string[] = [];
|
|
58
|
+
for (const name of wb.SheetNames) {
|
|
59
|
+
// Control separators become spaces — the same rule as the ods extractor:
|
|
60
|
+
// a crafted workbook's sheet name holding a tab or newline would split
|
|
61
|
+
// the `[sheet: …]` marker's own line and break grep line numbers.
|
|
62
|
+
lines.push(`[sheet: ${name.replace(/[\t\n\r]+/g, ' ')}]`);
|
|
63
|
+
const ws = wb.Sheets[name];
|
|
64
|
+
const ref = ws?.['!ref'];
|
|
65
|
+
if (!ws || !ref) continue; // empty sheet — marker only
|
|
66
|
+
const range = XLSX.utils.decode_range(ref);
|
|
67
|
+
const truncated: string[] = [];
|
|
68
|
+
if (range.e.r - range.s.r + 1 > MAX_ROWS_PER_SHEET) {
|
|
69
|
+
range.e.r = range.s.r + MAX_ROWS_PER_SHEET - 1;
|
|
70
|
+
truncated.push(`first ${MAX_ROWS_PER_SHEET} rows`);
|
|
71
|
+
}
|
|
72
|
+
if (range.e.c - range.s.c + 1 > MAX_COLS_PER_SHEET) {
|
|
73
|
+
range.e.c = range.s.c + MAX_COLS_PER_SHEET - 1;
|
|
74
|
+
truncated.push(`first ${MAX_COLS_PER_SHEET} columns`);
|
|
75
|
+
}
|
|
76
|
+
if (truncated.length > 0) {
|
|
77
|
+
lines.push(`[sheet truncated to the ${truncated.join(' and ')}]`);
|
|
78
|
+
ws['!ref'] = XLSX.utils.encode_range(range);
|
|
79
|
+
}
|
|
80
|
+
// Rows via sheet_to_json (formatted cell text, one array per worksheet
|
|
81
|
+
// row): sheet_to_csv CSV-quotes a cell containing a line break, so
|
|
82
|
+
// splitting its output on newlines turned ONE worksheet row into several
|
|
83
|
+
// extracted rows. Cell-internal tabs/newlines become single spaces
|
|
84
|
+
// instead — the same one-line-per-row contract as the ods extractor.
|
|
85
|
+
const rows = XLSX.utils
|
|
86
|
+
.sheet_to_json<unknown[]>(ws, { header: 1, raw: false, defval: '', blankrows: true })
|
|
87
|
+
.map((cells) => cells.map((c) => String(c).replace(/[\t\n\r]+/g, ' ')).join('\t'));
|
|
88
|
+
while (rows.length > 0 && rows[rows.length - 1].replace(/\t/g, '') === '') rows.pop();
|
|
89
|
+
lines.push(...rows);
|
|
90
|
+
}
|
|
91
|
+
return {
|
|
92
|
+
ok: true,
|
|
93
|
+
summary: `${wb.SheetNames.length} sheet${wb.SheetNames.length === 1 ? '' : 's'}, rows as tab-separated values; formulas, formatting and charts omitted`,
|
|
94
|
+
text: lines.join('\n'),
|
|
95
|
+
};
|
|
96
|
+
}
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
import { promises as fs } from 'node:fs';
|
|
3
|
+
import path from 'node:path';
|
|
4
|
+
import type { ExtractedDoc } from './doc-extract.types.js';
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Git BLOB sha of `bytes` — sha1 over `"blob <len>\0" + bytes`, exactly what
|
|
8
|
+
* `git hash-object` computes. Chosen as the cache key because the workspace is
|
|
9
|
+
* a git clone: for a clean tracked file this equals the sha `git ls-files -s`
|
|
10
|
+
* would report, WITHOUT spawning git — and because it is computed from the
|
|
11
|
+
* bytes actually read, it stays correct (it just stops matching the index)
|
|
12
|
+
* when the file is untracked or dirty. One code path, no fallback branch.
|
|
13
|
+
*/
|
|
14
|
+
export function gitBlobSha(bytes: Buffer): string {
|
|
15
|
+
return createHash('sha1').update(`blob ${bytes.length}\0`).update(bytes).digest('hex');
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/** Default size bound for the on-disk extraction cache. */
|
|
19
|
+
const DEFAULT_MAX_TOTAL_BYTES = 512 * 1024 * 1024; // 512MB
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* On-disk cache of document extractions. The KEY is the caller's business —
|
|
23
|
+
* `DocExtractService` passes git blob sha + extension (content hash, so a
|
|
24
|
+
* re-upload of identical bytes, or the same document on another branch or
|
|
25
|
+
* path, hits the same entry; extension, so identical bytes under another
|
|
26
|
+
* FORMAT never return the wrong extractor's output). One JSON file per entry
|
|
27
|
+
* (`<key>.json` holding the `{ summary, text }`), under a dedicated root
|
|
28
|
+
* that — like the spill
|
|
29
|
+
* store's — sits BESIDE the workspaces root, never inside a workspace and
|
|
30
|
+
* never committed.
|
|
31
|
+
*
|
|
32
|
+
* Bounding: writes best-effort prune OLDEST-MTIME entries until the total
|
|
33
|
+
* size fits `maxTotalBytes` (simple LRU-ish eviction; reads don't touch
|
|
34
|
+
* mtime, so it is closer to FIFO — good enough for a cache whose entries are
|
|
35
|
+
* cheap to rebuild). The full scan is deferred until the writes since the
|
|
36
|
+
* last one could plausibly have reached the bound (see `writtenSinceScan`).
|
|
37
|
+
* Every filesystem error here is swallowed: the cache is an accelerator,
|
|
38
|
+
* never a reason for a read to fail.
|
|
39
|
+
*/
|
|
40
|
+
export class DocExtractionCache {
|
|
41
|
+
constructor(
|
|
42
|
+
private readonly root: string,
|
|
43
|
+
private readonly maxTotalBytes: number = DEFAULT_MAX_TOTAL_BYTES,
|
|
44
|
+
) {}
|
|
45
|
+
|
|
46
|
+
/** The cached extraction for `sha`, or undefined on miss/corrupt entry. */
|
|
47
|
+
async get(sha: string): Promise<ExtractedDoc | undefined> {
|
|
48
|
+
try {
|
|
49
|
+
const parsed: unknown = JSON.parse(await fs.readFile(this.entryPath(sha), 'utf8'));
|
|
50
|
+
if (
|
|
51
|
+
typeof parsed === 'object' && parsed !== null &&
|
|
52
|
+
typeof (parsed as ExtractedDoc).summary === 'string' &&
|
|
53
|
+
typeof (parsed as ExtractedDoc).text === 'string'
|
|
54
|
+
) {
|
|
55
|
+
return { summary: (parsed as ExtractedDoc).summary, text: (parsed as ExtractedDoc).text };
|
|
56
|
+
}
|
|
57
|
+
return undefined;
|
|
58
|
+
} catch {
|
|
59
|
+
return undefined; // miss, unreadable or corrupt — treated identically
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Bytes written since the last full scan. The scan costs a `readdir` plus a
|
|
65
|
+
* `stat` per entry, and running it on EVERY cold extraction made a cache that
|
|
66
|
+
* is nowhere near its bound pay for the bound anyway. Writes are accumulated
|
|
67
|
+
* instead and the scan runs once they could plausibly have reached it.
|
|
68
|
+
*/
|
|
69
|
+
private writtenSinceScan = 0;
|
|
70
|
+
|
|
71
|
+
/** Total size the last scan found (after eviction); undefined before the first scan, so the first write scans. */
|
|
72
|
+
private totalAtLastScan: number | undefined;
|
|
73
|
+
|
|
74
|
+
/** The prune in flight, so concurrent puts share ONE scan instead of racing their own. */
|
|
75
|
+
private pruning: Promise<void> | undefined;
|
|
76
|
+
|
|
77
|
+
/** Store an extraction under `sha`; prunes towards the size bound. Never throws. */
|
|
78
|
+
async put(sha: string, doc: ExtractedDoc): Promise<void> {
|
|
79
|
+
try {
|
|
80
|
+
await fs.mkdir(this.root, { recursive: true });
|
|
81
|
+
const payload = JSON.stringify({ summary: doc.summary, text: doc.text });
|
|
82
|
+
await fs.writeFile(this.entryPath(sha), payload, 'utf8');
|
|
83
|
+
this.writtenSinceScan += Buffer.byteLength(payload);
|
|
84
|
+
if (
|
|
85
|
+
this.totalAtLastScan === undefined ||
|
|
86
|
+
this.totalAtLastScan + this.writtenSinceScan > this.maxTotalBytes
|
|
87
|
+
) {
|
|
88
|
+
this.pruning ??= this.prune().finally(() => {
|
|
89
|
+
this.pruning = undefined;
|
|
90
|
+
});
|
|
91
|
+
await this.pruning;
|
|
92
|
+
}
|
|
93
|
+
} catch {
|
|
94
|
+
// cache write failed — the extraction still returns; next read re-extracts
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* The file backing `key`, always INSIDE the cache root.
|
|
100
|
+
*
|
|
101
|
+
* The key is built from a content hash and an extension taken from a
|
|
102
|
+
* workspace path, so a path spelling `../` — or a Windows backslash — would
|
|
103
|
+
* otherwise resolve outside the root and let a read or a write reach an
|
|
104
|
+
* arbitrary file. Only the characters a real key uses survive.
|
|
105
|
+
*/
|
|
106
|
+
private entryPath(key: string): string {
|
|
107
|
+
const safe = key.replace(/[^A-Za-z0-9._-]/g, '_');
|
|
108
|
+
return path.join(this.root, `${safe}.json`);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/** Delete oldest-mtime entries until the total size fits the bound. */
|
|
112
|
+
private async prune(): Promise<void> {
|
|
113
|
+
// Snapshot FIRST: a write counted before the scan starts is on disk when
|
|
114
|
+
// `readdir` runs, so the scan settles exactly those bytes. A put that
|
|
115
|
+
// lands DURING the scan keeps its count (resetting to zero discarded it,
|
|
116
|
+
// and a later put then trusted a total the scan never saw), so the next
|
|
117
|
+
// put still prunes instead of leaving the cache above the bound.
|
|
118
|
+
const scanned = this.writtenSinceScan;
|
|
119
|
+
const entries: Array<{ p: string; size: number; mtime: number }> = [];
|
|
120
|
+
let total = 0;
|
|
121
|
+
for (const name of await fs.readdir(this.root)) {
|
|
122
|
+
try {
|
|
123
|
+
const st = await fs.stat(path.join(this.root, name));
|
|
124
|
+
if (!st.isFile()) continue;
|
|
125
|
+
entries.push({ p: path.join(this.root, name), size: st.size, mtime: st.mtimeMs });
|
|
126
|
+
total += st.size;
|
|
127
|
+
} catch {
|
|
128
|
+
// vanished mid-scan — ignore
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
if (total > this.maxTotalBytes) {
|
|
132
|
+
entries.sort((a, b) => a.mtime - b.mtime);
|
|
133
|
+
for (const e of entries) {
|
|
134
|
+
if (total <= this.maxTotalBytes) break;
|
|
135
|
+
await fs.rm(e.p, { force: true });
|
|
136
|
+
total -= e.size;
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
this.totalAtLastScan = total;
|
|
140
|
+
this.writtenSinceScan -= scanned;
|
|
141
|
+
}
|
|
142
|
+
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import type { DocExtractService } from './doc-extract.service.js';
|
|
2
|
+
import { DocumentReader } from './document-reader.js';
|
|
3
|
+
import { EmailReader } from './email-reader.js';
|
|
4
|
+
import { extractDocx } from './extract-docx.js';
|
|
5
|
+
import { extractEml } from './extract-eml.js';
|
|
6
|
+
import { extractMsg } from './extract-msg.js';
|
|
7
|
+
import { extractOdp } from './extract-odp.js';
|
|
8
|
+
import { extractOds } from './extract-ods.js';
|
|
9
|
+
import { extractOdt } from './extract-odt.js';
|
|
10
|
+
import { extractPdf } from './extract-pdf.js';
|
|
11
|
+
import { extractPptx } from './extract-pptx.js';
|
|
12
|
+
import { extractXlsx } from './extract-xlsx.js';
|
|
13
|
+
import { FileReaderRegistry } from './file-reader.js';
|
|
14
|
+
import { ImageReader } from './image-reader.js';
|
|
15
|
+
import { LegacyOfficeReader, TextReader } from './text-reader.js';
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* THE registry of file readers — the one place that says which reader owns
|
|
19
|
+
* which extension. Adding a format is one entry here (plus its reader/extract
|
|
20
|
+
* function); read_file, grep and the write-refusal all follow automatically.
|
|
21
|
+
*
|
|
22
|
+
* `docExtract` is the shared content-hash extraction cache the document
|
|
23
|
+
* readers wrap their pure extract functions with (see DocumentReader).
|
|
24
|
+
* Everything not claimed below falls back to the TextReader.
|
|
25
|
+
*/
|
|
26
|
+
export function createFileReaderRegistry(docExtract: DocExtractService): FileReaderRegistry {
|
|
27
|
+
return new FileReaderRegistry(
|
|
28
|
+
[
|
|
29
|
+
new DocumentReader('.docx', extractDocx, docExtract),
|
|
30
|
+
new DocumentReader('.pptx', extractPptx, docExtract),
|
|
31
|
+
new DocumentReader('.xlsx', extractXlsx, docExtract),
|
|
32
|
+
new DocumentReader('.pdf', extractPdf, docExtract),
|
|
33
|
+
new DocumentReader('.odt', extractOdt, docExtract),
|
|
34
|
+
new DocumentReader('.odp', extractOdp, docExtract),
|
|
35
|
+
new DocumentReader('.ods', extractOds, docExtract),
|
|
36
|
+
// Email files ride the same document machinery (cached extraction,
|
|
37
|
+
// greppable, not text-editable) with an email-honest write refusal.
|
|
38
|
+
new EmailReader('.eml', extractEml, docExtract),
|
|
39
|
+
new EmailReader('.msg', extractMsg, docExtract),
|
|
40
|
+
new ImageReader(),
|
|
41
|
+
new LegacyOfficeReader(),
|
|
42
|
+
],
|
|
43
|
+
new TextReader(),
|
|
44
|
+
);
|
|
45
|
+
}
|