sensemaking 0.15.4 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -5
- package/dist/cjs/chunk/extract.d.cts +2 -0
- package/dist/cjs/chunk/extract.d.ts +2 -0
- package/dist/cjs/chunk/extract.js +139 -0
- package/dist/cjs/chunk/extract.js.map +1 -0
- package/dist/cjs/chunk/group.d.cts +4 -0
- package/dist/cjs/chunk/group.d.ts +4 -0
- package/dist/cjs/chunk/group.js +489 -0
- package/dist/cjs/chunk/group.js.map +1 -0
- package/dist/cjs/chunk/index.d.cts +8 -0
- package/dist/cjs/chunk/index.d.ts +8 -0
- package/dist/cjs/chunk/index.js +44 -0
- package/dist/cjs/chunk/index.js.map +1 -0
- package/dist/cjs/chunk/parse.d.cts +2 -0
- package/dist/cjs/chunk/parse.d.ts +2 -0
- package/dist/cjs/chunk/parse.js +71 -0
- package/dist/cjs/chunk/parse.js.map +1 -0
- package/dist/cjs/chunk/types.d.cts +19 -0
- package/dist/cjs/chunk/types.d.ts +19 -0
- package/dist/cjs/chunk/types.js +5 -0
- package/dist/cjs/chunk/types.js.map +1 -0
- package/dist/cjs/chunk/version.d.cts +1 -0
- package/dist/cjs/chunk/version.d.ts +1 -0
- package/dist/cjs/chunk/version.js +14 -0
- package/dist/cjs/chunk/version.js.map +1 -0
- package/dist/cjs/cli/download.js +12 -11
- package/dist/cjs/cli/download.js.map +1 -1
- package/dist/cjs/cli/index.d.cts +1 -1
- package/dist/cjs/cli/index.d.ts +1 -1
- package/dist/cjs/cli/index.js +1 -1
- package/dist/cjs/cli/index.js.map +1 -1
- package/dist/cjs/cli/init.js +31 -5
- package/dist/cjs/cli/init.js.map +1 -1
- package/dist/cjs/cli/map.js +1 -1
- package/dist/cjs/cli/map.js.map +1 -1
- package/dist/cjs/cli/named.js +1 -1
- package/dist/cjs/cli/named.js.map +1 -1
- package/dist/cjs/cli/path.js +2 -2
- package/dist/cjs/cli/path.js.map +1 -1
- package/dist/cjs/cli/peek.js +1 -1
- package/dist/cjs/cli/peek.js.map +1 -1
- package/dist/cjs/cli/related.js +1 -1
- package/dist/cjs/cli/related.js.map +1 -1
- package/dist/cjs/cli/search.js +1 -1
- package/dist/cjs/cli/search.js.map +1 -1
- package/dist/cjs/cli/shared.d.cts +1 -1
- package/dist/cjs/cli/shared.d.ts +1 -1
- package/dist/cjs/cli/shared.js +3 -3
- package/dist/cjs/cli/shared.js.map +1 -1
- package/dist/cjs/cli/status.js +344 -100
- package/dist/cjs/cli/status.js.map +1 -1
- package/dist/cjs/commands/map.d.cts +1 -1
- package/dist/cjs/commands/map.d.ts +1 -1
- package/dist/cjs/commands/map.js.map +1 -1
- package/dist/cjs/commands/peek.d.cts +1 -1
- package/dist/cjs/commands/peek.d.ts +1 -1
- package/dist/cjs/commands/peek.js +13 -6
- package/dist/cjs/commands/peek.js.map +1 -1
- package/dist/cjs/commands/related.js +17 -13
- package/dist/cjs/commands/related.js.map +1 -1
- package/dist/cjs/commands/scope.js +1 -1
- package/dist/cjs/commands/scope.js.map +1 -1
- package/dist/cjs/commands/search.d.cts +1 -1
- package/dist/cjs/commands/search.d.ts +1 -1
- package/dist/cjs/commands/search.js +28 -99
- package/dist/cjs/commands/search.js.map +1 -1
- package/dist/cjs/commands/signals.d.cts +18 -0
- package/dist/cjs/commands/signals.d.ts +18 -0
- package/dist/cjs/commands/signals.js +299 -0
- package/dist/cjs/commands/signals.js.map +1 -0
- package/dist/cjs/commands/status.d.cts +2 -2
- package/dist/cjs/commands/status.d.ts +2 -2
- package/dist/cjs/commands/status.js +1 -1
- package/dist/cjs/commands/status.js.map +1 -1
- package/dist/cjs/config/access.d.cts +8 -3
- package/dist/cjs/config/access.d.ts +8 -3
- package/dist/cjs/config/access.js +35 -17
- package/dist/cjs/config/access.js.map +1 -1
- package/dist/cjs/config/index.d.cts +5 -1
- package/dist/cjs/config/index.d.ts +5 -1
- package/dist/cjs/config/index.js +16 -2
- package/dist/cjs/config/index.js.map +1 -1
- package/dist/cjs/config/load.d.cts +6 -1
- package/dist/cjs/config/load.d.ts +6 -1
- package/dist/cjs/config/load.js +83 -8
- package/dist/cjs/config/load.js.map +1 -1
- package/dist/cjs/config/resolve.js +1 -1
- package/dist/cjs/config/resolve.js.map +1 -1
- package/dist/cjs/config/signals.d.cts +4 -0
- package/dist/cjs/config/signals.d.ts +4 -0
- package/dist/cjs/config/signals.js +30 -0
- package/dist/cjs/config/signals.js.map +1 -0
- package/dist/cjs/config/types.d.cts +7 -4
- package/dist/cjs/config/types.d.ts +7 -4
- package/dist/cjs/config/types.js +1 -1
- package/dist/cjs/config/types.js.map +1 -1
- package/dist/cjs/config/validate.d.cts +1 -0
- package/dist/cjs/config/validate.d.ts +1 -0
- package/dist/cjs/config/validate.js +127 -11
- package/dist/cjs/config/validate.js.map +1 -1
- package/dist/cjs/db/open.d.cts +1 -1
- package/dist/cjs/db/open.d.ts +1 -1
- package/dist/cjs/db/open.js +9 -4
- package/dist/cjs/db/open.js.map +1 -1
- package/dist/cjs/db/reconcile.d.cts +1 -0
- package/dist/cjs/db/reconcile.d.ts +1 -0
- package/dist/cjs/db/reconcile.js +23 -8
- package/dist/cjs/db/reconcile.js.map +1 -1
- package/dist/cjs/{sql-functions.js → db/sql-functions.js} +1 -1
- package/dist/cjs/db/sql-functions.js.map +1 -0
- package/dist/cjs/embed/cohere.d.cts +2 -0
- package/dist/cjs/embed/cohere.d.ts +2 -0
- package/dist/cjs/embed/cohere.js +238 -0
- package/dist/cjs/embed/cohere.js.map +1 -0
- package/dist/cjs/embed/http.d.cts +2 -0
- package/dist/cjs/embed/http.d.ts +2 -0
- package/dist/cjs/embed/http.js +258 -0
- package/dist/cjs/embed/http.js.map +1 -0
- package/dist/cjs/embed/identity.d.cts +13 -0
- package/dist/cjs/embed/identity.d.ts +13 -0
- package/dist/cjs/embed/identity.js +194 -0
- package/dist/cjs/embed/identity.js.map +1 -0
- package/dist/cjs/embed/langfit.d.cts +6 -0
- package/dist/cjs/embed/langfit.d.ts +6 -0
- package/dist/cjs/embed/langfit.js +286 -0
- package/dist/cjs/embed/langfit.js.map +1 -0
- package/dist/cjs/embed/languages.d.cts +3 -0
- package/dist/cjs/embed/languages.d.ts +3 -0
- package/dist/cjs/embed/languages.js +98 -0
- package/dist/cjs/embed/languages.js.map +1 -0
- package/dist/cjs/embed/openai.d.cts +2 -0
- package/dist/cjs/embed/openai.d.ts +2 -0
- package/dist/cjs/embed/openai.js +268 -0
- package/dist/cjs/embed/openai.js.map +1 -0
- package/dist/cjs/embed/query.d.cts +21 -0
- package/dist/cjs/embed/query.d.ts +21 -0
- package/dist/cjs/embed/query.js +610 -0
- package/dist/cjs/embed/query.js.map +1 -0
- package/dist/cjs/embed/registry.d.cts +3 -0
- package/dist/cjs/embed/registry.d.ts +3 -0
- package/dist/cjs/embed/registry.js +114 -0
- package/dist/cjs/embed/registry.js.map +1 -0
- package/dist/cjs/embed/static.d.cts +2 -0
- package/dist/cjs/embed/static.d.ts +2 -0
- package/dist/cjs/embed/static.js +330 -0
- package/dist/cjs/embed/static.js.map +1 -0
- package/dist/cjs/embed/store.d.cts +8 -0
- package/dist/cjs/embed/store.d.ts +8 -0
- package/dist/cjs/embed/store.js +498 -0
- package/dist/cjs/embed/store.js.map +1 -0
- package/dist/cjs/embed/types.d.cts +8 -0
- package/dist/cjs/embed/types.d.ts +8 -0
- package/dist/cjs/embed/types.js +7 -0
- package/dist/cjs/embed/types.js.map +1 -0
- package/dist/cjs/errors.d.cts +1 -1
- package/dist/cjs/errors.d.ts +1 -1
- package/dist/cjs/errors.js.map +1 -1
- package/dist/cjs/features/embed.d.cts +2 -28
- package/dist/cjs/features/embed.d.ts +2 -28
- package/dist/cjs/features/embed.js +39 -1027
- package/dist/cjs/features/embed.js.map +1 -1
- package/dist/cjs/{fences.js → features/fences.js} +1 -1
- package/dist/cjs/features/fences.js.map +1 -0
- package/dist/cjs/features/links.js +1 -1
- package/dist/cjs/features/links.js.map +1 -1
- package/dist/cjs/features/rank.js +1 -1
- package/dist/cjs/features/rank.js.map +1 -1
- package/dist/cjs/features/sections.js +4 -3
- package/dist/cjs/features/sections.js.map +1 -1
- package/dist/cjs/features/tags.js +1 -1
- package/dist/cjs/features/tags.js.map +1 -1
- package/dist/cjs/features/types.d.cts +4 -2
- package/dist/cjs/features/types.d.ts +4 -2
- package/dist/cjs/graph/graph.js.map +1 -0
- package/dist/cjs/graph/traverse.js.map +1 -0
- package/dist/cjs/index.d.cts +4 -4
- package/dist/cjs/index.d.ts +4 -4
- package/dist/cjs/index.js +4 -1
- package/dist/cjs/index.js.map +1 -1
- package/dist/cjs/output/column-hint.js.map +1 -0
- package/dist/cjs/{output.d.cts → output/output.d.cts} +2 -2
- package/dist/cjs/{output.d.ts → output/output.d.ts} +2 -2
- package/dist/cjs/{output.js → output/output.js} +9 -2
- package/dist/cjs/output/output.js.map +1 -0
- package/dist/cjs/output/progress.js.map +1 -0
- package/dist/cjs/{search-error.js → output/search-error.js} +1 -1
- package/dist/cjs/output/search-error.js.map +1 -0
- package/dist/cjs/scan/frontmatter.d.cts +13 -0
- package/dist/cjs/scan/frontmatter.d.ts +13 -0
- package/dist/cjs/scan/frontmatter.js +229 -0
- package/dist/cjs/scan/frontmatter.js.map +1 -0
- package/dist/cjs/scan/index.d.cts +25 -0
- package/dist/cjs/scan/index.d.ts +25 -0
- package/dist/cjs/scan/index.js +107 -0
- package/dist/cjs/scan/index.js.map +1 -0
- package/dist/cjs/scan/list.d.cts +12 -0
- package/dist/cjs/scan/list.d.ts +12 -0
- package/dist/cjs/scan/list.js +154 -0
- package/dist/cjs/scan/list.js.map +1 -0
- package/dist/cjs/text/segment.d.cts +3 -0
- package/dist/cjs/text/segment.d.ts +3 -0
- package/dist/cjs/text/segment.js +154 -0
- package/dist/cjs/text/segment.js.map +1 -0
- package/dist/cjs/text/strip.d.cts +3 -0
- package/dist/cjs/text/strip.d.ts +3 -0
- package/dist/cjs/text/strip.js +32 -0
- package/dist/cjs/text/strip.js.map +1 -0
- package/dist/esm/chunk/extract.d.ts +2 -0
- package/dist/esm/chunk/extract.js +114 -0
- package/dist/esm/chunk/extract.js.map +1 -0
- package/dist/esm/chunk/group.d.ts +4 -0
- package/dist/esm/chunk/group.js +253 -0
- package/dist/esm/chunk/group.js.map +1 -0
- package/dist/esm/chunk/index.d.ts +8 -0
- package/dist/esm/chunk/index.js +16 -0
- package/dist/esm/chunk/index.js.map +1 -0
- package/dist/esm/chunk/parse.d.ts +2 -0
- package/dist/esm/chunk/parse.js +63 -0
- package/dist/esm/chunk/parse.js.map +1 -0
- package/dist/esm/chunk/types.d.ts +19 -0
- package/dist/esm/chunk/types.js +2 -0
- package/dist/esm/chunk/types.js.map +1 -0
- package/dist/esm/chunk/version.d.ts +1 -0
- package/dist/esm/chunk/version.js +3 -0
- package/dist/esm/chunk/version.js.map +1 -0
- package/dist/esm/cli/download.js +9 -8
- package/dist/esm/cli/download.js.map +1 -1
- package/dist/esm/cli/index.d.ts +1 -1
- package/dist/esm/cli/index.js +1 -1
- package/dist/esm/cli/index.js.map +1 -1
- package/dist/esm/cli/init.js +32 -6
- package/dist/esm/cli/init.js.map +1 -1
- package/dist/esm/cli/map.js +1 -1
- package/dist/esm/cli/map.js.map +1 -1
- package/dist/esm/cli/named.js +1 -1
- package/dist/esm/cli/named.js.map +1 -1
- package/dist/esm/cli/path.js +2 -2
- package/dist/esm/cli/path.js.map +1 -1
- package/dist/esm/cli/peek.js +1 -1
- package/dist/esm/cli/peek.js.map +1 -1
- package/dist/esm/cli/related.js +1 -1
- package/dist/esm/cli/related.js.map +1 -1
- package/dist/esm/cli/search.js +1 -1
- package/dist/esm/cli/search.js.map +1 -1
- package/dist/esm/cli/shared.d.ts +1 -1
- package/dist/esm/cli/shared.js +3 -3
- package/dist/esm/cli/shared.js.map +1 -1
- package/dist/esm/cli/status.js +37 -9
- package/dist/esm/cli/status.js.map +1 -1
- package/dist/esm/commands/map.d.ts +1 -1
- package/dist/esm/commands/map.js.map +1 -1
- package/dist/esm/commands/peek.d.ts +1 -1
- package/dist/esm/commands/peek.js +8 -1
- package/dist/esm/commands/peek.js.map +1 -1
- package/dist/esm/commands/related.js +14 -10
- package/dist/esm/commands/related.js.map +1 -1
- package/dist/esm/commands/scope.js +1 -1
- package/dist/esm/commands/scope.js.map +1 -1
- package/dist/esm/commands/search.d.ts +1 -1
- package/dist/esm/commands/search.js +30 -81
- package/dist/esm/commands/search.js.map +1 -1
- package/dist/esm/commands/signals.d.ts +18 -0
- package/dist/esm/commands/signals.js +78 -0
- package/dist/esm/commands/signals.js.map +1 -0
- package/dist/esm/commands/status.d.ts +2 -2
- package/dist/esm/commands/status.js +2 -2
- package/dist/esm/commands/status.js.map +1 -1
- package/dist/esm/config/access.d.ts +8 -3
- package/dist/esm/config/access.js +34 -15
- package/dist/esm/config/access.js.map +1 -1
- package/dist/esm/config/index.d.ts +5 -1
- package/dist/esm/config/index.js +3 -1
- package/dist/esm/config/index.js.map +1 -1
- package/dist/esm/config/load.d.ts +6 -1
- package/dist/esm/config/load.js +67 -9
- package/dist/esm/config/load.js.map +1 -1
- package/dist/esm/config/resolve.js +2 -2
- package/dist/esm/config/resolve.js.map +1 -1
- package/dist/esm/config/signals.d.ts +4 -0
- package/dist/esm/config/signals.js +14 -0
- package/dist/esm/config/signals.js.map +1 -0
- package/dist/esm/config/types.d.ts +7 -4
- package/dist/esm/config/types.js +1 -1
- package/dist/esm/config/types.js.map +1 -1
- package/dist/esm/config/validate.d.ts +1 -0
- package/dist/esm/config/validate.js +65 -12
- package/dist/esm/config/validate.js.map +1 -1
- package/dist/esm/db/open.d.ts +1 -1
- package/dist/esm/db/open.js +11 -6
- package/dist/esm/db/open.js.map +1 -1
- package/dist/esm/db/reconcile.d.ts +1 -0
- package/dist/esm/db/reconcile.js +14 -4
- package/dist/esm/db/reconcile.js.map +1 -1
- package/dist/esm/{sql-functions.js → db/sql-functions.js} +1 -1
- package/dist/esm/db/sql-functions.js.map +1 -0
- package/dist/esm/embed/cohere.d.ts +2 -0
- package/dist/esm/embed/cohere.js +46 -0
- package/dist/esm/embed/cohere.js.map +1 -0
- package/dist/esm/embed/http.d.ts +2 -0
- package/dist/esm/embed/http.js +33 -0
- package/dist/esm/embed/http.js.map +1 -0
- package/dist/esm/embed/identity.d.ts +13 -0
- package/dist/esm/embed/identity.js +92 -0
- package/dist/esm/embed/identity.js.map +1 -0
- package/dist/esm/embed/langfit.d.ts +6 -0
- package/dist/esm/embed/langfit.js +44 -0
- package/dist/esm/embed/langfit.js.map +1 -0
- package/dist/esm/embed/languages.d.ts +3 -0
- package/dist/esm/embed/languages.js +80 -0
- package/dist/esm/embed/languages.js.map +1 -0
- package/dist/esm/embed/openai.d.ts +2 -0
- package/dist/esm/embed/openai.js +49 -0
- package/dist/esm/embed/openai.js.map +1 -0
- package/dist/esm/embed/query.d.ts +21 -0
- package/dist/esm/embed/query.js +181 -0
- package/dist/esm/embed/query.js.map +1 -0
- package/dist/esm/embed/registry.d.ts +3 -0
- package/dist/esm/embed/registry.js +45 -0
- package/dist/esm/embed/registry.js.map +1 -0
- package/dist/esm/embed/static.d.ts +2 -0
- package/dist/esm/embed/static.js +71 -0
- package/dist/esm/embed/static.js.map +1 -0
- package/dist/esm/embed/store.d.ts +8 -0
- package/dist/esm/embed/store.js +123 -0
- package/dist/esm/embed/store.js.map +1 -0
- package/dist/esm/embed/types.d.ts +8 -0
- package/dist/esm/embed/types.js +3 -0
- package/dist/esm/embed/types.js.map +1 -0
- package/dist/esm/errors.d.ts +1 -1
- package/dist/esm/errors.js.map +1 -1
- package/dist/esm/features/embed.d.ts +2 -28
- package/dist/esm/features/embed.js +33 -377
- package/dist/esm/features/embed.js.map +1 -1
- package/dist/esm/{fences.js → features/fences.js} +1 -1
- package/dist/esm/features/fences.js.map +1 -0
- package/dist/esm/features/links.js +1 -1
- package/dist/esm/features/links.js.map +1 -1
- package/dist/esm/features/rank.js +1 -1
- package/dist/esm/features/rank.js.map +1 -1
- package/dist/esm/features/sections.js +4 -3
- package/dist/esm/features/sections.js.map +1 -1
- package/dist/esm/features/tags.js +1 -1
- package/dist/esm/features/tags.js.map +1 -1
- package/dist/esm/features/types.d.ts +4 -2
- package/dist/esm/features/types.js.map +1 -1
- package/dist/esm/graph/graph.js.map +1 -0
- package/dist/esm/graph/traverse.js.map +1 -0
- package/dist/esm/index.d.ts +4 -4
- package/dist/esm/index.js +2 -2
- package/dist/esm/index.js.map +1 -1
- package/dist/esm/output/column-hint.js.map +1 -0
- package/dist/esm/{output.d.ts → output/output.d.ts} +2 -2
- package/dist/esm/{output.js → output/output.js} +6 -2
- package/dist/esm/output/output.js.map +1 -0
- package/dist/esm/output/progress.js.map +1 -0
- package/dist/esm/{search-error.js → output/search-error.js} +1 -1
- package/dist/esm/output/search-error.js.map +1 -0
- package/dist/esm/scan/frontmatter.d.ts +13 -0
- package/dist/esm/{scan.js → scan/frontmatter.js} +4 -120
- package/dist/esm/scan/frontmatter.js.map +1 -0
- package/dist/esm/scan/index.d.ts +25 -0
- package/dist/esm/scan/index.js +58 -0
- package/dist/esm/scan/index.js.map +1 -0
- package/dist/esm/scan/list.d.ts +12 -0
- package/dist/esm/scan/list.js +58 -0
- package/dist/esm/scan/list.js.map +1 -0
- package/dist/esm/text/segment.d.ts +3 -0
- package/dist/esm/text/segment.js +95 -0
- package/dist/esm/text/segment.js.map +1 -0
- package/dist/esm/text/strip.d.ts +3 -0
- package/dist/esm/text/strip.js +12 -0
- package/dist/esm/text/strip.js.map +1 -0
- package/package.json +23 -2
- package/schema.json +22 -8
- package/skills/sense/SKILL.md +21 -5
- package/skills/sense-setup/EXAMPLES.md +6 -6
- package/skills/sense-setup/SKILL.md +7 -5
- package/dist/cjs/column-hint.js.map +0 -1
- package/dist/cjs/fences.js.map +0 -1
- package/dist/cjs/graph.js.map +0 -1
- package/dist/cjs/output.js.map +0 -1
- package/dist/cjs/progress.js.map +0 -1
- package/dist/cjs/scan.d.cts +0 -35
- package/dist/cjs/scan.d.ts +0 -35
- package/dist/cjs/scan.js +0 -433
- package/dist/cjs/scan.js.map +0 -1
- package/dist/cjs/search-error.js.map +0 -1
- package/dist/cjs/segment.d.cts +0 -2
- package/dist/cjs/segment.d.ts +0 -2
- package/dist/cjs/segment.js +0 -98
- package/dist/cjs/segment.js.map +0 -1
- package/dist/cjs/sql-functions.js.map +0 -1
- package/dist/cjs/traverse.js.map +0 -1
- package/dist/esm/column-hint.js.map +0 -1
- package/dist/esm/fences.js.map +0 -1
- package/dist/esm/graph.js.map +0 -1
- package/dist/esm/output.js.map +0 -1
- package/dist/esm/progress.js.map +0 -1
- package/dist/esm/scan.d.ts +0 -35
- package/dist/esm/scan.js.map +0 -1
- package/dist/esm/search-error.js.map +0 -1
- package/dist/esm/segment.d.ts +0 -2
- package/dist/esm/segment.js +0 -65
- package/dist/esm/segment.js.map +0 -1
- package/dist/esm/sql-functions.js.map +0 -1
- package/dist/esm/traverse.js.map +0 -1
- /package/dist/cjs/{sql-functions.d.cts → db/sql-functions.d.cts} +0 -0
- /package/dist/cjs/{sql-functions.d.ts → db/sql-functions.d.ts} +0 -0
- /package/dist/cjs/{fences.d.cts → features/fences.d.cts} +0 -0
- /package/dist/cjs/{fences.d.ts → features/fences.d.ts} +0 -0
- /package/dist/cjs/{graph.d.cts → graph/graph.d.cts} +0 -0
- /package/dist/cjs/{graph.d.ts → graph/graph.d.ts} +0 -0
- /package/dist/cjs/{graph.js → graph/graph.js} +0 -0
- /package/dist/cjs/{traverse.d.cts → graph/traverse.d.cts} +0 -0
- /package/dist/cjs/{traverse.d.ts → graph/traverse.d.ts} +0 -0
- /package/dist/cjs/{traverse.js → graph/traverse.js} +0 -0
- /package/dist/cjs/{column-hint.d.cts → output/column-hint.d.cts} +0 -0
- /package/dist/cjs/{column-hint.d.ts → output/column-hint.d.ts} +0 -0
- /package/dist/cjs/{column-hint.js → output/column-hint.js} +0 -0
- /package/dist/cjs/{progress.d.cts → output/progress.d.cts} +0 -0
- /package/dist/cjs/{progress.d.ts → output/progress.d.ts} +0 -0
- /package/dist/cjs/{progress.js → output/progress.js} +0 -0
- /package/dist/cjs/{search-error.d.cts → output/search-error.d.cts} +0 -0
- /package/dist/cjs/{search-error.d.ts → output/search-error.d.ts} +0 -0
- /package/dist/esm/{sql-functions.d.ts → db/sql-functions.d.ts} +0 -0
- /package/dist/esm/{fences.d.ts → features/fences.d.ts} +0 -0
- /package/dist/esm/{graph.d.ts → graph/graph.d.ts} +0 -0
- /package/dist/esm/{graph.js → graph/graph.js} +0 -0
- /package/dist/esm/{traverse.d.ts → graph/traverse.d.ts} +0 -0
- /package/dist/esm/{traverse.js → graph/traverse.js} +0 -0
- /package/dist/esm/{column-hint.d.ts → output/column-hint.d.ts} +0 -0
- /package/dist/esm/{column-hint.js → output/column-hint.js} +0 -0
- /package/dist/esm/{progress.d.ts → output/progress.d.ts} +0 -0
- /package/dist/esm/{progress.js → output/progress.js} +0 -0
- /package/dist/esm/{search-error.d.ts → output/search-error.d.ts} +0 -0
|
@@ -1,9 +1,4 @@
|
|
|
1
|
-
import { globSync, readFileSync, statSync } from 'node:fs';
|
|
2
|
-
import { join, sep } from 'node:path';
|
|
3
|
-
import removeMarkdown from 'remove-markdown';
|
|
4
1
|
import { isCollection, parseDocument, visit } from 'yaml';
|
|
5
|
-
import { embedEnabled, presetNames, presetSemanticEnabled } from './config/index.js';
|
|
6
|
-
// Filesystem -> rows. Pure data in, data + warnings out; db.ts does the SQL.
|
|
7
2
|
// Frontmatter keys that would collide with table columns. Exported so db.ts's upsert can tell
|
|
8
3
|
// a feature-owned column (`_rank`) from a parsed one and leave it alone on reparse.
|
|
9
4
|
export const RESERVED_COLUMNS = new Set([
|
|
@@ -26,73 +21,10 @@ export const RESERVED_COLUMNS = new Set([
|
|
|
26
21
|
const ACCEPTED_YAML_CODES = new Set([
|
|
27
22
|
'BAD_SCALAR_START'
|
|
28
23
|
]);
|
|
29
|
-
function normalizeText(value) {
|
|
24
|
+
export function normalizeText(value) {
|
|
30
25
|
if (value === null || value === undefined) return '';
|
|
31
26
|
return String(value).replace(/\s+/g, ' ').trim();
|
|
32
27
|
}
|
|
33
|
-
// Keeps URL query strings, asset filenames, and HTML attributes out of the index: rare terms
|
|
34
|
-
// carry high IDF, so they outrank prose. remove-markdown misses wikilinks and tables.
|
|
35
|
-
function stripText(value) {
|
|
36
|
-
const withoutWikilinks = value.replace(/\[\[([^\]|]+)\|([^\]]+)\]\]/g, '$2').replace(/\[\[([^\]]+)\]\]/g, '$1');
|
|
37
|
-
const withoutMarkdown = removeMarkdown(withoutWikilinks);
|
|
38
|
-
const withoutTables = withoutMarkdown.replace(/^\s*\|?[-\s|:]+\|\s*$/gm, '').replace(/\|/g, ' ');
|
|
39
|
-
return normalizeText(withoutTables);
|
|
40
|
-
}
|
|
41
|
-
// Presets are views, not partitions: they overlap freely, and a file's covering set (not one
|
|
42
|
-
// owner) drives indexing. Globs resolve relative to baseDir; unmatched files are not indexed.
|
|
43
|
-
export function toPosixPath(relPath, separator = sep) {
|
|
44
|
-
return separator === '\\' ? relPath.split(separator).join('/') : relPath;
|
|
45
|
-
}
|
|
46
|
-
// Every command pays listFiles before it answers (the freshness check stats each file), so
|
|
47
|
-
// per-file work here is the hottest path in the package. Everything derivable from the config
|
|
48
|
-
// alone is computed once, above the loop.
|
|
49
|
-
const NO_THROW = {
|
|
50
|
-
throwIfNoEntry: false
|
|
51
|
-
};
|
|
52
|
-
export function listFiles(cfg, baseDir) {
|
|
53
|
-
const coverage = new Map();
|
|
54
|
-
const posixNeeded = sep === '\\';
|
|
55
|
-
for (const name of presetNames(cfg)){
|
|
56
|
-
const preset = cfg.presets[name];
|
|
57
|
-
for (const matched of globSync(preset.include, {
|
|
58
|
-
cwd: baseDir,
|
|
59
|
-
exclude: preset.exclude
|
|
60
|
-
})){
|
|
61
|
-
var _coverage_get;
|
|
62
|
-
const relPath = posixNeeded ? toPosixPath(matched) : matched;
|
|
63
|
-
const set = (_coverage_get = coverage.get(relPath)) !== null && _coverage_get !== void 0 ? _coverage_get : new Set();
|
|
64
|
-
set.add(name);
|
|
65
|
-
coverage.set(relPath, set);
|
|
66
|
-
}
|
|
67
|
-
}
|
|
68
|
-
// Which presets want vectors is a property of the config, not of any file.
|
|
69
|
-
const embedding = embedEnabled(cfg);
|
|
70
|
-
const semanticPresets = embedding ? new Set(presetNames(cfg).filter((name)=>presetSemanticEnabled(cfg, name))) : null;
|
|
71
|
-
const files = [];
|
|
72
|
-
for (const relPath of [
|
|
73
|
-
...coverage.keys()
|
|
74
|
-
].sort()){
|
|
75
|
-
const absPath = join(baseDir, relPath); // join re-applies the platform separator for fs calls
|
|
76
|
-
// node:fs glob matches directories and dangling symlinks; fast-glob returned neither, so
|
|
77
|
-
// one stat filters both back out (throwIfNoEntry keeps a dangling link from throwing).
|
|
78
|
-
const st = statSync(absPath, NO_THROW);
|
|
79
|
-
if (!(st === null || st === void 0 ? void 0 : st.isFile())) continue;
|
|
80
|
-
const presets = [
|
|
81
|
-
...coverage.get(relPath)
|
|
82
|
-
].sort();
|
|
83
|
-
const embed = semanticPresets !== null && presets.some((name)=>semanticPresets.has(name));
|
|
84
|
-
files.push({
|
|
85
|
-
relPath,
|
|
86
|
-
absPath,
|
|
87
|
-
mtimeMs: st.mtimeMs,
|
|
88
|
-
ctimeMs: st.birthtimeMs,
|
|
89
|
-
size: st.size,
|
|
90
|
-
presets,
|
|
91
|
-
embed
|
|
92
|
-
});
|
|
93
|
-
}
|
|
94
|
-
return files;
|
|
95
|
-
}
|
|
96
28
|
// SQLite's datetime() rejects a colonless offset (`-0800`) and a space separator, which ISO 8601
|
|
97
29
|
// allows and producers emit. A rejected date is invisible, not excluded: every comparison is NULL.
|
|
98
30
|
const ISO_DATETIME = /^(\d{4}-\d{2}-\d{2})[T ](\d{2}:\d{2}(?::\d{2})?(?:\.\d+)?)(Z|[+-]\d{2}(?::?\d{2})?)?$/;
|
|
@@ -115,7 +47,7 @@ export function normalizeDate(value) {
|
|
|
115
47
|
}
|
|
116
48
|
// Storage class follows the YAML scalar. Booleans store as 1/0, so `WHERE flag = 1` matches
|
|
117
49
|
// and `WHERE flag = 'true'` cannot; `map` prints observed types so the mismatch is visible.
|
|
118
|
-
function mapValue(value) {
|
|
50
|
+
export function mapValue(value) {
|
|
119
51
|
if (value === null || value === undefined) return null;
|
|
120
52
|
if (typeof value === 'boolean') return BigInt(value ? 1 : 0);
|
|
121
53
|
if (typeof value === 'number') return Number.isSafeInteger(value) ? BigInt(value) : value;
|
|
@@ -123,7 +55,7 @@ function mapValue(value) {
|
|
|
123
55
|
return JSON.stringify(value);
|
|
124
56
|
}
|
|
125
57
|
// The delimiter split is all this package used gray-matter for.
|
|
126
|
-
function splitFrontmatter(raw) {
|
|
58
|
+
export function splitFrontmatter(raw) {
|
|
127
59
|
const open = raw.match(/^---\r?\n/);
|
|
128
60
|
if (!open) return {
|
|
129
61
|
fm: null,
|
|
@@ -169,7 +101,7 @@ function warnStringifiedKeys(relPath, doc, warnings) {
|
|
|
169
101
|
function firstLine(message) {
|
|
170
102
|
return message.split('\n')[0].replace(/:\s*$/, '');
|
|
171
103
|
}
|
|
172
|
-
function parseFrontmatter(relPath, fm, warnings) {
|
|
104
|
+
export function parseFrontmatter(relPath, fm, warnings) {
|
|
173
105
|
// logLevel silences yaml's own pathless warnings; warnStringifiedKeys re-reports the one
|
|
174
106
|
// that carries information, with the file it came from.
|
|
175
107
|
const doc = parseDocument(fm, {
|
|
@@ -216,51 +148,3 @@ function parseFrontmatter(relPath, fm, warnings) {
|
|
|
216
148
|
parseError: null
|
|
217
149
|
};
|
|
218
150
|
}
|
|
219
|
-
export function parseFile(file, extractors = []) {
|
|
220
|
-
const raw = readFileSync(file.absPath, 'utf8');
|
|
221
|
-
const warnings = [];
|
|
222
|
-
const { fm, body: content } = splitFrontmatter(raw);
|
|
223
|
-
const { data, parseError } = fm === null ? {
|
|
224
|
-
data: {},
|
|
225
|
-
parseError: null
|
|
226
|
-
} : parseFrontmatter(file.relPath, fm, warnings);
|
|
227
|
-
const mapped = {};
|
|
228
|
-
for (const key of Object.keys(data)){
|
|
229
|
-
if (RESERVED_COLUMNS.has(key)) {
|
|
230
|
-
warnings.push(`warning: ${file.relPath} has a frontmatter key named "${key}", which is reserved; ignoring it`);
|
|
231
|
-
continue;
|
|
232
|
-
}
|
|
233
|
-
const value = mapValue(data[key]);
|
|
234
|
-
if (typeof value === 'string' && looksLikeDatetime(value) && Number.isNaN(Date.parse(value))) {
|
|
235
|
-
warnings.push(`warning: ${file.relPath}: ${key} is not a valid date (${value}), so it is invisible to every date comparison`);
|
|
236
|
-
}
|
|
237
|
-
mapped[key] = value;
|
|
238
|
-
}
|
|
239
|
-
// title/summary are plain YAML strings -- whitespace-collapse only;
|
|
240
|
-
// the prose gets the full markdown strip.
|
|
241
|
-
const search = {
|
|
242
|
-
title: normalizeText(data.title),
|
|
243
|
-
summary: normalizeText(data.summary),
|
|
244
|
-
text: stripText(content)
|
|
245
|
-
};
|
|
246
|
-
return {
|
|
247
|
-
doc: {
|
|
248
|
-
relPath: file.relPath,
|
|
249
|
-
mtimeMs: file.mtimeMs,
|
|
250
|
-
ctimeMs: file.ctimeMs,
|
|
251
|
-
size: file.size,
|
|
252
|
-
presets: file.presets,
|
|
253
|
-
data: mapped,
|
|
254
|
-
parseError,
|
|
255
|
-
search,
|
|
256
|
-
extracted: Object.fromEntries(extractors.filter((f)=>f.extract).map((f)=>{
|
|
257
|
-
var _f_extract;
|
|
258
|
-
return [
|
|
259
|
-
f.name,
|
|
260
|
-
(_f_extract = f.extract) === null || _f_extract === void 0 ? void 0 : _f_extract.call(f, raw, content, search, data)
|
|
261
|
-
];
|
|
262
|
-
}))
|
|
263
|
-
},
|
|
264
|
-
warnings
|
|
265
|
-
};
|
|
266
|
-
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/scan/frontmatter.ts"],"sourcesContent":["import { isCollection, parseDocument, visit } from 'yaml';\n\n// Frontmatter keys that would collide with table columns. Exported so db.ts's upsert can tell\n// a feature-owned column (`_rank`) from a parsed one and leave it alone on reparse.\nexport const RESERVED_COLUMNS = new Set(['path', '_mtime', '_ctime', '_size', '_rank', '_parse_error', 'content', 'links', 'sections']);\n\n// YAML error codes whose recovery is unambiguous, so the parse is accepted rather than\n// quarantined. Only one qualifies: YAML 1.2 reserves `@` and `` ` `` at the start of a plain\n// scalar for future use, so they can never be valid and the text can only be what was typed\n// (`aliases: [@handle]` -> [\"@handle\"]). Every other code has a second reading -- an unquoted\n// `:` swallows the keys after it, an unquoted `[..](..)` drops the URL, a duplicate key picks\n// one value in silence -- so it writes values nobody wrote.\nconst ACCEPTED_YAML_CODES = new Set(['BAD_SCALAR_START']);\n\nexport function normalizeText(value: unknown): string {\n if (value === null || value === undefined) return '';\n return String(value).replace(/\\s+/g, ' ').trim();\n}\n\n// SQLite's datetime() rejects a colonless offset (`-0800`) and a space separator, which ISO 8601\n// allows and producers emit. A rejected date is invisible, not excluded: every comparison is NULL.\nconst ISO_DATETIME = /^(\\d{4}-\\d{2}-\\d{2})[T ](\\d{2}:\\d{2}(?::\\d{2})?(?:\\.\\d+)?)(Z|[+-]\\d{2}(?::?\\d{2})?)?$/;\n\n// A value opening `YYYY-MM-DDT` was meant to be a datetime; prose never is. Reported when it\n// cannot be normalized, so a typo surfaces on the next crawl instead of at some later audit.\nconst MEANT_AS_DATETIME = /^\\d{4}-\\d{2}-\\d{2}[T ]\\d/;\n\nexport function looksLikeDatetime(value: string): boolean {\n return MEANT_AS_DATETIME.test(value);\n}\n\n// Punctuation only, never a timezone conversion: the offset survives, so substr(d,1,10) is still\n// the local date. A shape that is not a real instant is left as written, and stays auditable.\nexport function normalizeDate(value: string): string {\n const m = ISO_DATETIME.exec(value);\n if (m === null) return value;\n const [, date, time, zone] = m;\n const digits = zone === undefined || zone === 'Z' ? '' : zone.replace(':', '');\n const offset = digits === '' ? (zone ?? '') : digits.length === 3 ? `${digits}:00` : `${digits.slice(0, 3)}:${digits.slice(3)}`;\n const normalized = `${date}T${time}${offset}`;\n return Number.isNaN(Date.parse(normalized)) ? value : normalized;\n}\n\n// Storage class follows the YAML scalar. Booleans store as 1/0, so `WHERE flag = 1` matches\n// and `WHERE flag = 'true'` cannot; `map` prints observed types so the mismatch is visible.\nexport function mapValue(value: unknown): string | number | bigint | null {\n if (value === null || value === undefined) return null;\n if (typeof value === 'boolean') return BigInt(value ? 1 : 0);\n if (typeof value === 'number') return Number.isSafeInteger(value) ? BigInt(value) : value;\n if (typeof value === 'string') return normalizeDate(value);\n return JSON.stringify(value);\n}\n\n// The delimiter split is all this package used gray-matter for.\nexport function splitFrontmatter(raw: string): { fm: string | null; body: string } {\n const open = raw.match(/^---\\r?\\n/);\n if (!open) return { fm: null, body: raw };\n const rest = raw.slice(open[0].length);\n const close = rest.match(/^---\\r?(\\n|$)/m);\n if (!close || close.index === undefined) return { fm: null, body: raw };\n return { fm: rest.slice(0, close.index), body: rest.slice(close.index + close[0].length) };\n}\n\n// A well-formed document can still hold a value nobody meant: `created: {{date}}` is valid\n// YAML for a flow map used as a mapping key, so it raises no error and stores\n// {\"{ date }\": null}. No error code can catch that, but yaml notices the stringified key, so\n// this reports it with the path instead (yaml's own warning has none, fires once per document,\n// and is what trains readers to discard stderr).\nfunction warnStringifiedKeys(relPath: string, doc: ReturnType<typeof parseDocument>, warnings: string[]): void {\n let found = false;\n // Nested, not top level: `created: {{date}}` puts the collection key one level down, inside\n // the flow map that `{{...}}` parses as.\n visit(doc, {\n Pair(_key, pair) {\n if (!isCollection(pair.key)) return undefined;\n found = true;\n return visit.BREAK;\n },\n });\n // One per file: a template repeats the same mistake on every field it stamps.\n if (found) warnings.push(`warning: ${relPath} frontmatter has a key that is itself a list or mapping, stored as text; this is usually an unrendered template placeholder like {{date}}`);\n}\n\n// Accept a clean parse, and one whose every error is unambiguous (ACCEPTED_YAML_CODES).\n// Anything else is quarantined: no frontmatter columns at all, and `_parse_error` carries the\n// reason. Recovering it would write values nobody wrote, which is worse than absence because\n// no query can see it. The file is still indexed -- content, links and sections never touch\n// frontmatter -- so a broken note stays searchable while it is being hunted for.\n// yaml's message continues onto a source excerpt, so the first line is the sentence -- minus\n// the colon that introduced the part being dropped.\nfunction firstLine(message: string): string {\n return message.split('\\n')[0].replace(/:\\s*$/, '');\n}\n\nexport function parseFrontmatter(relPath: string, fm: string, warnings: string[]): { data: Record<string, unknown>; parseError: string | null } {\n // logLevel silences yaml's own pathless warnings; warnStringifiedKeys re-reports the one\n // that carries information, with the file it came from.\n const doc = parseDocument(fm, { logLevel: 'silent' });\n const refused = doc.errors.filter((err) => !ACCEPTED_YAML_CODES.has(err.code));\n if (refused.length > 0) {\n const detail = refused.length > 1 ? ` (and ${refused.length - 1} more)` : '';\n const parseError = `${firstLine(refused[0].message)}${detail}`;\n warnings.push(`warning: ${relPath} frontmatter did not parse, so none of it is indexed: ${parseError}`);\n return { data: {}, parseError };\n }\n\n let data: unknown;\n try {\n data = doc.toJS();\n } catch (err) {\n // Reaches here with doc.errors empty: `title: **Bold**` parses, then opens an alias on\n // materialisation. An empty error list is not a successful parse.\n const parseError = firstLine((err as Error).message);\n warnings.push(`warning: ${relPath} frontmatter did not parse, so none of it is indexed: ${parseError}`);\n return { data: {}, parseError };\n }\n\n if (data === null || data === undefined) return { data: {}, parseError: null };\n if (typeof data !== 'object' || Array.isArray(data)) {\n const parseError = 'frontmatter is not a key-value mapping';\n warnings.push(`warning: ${relPath} ${parseError}; none of it is indexed`);\n return { data: {}, parseError };\n }\n warnStringifiedKeys(relPath, doc, warnings);\n return { data: data as Record<string, unknown>, parseError: null };\n}\n"],"names":["isCollection","parseDocument","visit","RESERVED_COLUMNS","Set","ACCEPTED_YAML_CODES","normalizeText","value","undefined","String","replace","trim","ISO_DATETIME","MEANT_AS_DATETIME","looksLikeDatetime","test","normalizeDate","m","exec","date","time","zone","digits","offset","length","slice","normalized","Number","isNaN","Date","parse","mapValue","BigInt","isSafeInteger","JSON","stringify","splitFrontmatter","raw","open","match","fm","body","rest","close","index","warnStringifiedKeys","relPath","doc","warnings","found","Pair","_key","pair","key","BREAK","push","firstLine","message","split","parseFrontmatter","logLevel","refused","errors","filter","err","has","code","detail","parseError","data","toJS","Array","isArray"],"mappings":"AAAA,SAASA,YAAY,EAAEC,aAAa,EAAEC,KAAK,QAAQ,OAAO;AAE1D,8FAA8F;AAC9F,oFAAoF;AACpF,OAAO,MAAMC,mBAAmB,IAAIC,IAAI;IAAC;IAAQ;IAAU;IAAU;IAAS;IAAS;IAAgB;IAAW;IAAS;CAAW,EAAE;AAExI,uFAAuF;AACvF,6FAA6F;AAC7F,4FAA4F;AAC5F,8FAA8F;AAC9F,8FAA8F;AAC9F,4DAA4D;AAC5D,MAAMC,sBAAsB,IAAID,IAAI;IAAC;CAAmB;AAExD,OAAO,SAASE,cAAcC,KAAc;IAC1C,IAAIA,UAAU,QAAQA,UAAUC,WAAW,OAAO;IAClD,OAAOC,OAAOF,OAAOG,OAAO,CAAC,QAAQ,KAAKC,IAAI;AAChD;AAEA,iGAAiG;AACjG,mGAAmG;AACnG,MAAMC,eAAe;AAErB,6FAA6F;AAC7F,6FAA6F;AAC7F,MAAMC,oBAAoB;AAE1B,OAAO,SAASC,kBAAkBP,KAAa;IAC7C,OAAOM,kBAAkBE,IAAI,CAACR;AAChC;AAEA,iGAAiG;AACjG,8FAA8F;AAC9F,OAAO,SAASS,cAAcT,KAAa;IACzC,MAAMU,IAAIL,aAAaM,IAAI,CAACX;IAC5B,IAAIU,MAAM,MAAM,OAAOV;IACvB,MAAM,GAAGY,MAAMC,MAAMC,KAAK,GAAGJ;IAC7B,MAAMK,SAASD,SAASb,aAAaa,SAAS,MAAM,KAAKA,KAAKX,OAAO,CAAC,KAAK;IAC3E,MAAMa,SAASD,WAAW,KAAMD,iBAAAA,kBAAAA,OAAQ,KAAMC,OAAOE,MAAM,KAAK,IAAI,GAAGF,OAAO,GAAG,CAAC,GAAG,GAAGA,OAAOG,KAAK,CAAC,GAAG,GAAG,CAAC,EAAEH,OAAOG,KAAK,CAAC,IAAI;IAC/H,MAAMC,aAAa,GAAGP,KAAK,CAAC,EAAEC,OAAOG,QAAQ;IAC7C,OAAOI,OAAOC,KAAK,CAACC,KAAKC,KAAK,CAACJ,eAAenB,QAAQmB;AACxD;AAEA,4FAA4F;AAC5F,4FAA4F;AAC5F,OAAO,SAASK,SAASxB,KAAc;IACrC,IAAIA,UAAU,QAAQA,UAAUC,WAAW,OAAO;IAClD,IAAI,OAAOD,UAAU,WAAW,OAAOyB,OAAOzB,QAAQ,IAAI;IAC1D,IAAI,OAAOA,UAAU,UAAU,OAAOoB,OAAOM,aAAa,CAAC1B,SAASyB,OAAOzB,SAASA;IACpF,IAAI,OAAOA,UAAU,UAAU,OAAOS,cAAcT;IACpD,OAAO2B,KAAKC,SAAS,CAAC5B;AACxB;AAEA,gEAAgE;AAChE,OAAO,SAAS6B,iBAAiBC,GAAW;IAC1C,MAAMC,OAAOD,IAAIE,KAAK,CAAC;IACvB,IAAI,CAACD,MAAM,OAAO;QAAEE,IAAI;QAAMC,MAAMJ;IAAI;IACxC,MAAMK,OAAOL,IAAIZ,KAAK,CAACa,IAAI,CAAC,EAAE,CAACd,MAAM;IACrC,MAAMmB,QAAQD,KAAKH,KAAK,CAAC;IACzB,IAAI,CAACI,SAASA,MAAMC,KAAK,KAAKpC,WAAW,OAAO;QAAEgC,IAAI;QAAMC,MAAMJ;IAAI;IACtE,OAAO;QAAEG,IAAIE,KAAKjB,KAAK,CAAC,GAAGkB,MAAMC,KAAK;QAAGH,MAAMC,KAAKjB,KAAK,CAACkB,MAAMC,KAAK,GAAGD,KAAK,CAAC,EAAE,CAACnB,MAAM;IAAE;AAC3F;AAEA,2FAA2F;AAC3F,8EAA8E;AAC9E,6FAA6F;AAC7F,+FAA+F;AAC/F,iDAAiD;AACjD,SAASqB,oBAAoBC,OAAe,EAAEC,GAAqC,EAAEC,QAAkB;IACrG,IAAIC,QAAQ;IACZ,4FAA4F;IAC5F,yCAAyC;IACzC/C,MAAM6C,KAAK;QACTG,MAAKC,IAAI,EAAEC,IAAI;YACb,IAAI,CAACpD,aAAaoD,KAAKC,GAAG,GAAG,OAAO7C;YACpCyC,QAAQ;YACR,OAAO/C,MAAMoD,KAAK;QACpB;IACF;IACA,8EAA8E;IAC9E,IAAIL,OAAOD,SAASO,IAAI,CAAC,CAAC,SAAS,EAAET,QAAQ,yIAAyI,CAAC;AACzL;AAEA,wFAAwF;AACxF,8FAA8F;AAC9F,6FAA6F;AAC7F,4FAA4F;AAC5F,iFAAiF;AACjF,6FAA6F;AAC7F,oDAAoD;AACpD,SAASU,UAAUC,OAAe;IAChC,OAAOA,QAAQC,KAAK,CAAC,KAAK,CAAC,EAAE,CAAChD,OAAO,CAAC,SAAS;AACjD;AAEA,OAAO,SAASiD,iBAAiBb,OAAe,EAAEN,EAAU,EAAEQ,QAAkB;IAC9E,yFAAyF;IACzF,wDAAwD;IACxD,MAAMD,MAAM9C,cAAcuC,IAAI;QAAEoB,UAAU;IAAS;IACnD,MAAMC,UAAUd,IAAIe,MAAM,CAACC,MAAM,CAAC,CAACC,MAAQ,CAAC3D,oBAAoB4D,GAAG,CAACD,IAAIE,IAAI;IAC5E,IAAIL,QAAQrC,MAAM,GAAG,GAAG;QACtB,MAAM2C,SAASN,QAAQrC,MAAM,GAAG,IAAI,CAAC,MAAM,EAAEqC,QAAQrC,MAAM,GAAG,EAAE,MAAM,CAAC,GAAG;QAC1E,MAAM4C,aAAa,GAAGZ,UAAUK,OAAO,CAAC,EAAE,CAACJ,OAAO,IAAIU,QAAQ;QAC9DnB,SAASO,IAAI,CAAC,CAAC,SAAS,EAAET,QAAQ,sDAAsD,EAAEsB,YAAY;QACtG,OAAO;YAAEC,MAAM,CAAC;YAAGD;QAAW;IAChC;IAEA,IAAIC;IACJ,IAAI;QACFA,OAAOtB,IAAIuB,IAAI;IACjB,EAAE,OAAON,KAAK;QACZ,uFAAuF;QACvF,kEAAkE;QAClE,MAAMI,aAAaZ,UAAU,AAACQ,IAAcP,OAAO;QACnDT,SAASO,IAAI,CAAC,CAAC,SAAS,EAAET,QAAQ,sDAAsD,EAAEsB,YAAY;QACtG,OAAO;YAAEC,MAAM,CAAC;YAAGD;QAAW;IAChC;IAEA,IAAIC,SAAS,QAAQA,SAAS7D,WAAW,OAAO;QAAE6D,MAAM,CAAC;QAAGD,YAAY;IAAK;IAC7E,IAAI,OAAOC,SAAS,YAAYE,MAAMC,OAAO,CAACH,OAAO;QACnD,MAAMD,aAAa;QACnBpB,SAASO,IAAI,CAAC,CAAC,SAAS,EAAET,QAAQ,CAAC,EAAEsB,WAAW,uBAAuB,CAAC;QACxE,OAAO;YAAEC,MAAM,CAAC;YAAGD;QAAW;IAChC;IACAvB,oBAAoBC,SAASC,KAAKC;IAClC,OAAO;QAAEqB,MAAMA;QAAiCD,YAAY;IAAK;AACnE"}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import type { Config } from '../config/index.js';
|
|
2
|
+
import type { Feature } from '../features/types.js';
|
|
3
|
+
import type { FileStat } from './list.js';
|
|
4
|
+
export { looksLikeDatetime, normalizeDate, RESERVED_COLUMNS } from './frontmatter.js';
|
|
5
|
+
export type { FileStat } from './list.js';
|
|
6
|
+
export { listFiles, toPosixPath } from './list.js';
|
|
7
|
+
export interface ParsedDoc {
|
|
8
|
+
relPath: string;
|
|
9
|
+
mtimeMs: number;
|
|
10
|
+
ctimeMs: number;
|
|
11
|
+
size: number;
|
|
12
|
+
presets: string[];
|
|
13
|
+
data: Record<string, string | number | bigint | null>;
|
|
14
|
+
parseError: string | null;
|
|
15
|
+
search: {
|
|
16
|
+
title: string;
|
|
17
|
+
summary: string;
|
|
18
|
+
text: string;
|
|
19
|
+
};
|
|
20
|
+
extracted: Record<string, unknown>;
|
|
21
|
+
}
|
|
22
|
+
export declare function parseFile(file: FileStat, extractors?: Feature[], cfg?: Config): {
|
|
23
|
+
doc: ParsedDoc;
|
|
24
|
+
warnings: string[];
|
|
25
|
+
};
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { readFileSync } from 'node:fs';
|
|
2
|
+
import { parse } from '../chunk/index.js';
|
|
3
|
+
import { textFromBlocks } from '../text/strip.js';
|
|
4
|
+
import { looksLikeDatetime, mapValue, normalizeText, parseFrontmatter, RESERVED_COLUMNS, splitFrontmatter } from './frontmatter.js';
|
|
5
|
+
// Filesystem -> rows. Pure data in, data + warnings out; db.ts does the SQL.
|
|
6
|
+
export { looksLikeDatetime, normalizeDate, RESERVED_COLUMNS } from './frontmatter.js';
|
|
7
|
+
export { listFiles, toPosixPath } from './list.js';
|
|
8
|
+
export function parseFile(file, extractors = [], cfg) {
|
|
9
|
+
const raw = readFileSync(file.absPath, 'utf8');
|
|
10
|
+
const warnings = [];
|
|
11
|
+
const { fm, body: content } = splitFrontmatter(raw);
|
|
12
|
+
const { data, parseError } = fm === null ? {
|
|
13
|
+
data: {},
|
|
14
|
+
parseError: null
|
|
15
|
+
} : parseFrontmatter(file.relPath, fm, warnings);
|
|
16
|
+
const mapped = {};
|
|
17
|
+
for (const key of Object.keys(data)){
|
|
18
|
+
if (RESERVED_COLUMNS.has(key)) {
|
|
19
|
+
warnings.push(`warning: ${file.relPath} has a frontmatter key named "${key}", which is reserved; ignoring it`);
|
|
20
|
+
continue;
|
|
21
|
+
}
|
|
22
|
+
const value = mapValue(data[key]);
|
|
23
|
+
if (typeof value === 'string' && looksLikeDatetime(value) && Number.isNaN(Date.parse(value))) {
|
|
24
|
+
warnings.push(`warning: ${file.relPath}: ${key} is not a valid date (${value}), so it is invisible to every date comparison`);
|
|
25
|
+
}
|
|
26
|
+
mapped[key] = value;
|
|
27
|
+
}
|
|
28
|
+
// Parsed once and shared: the FTS text path (textFromBlocks, ~ strip.ts) and any feature
|
|
29
|
+
// that needs a parse tree (embed's chunker) both read this same block list.
|
|
30
|
+
const blocks = parse(content);
|
|
31
|
+
// title/summary are plain YAML strings -- whitespace-collapse only;
|
|
32
|
+
// the prose gets the full markdown strip.
|
|
33
|
+
const search = {
|
|
34
|
+
title: normalizeText(data.title),
|
|
35
|
+
summary: normalizeText(data.summary),
|
|
36
|
+
text: textFromBlocks(blocks)
|
|
37
|
+
};
|
|
38
|
+
return {
|
|
39
|
+
doc: {
|
|
40
|
+
relPath: file.relPath,
|
|
41
|
+
mtimeMs: file.mtimeMs,
|
|
42
|
+
ctimeMs: file.ctimeMs,
|
|
43
|
+
size: file.size,
|
|
44
|
+
presets: file.presets,
|
|
45
|
+
data: mapped,
|
|
46
|
+
parseError,
|
|
47
|
+
search,
|
|
48
|
+
extracted: Object.fromEntries(extractors.filter((f)=>f.extract).map((f)=>{
|
|
49
|
+
var _f_extract;
|
|
50
|
+
return [
|
|
51
|
+
f.name,
|
|
52
|
+
(_f_extract = f.extract) === null || _f_extract === void 0 ? void 0 : _f_extract.call(f, raw, content, search, data, cfg, blocks)
|
|
53
|
+
];
|
|
54
|
+
}))
|
|
55
|
+
},
|
|
56
|
+
warnings
|
|
57
|
+
};
|
|
58
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/scan/index.ts"],"sourcesContent":["import { readFileSync } from 'node:fs';\nimport { parse } from '../chunk/index.ts';\nimport type { Config } from '../config/index.ts';\nimport type { Feature } from '../features/types.ts';\nimport { textFromBlocks } from '../text/strip.ts';\nimport { looksLikeDatetime, mapValue, normalizeText, parseFrontmatter, RESERVED_COLUMNS, splitFrontmatter } from './frontmatter.ts';\nimport type { FileStat } from './list.ts';\n\n// Filesystem -> rows. Pure data in, data + warnings out; db.ts does the SQL.\n\nexport { looksLikeDatetime, normalizeDate, RESERVED_COLUMNS } from './frontmatter.ts';\nexport type { FileStat } from './list.ts';\nexport { listFiles, toPosixPath } from './list.ts';\n\nexport interface ParsedDoc {\n relPath: string;\n mtimeMs: number;\n ctimeMs: number;\n size: number;\n presets: string[];\n data: Record<string, string | number | bigint | null>;\n // NULL when the frontmatter parsed, the first YAML message otherwise. In the row rather than\n // a side table so `SELECT *` and any `IS NULL` investigation trip over it without being asked.\n parseError: string | null;\n // title/summary are duplicated from frontmatter so bm25() can weight them above the body text.\n search: { title: string; summary: string; text: string };\n // Per-feature extraction results, keyed by feature name; features store them at reconcile.\n extracted: Record<string, unknown>;\n}\n\nexport function parseFile(file: FileStat, extractors: Feature[] = [], cfg?: Config): { doc: ParsedDoc; warnings: string[] } {\n const raw = readFileSync(file.absPath, 'utf8');\n const warnings: string[] = [];\n\n const { fm, body: content } = splitFrontmatter(raw);\n const { data, parseError } = fm === null ? { data: {} as Record<string, unknown>, parseError: null } : parseFrontmatter(file.relPath, fm, warnings);\n const mapped: Record<string, string | number | bigint | null> = {};\n\n for (const key of Object.keys(data)) {\n if (RESERVED_COLUMNS.has(key)) {\n warnings.push(`warning: ${file.relPath} has a frontmatter key named \"${key}\", which is reserved; ignoring it`);\n continue;\n }\n const value = mapValue(data[key]);\n if (typeof value === 'string' && looksLikeDatetime(value) && Number.isNaN(Date.parse(value))) {\n warnings.push(`warning: ${file.relPath}: ${key} is not a valid date (${value}), so it is invisible to every date comparison`);\n }\n mapped[key] = value;\n }\n\n // Parsed once and shared: the FTS text path (textFromBlocks, ~ strip.ts) and any feature\n // that needs a parse tree (embed's chunker) both read this same block list.\n const blocks = parse(content);\n\n // title/summary are plain YAML strings -- whitespace-collapse only;\n // the prose gets the full markdown strip.\n const search = { title: normalizeText(data.title), summary: normalizeText(data.summary), text: textFromBlocks(blocks) };\n\n return {\n doc: {\n relPath: file.relPath,\n mtimeMs: file.mtimeMs,\n ctimeMs: file.ctimeMs,\n size: file.size,\n presets: file.presets,\n data: mapped,\n parseError,\n search,\n extracted: Object.fromEntries(extractors.filter((f) => f.extract).map((f) => [f.name, f.extract?.(raw, content, search, data, cfg, blocks)])),\n },\n warnings,\n };\n}\n"],"names":["readFileSync","parse","textFromBlocks","looksLikeDatetime","mapValue","normalizeText","parseFrontmatter","RESERVED_COLUMNS","splitFrontmatter","normalizeDate","listFiles","toPosixPath","parseFile","file","extractors","cfg","raw","absPath","warnings","fm","body","content","data","parseError","relPath","mapped","key","Object","keys","has","push","value","Number","isNaN","Date","blocks","search","title","summary","text","doc","mtimeMs","ctimeMs","size","presets","extracted","fromEntries","filter","f","extract","map","name"],"mappings":"AAAA,SAASA,YAAY,QAAQ,UAAU;AACvC,SAASC,KAAK,QAAQ,oBAAoB;AAG1C,SAASC,cAAc,QAAQ,mBAAmB;AAClD,SAASC,iBAAiB,EAAEC,QAAQ,EAAEC,aAAa,EAAEC,gBAAgB,EAAEC,gBAAgB,EAAEC,gBAAgB,QAAQ,mBAAmB;AAGpI,6EAA6E;AAE7E,SAASL,iBAAiB,EAAEM,aAAa,EAAEF,gBAAgB,QAAQ,mBAAmB;AAEtF,SAASG,SAAS,EAAEC,WAAW,QAAQ,YAAY;AAkBnD,OAAO,SAASC,UAAUC,IAAc,EAAEC,aAAwB,EAAE,EAAEC,GAAY;IAChF,MAAMC,MAAMhB,aAAaa,KAAKI,OAAO,EAAE;IACvC,MAAMC,WAAqB,EAAE;IAE7B,MAAM,EAAEC,EAAE,EAAEC,MAAMC,OAAO,EAAE,GAAGb,iBAAiBQ;IAC/C,MAAM,EAAEM,IAAI,EAAEC,UAAU,EAAE,GAAGJ,OAAO,OAAO;QAAEG,MAAM,CAAC;QAA8BC,YAAY;IAAK,IAAIjB,iBAAiBO,KAAKW,OAAO,EAAEL,IAAID;IAC1I,MAAMO,SAA0D,CAAC;IAEjE,KAAK,MAAMC,OAAOC,OAAOC,IAAI,CAACN,MAAO;QACnC,IAAIf,iBAAiBsB,GAAG,CAACH,MAAM;YAC7BR,SAASY,IAAI,CAAC,CAAC,SAAS,EAAEjB,KAAKW,OAAO,CAAC,8BAA8B,EAAEE,IAAI,iCAAiC,CAAC;YAC7G;QACF;QACA,MAAMK,QAAQ3B,SAASkB,IAAI,CAACI,IAAI;QAChC,IAAI,OAAOK,UAAU,YAAY5B,kBAAkB4B,UAAUC,OAAOC,KAAK,CAACC,KAAKjC,KAAK,CAAC8B,SAAS;YAC5Fb,SAASY,IAAI,CAAC,CAAC,SAAS,EAAEjB,KAAKW,OAAO,CAAC,EAAE,EAAEE,IAAI,sBAAsB,EAAEK,MAAM,8CAA8C,CAAC;QAC9H;QACAN,MAAM,CAACC,IAAI,GAAGK;IAChB;IAEA,yFAAyF;IACzF,4EAA4E;IAC5E,MAAMI,SAASlC,MAAMoB;IAErB,oEAAoE;IACpE,0CAA0C;IAC1C,MAAMe,SAAS;QAAEC,OAAOhC,cAAciB,KAAKe,KAAK;QAAGC,SAASjC,cAAciB,KAAKgB,OAAO;QAAGC,MAAMrC,eAAeiC;IAAQ;IAEtH,OAAO;QACLK,KAAK;YACHhB,SAASX,KAAKW,OAAO;YACrBiB,SAAS5B,KAAK4B,OAAO;YACrBC,SAAS7B,KAAK6B,OAAO;YACrBC,MAAM9B,KAAK8B,IAAI;YACfC,SAAS/B,KAAK+B,OAAO;YACrBtB,MAAMG;YACNF;YACAa;YACAS,WAAWlB,OAAOmB,WAAW,CAAChC,WAAWiC,MAAM,CAAC,CAACC,IAAMA,EAAEC,OAAO,EAAEC,GAAG,CAAC,CAACF;oBAAeA;uBAAT;oBAACA,EAAEG,IAAI;qBAAEH,aAAAA,EAAEC,OAAO,cAATD,iCAAAA,gBAAAA,GAAYhC,KAAKK,SAASe,QAAQd,MAAMP,KAAKoB;iBAAQ;;QAC7I;QACAjB;IACF;AACF"}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { Config } from '../config/index.js';
|
|
2
|
+
export interface FileStat {
|
|
3
|
+
relPath: string;
|
|
4
|
+
absPath: string;
|
|
5
|
+
mtimeMs: number;
|
|
6
|
+
ctimeMs: number;
|
|
7
|
+
size: number;
|
|
8
|
+
presets: string[];
|
|
9
|
+
embed: boolean;
|
|
10
|
+
}
|
|
11
|
+
export declare function toPosixPath(relPath: string, separator?: string): string;
|
|
12
|
+
export declare function listFiles(cfg: Config, baseDir: string): FileStat[];
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
import { globSync, statSync } from 'node:fs';
|
|
2
|
+
import { join, sep } from 'node:path';
|
|
3
|
+
import { embedEnabled, presetHasSignal, presetNames } from '../config/index.js';
|
|
4
|
+
// Presets are views, not partitions: they overlap freely, and a file's covering set (not one
|
|
5
|
+
// owner) drives indexing. Globs resolve relative to baseDir; unmatched files are not indexed.
|
|
6
|
+
export function toPosixPath(relPath, separator = sep) {
|
|
7
|
+
return separator === '\\' ? relPath.split(separator).join('/') : relPath;
|
|
8
|
+
}
|
|
9
|
+
// Every command pays listFiles before it answers (the freshness check stats each file), so
|
|
10
|
+
// per-file work here is the hottest path in the package. Everything derivable from the config
|
|
11
|
+
// alone is computed once, above the loop.
|
|
12
|
+
const NO_THROW = {
|
|
13
|
+
throwIfNoEntry: false
|
|
14
|
+
};
|
|
15
|
+
export function listFiles(cfg, baseDir) {
|
|
16
|
+
const coverage = new Map();
|
|
17
|
+
const posixNeeded = sep === '\\';
|
|
18
|
+
for (const name of presetNames(cfg)){
|
|
19
|
+
const preset = cfg.presets[name];
|
|
20
|
+
for (const matched of globSync(preset.include, {
|
|
21
|
+
cwd: baseDir,
|
|
22
|
+
exclude: preset.exclude
|
|
23
|
+
})){
|
|
24
|
+
var _coverage_get;
|
|
25
|
+
const relPath = posixNeeded ? toPosixPath(matched) : matched;
|
|
26
|
+
const set = (_coverage_get = coverage.get(relPath)) !== null && _coverage_get !== void 0 ? _coverage_get : new Set();
|
|
27
|
+
set.add(name);
|
|
28
|
+
coverage.set(relPath, set);
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
// Which presets want vectors is a property of the config, not of any file.
|
|
32
|
+
const embedding = embedEnabled(cfg);
|
|
33
|
+
const vectorPresets = embedding ? new Set(presetNames(cfg).filter((name)=>presetHasSignal(cfg, name, 'vectors'))) : null;
|
|
34
|
+
const files = [];
|
|
35
|
+
for (const relPath of [
|
|
36
|
+
...coverage.keys()
|
|
37
|
+
].sort()){
|
|
38
|
+
const absPath = join(baseDir, relPath); // join re-applies the platform separator for fs calls
|
|
39
|
+
// node:fs glob matches directories and dangling symlinks; fast-glob returned neither, so
|
|
40
|
+
// one stat filters both back out (throwIfNoEntry keeps a dangling link from throwing).
|
|
41
|
+
const st = statSync(absPath, NO_THROW);
|
|
42
|
+
if (!(st === null || st === void 0 ? void 0 : st.isFile())) continue;
|
|
43
|
+
const presets = [
|
|
44
|
+
...coverage.get(relPath)
|
|
45
|
+
].sort();
|
|
46
|
+
const embed = vectorPresets !== null && presets.some((name)=>vectorPresets.has(name));
|
|
47
|
+
files.push({
|
|
48
|
+
relPath,
|
|
49
|
+
absPath,
|
|
50
|
+
mtimeMs: st.mtimeMs,
|
|
51
|
+
ctimeMs: st.birthtimeMs,
|
|
52
|
+
size: st.size,
|
|
53
|
+
presets,
|
|
54
|
+
embed
|
|
55
|
+
});
|
|
56
|
+
}
|
|
57
|
+
return files;
|
|
58
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/scan/list.ts"],"sourcesContent":["import { globSync, statSync } from 'node:fs';\nimport { join, sep } from 'node:path';\nimport type { Config } from '../config/index.ts';\nimport { embedEnabled, presetHasSignal, presetNames } from '../config/index.ts';\n\nexport interface FileStat {\n relPath: string;\n absPath: string;\n mtimeMs: number;\n ctimeMs: number; // filesystem birthtime; a clone or copy resets it, like _mtime\n size: number;\n presets: string[]; // every declared preset covering this file (>= 1; union, overlap allowed)\n embed: boolean; // true iff a model is named and some covering preset's signals include vectors\n}\n\n// Presets are views, not partitions: they overlap freely, and a file's covering set (not one\n// owner) drives indexing. Globs resolve relative to baseDir; unmatched files are not indexed.\nexport function toPosixPath(relPath: string, separator: string = sep): string {\n return separator === '\\\\' ? relPath.split(separator).join('/') : relPath;\n}\n\n// Every command pays listFiles before it answers (the freshness check stats each file), so\n// per-file work here is the hottest path in the package. Everything derivable from the config\n// alone is computed once, above the loop.\nconst NO_THROW = { throwIfNoEntry: false } as const;\n\nexport function listFiles(cfg: Config, baseDir: string): FileStat[] {\n const coverage = new Map<string, Set<string>>();\n const posixNeeded = sep === '\\\\';\n for (const name of presetNames(cfg)) {\n const preset = cfg.presets[name];\n for (const matched of globSync(preset.include, { cwd: baseDir, exclude: preset.exclude })) {\n const relPath = posixNeeded ? toPosixPath(matched) : matched;\n const set = coverage.get(relPath) ?? new Set<string>();\n set.add(name);\n coverage.set(relPath, set);\n }\n }\n\n // Which presets want vectors is a property of the config, not of any file.\n const embedding = embedEnabled(cfg);\n const vectorPresets = embedding ? new Set(presetNames(cfg).filter((name) => presetHasSignal(cfg, name, 'vectors'))) : null;\n\n const files: FileStat[] = [];\n for (const relPath of [...coverage.keys()].sort()) {\n const absPath = join(baseDir, relPath); // join re-applies the platform separator for fs calls\n // node:fs glob matches directories and dangling symlinks; fast-glob returned neither, so\n // one stat filters both back out (throwIfNoEntry keeps a dangling link from throwing).\n const st = statSync(absPath, NO_THROW);\n if (!st?.isFile()) continue;\n const presets = [...(coverage.get(relPath) as Set<string>)].sort();\n const embed = vectorPresets !== null && presets.some((name) => vectorPresets.has(name));\n files.push({ relPath, absPath, mtimeMs: st.mtimeMs, ctimeMs: st.birthtimeMs, size: st.size, presets, embed });\n }\n return files;\n}\n"],"names":["globSync","statSync","join","sep","embedEnabled","presetHasSignal","presetNames","toPosixPath","relPath","separator","split","NO_THROW","throwIfNoEntry","listFiles","cfg","baseDir","coverage","Map","posixNeeded","name","preset","presets","matched","include","cwd","exclude","set","get","Set","add","embedding","vectorPresets","filter","files","keys","sort","absPath","st","isFile","embed","some","has","push","mtimeMs","ctimeMs","birthtimeMs","size"],"mappings":"AAAA,SAASA,QAAQ,EAAEC,QAAQ,QAAQ,UAAU;AAC7C,SAASC,IAAI,EAAEC,GAAG,QAAQ,YAAY;AAEtC,SAASC,YAAY,EAAEC,eAAe,EAAEC,WAAW,QAAQ,qBAAqB;AAYhF,6FAA6F;AAC7F,8FAA8F;AAC9F,OAAO,SAASC,YAAYC,OAAe,EAAEC,YAAoBN,GAAG;IAClE,OAAOM,cAAc,OAAOD,QAAQE,KAAK,CAACD,WAAWP,IAAI,CAAC,OAAOM;AACnE;AAEA,2FAA2F;AAC3F,8FAA8F;AAC9F,0CAA0C;AAC1C,MAAMG,WAAW;IAAEC,gBAAgB;AAAM;AAEzC,OAAO,SAASC,UAAUC,GAAW,EAAEC,OAAe;IACpD,MAAMC,WAAW,IAAIC;IACrB,MAAMC,cAAcf,QAAQ;IAC5B,KAAK,MAAMgB,QAAQb,YAAYQ,KAAM;QACnC,MAAMM,SAASN,IAAIO,OAAO,CAACF,KAAK;QAChC,KAAK,MAAMG,WAAWtB,SAASoB,OAAOG,OAAO,EAAE;YAAEC,KAAKT;YAASU,SAASL,OAAOK,OAAO;QAAC,GAAI;gBAE7ET;YADZ,MAAMR,UAAUU,cAAcX,YAAYe,WAAWA;YACrD,MAAMI,OAAMV,gBAAAA,SAASW,GAAG,CAACnB,sBAAbQ,2BAAAA,gBAAyB,IAAIY;YACzCF,IAAIG,GAAG,CAACV;YACRH,SAASU,GAAG,CAAClB,SAASkB;QACxB;IACF;IAEA,2EAA2E;IAC3E,MAAMI,YAAY1B,aAAaU;IAC/B,MAAMiB,gBAAgBD,YAAY,IAAIF,IAAItB,YAAYQ,KAAKkB,MAAM,CAAC,CAACb,OAASd,gBAAgBS,KAAKK,MAAM,eAAe;IAEtH,MAAMc,QAAoB,EAAE;IAC5B,KAAK,MAAMzB,WAAW;WAAIQ,SAASkB,IAAI;KAAG,CAACC,IAAI,GAAI;QACjD,MAAMC,UAAUlC,KAAKa,SAASP,UAAU,sDAAsD;QAC9F,yFAAyF;QACzF,uFAAuF;QACvF,MAAM6B,KAAKpC,SAASmC,SAASzB;QAC7B,IAAI,EAAC0B,eAAAA,yBAAAA,GAAIC,MAAM,KAAI;QACnB,MAAMjB,UAAU;eAAKL,SAASW,GAAG,CAACnB;SAAyB,CAAC2B,IAAI;QAChE,MAAMI,QAAQR,kBAAkB,QAAQV,QAAQmB,IAAI,CAAC,CAACrB,OAASY,cAAcU,GAAG,CAACtB;QACjFc,MAAMS,IAAI,CAAC;YAAElC;YAAS4B;YAASO,SAASN,GAAGM,OAAO;YAAEC,SAASP,GAAGQ,WAAW;YAAEC,MAAMT,GAAGS,IAAI;YAAEzB;YAASkB;QAAM;IAC7G;IACA,OAAON;AACT"}
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
// Word boundaries FTS5's tokenizers cannot find on their own.
|
|
2
|
+
//
|
|
3
|
+
// Segmentation is context-free: seg(query) appears inside seg(document) whenever the query is
|
|
4
|
+
// a substring of the document. That is the contract -- substring semantics, what `grep` and
|
|
5
|
+
// `LIKE '%..%'` give -- and it is what makes index and query agree on every input, always.
|
|
6
|
+
// Intl.Segmenter's word mode is context-dependent (`东京都政府` splits as `东 | 京都 | 政府`
|
|
7
|
+
// while the query `东京` splits as one word) and is rejected for that reason. Its grapheme
|
|
8
|
+
// mode has no such dependency -- UAX #29 cluster boundaries never look past adjacent
|
|
9
|
+
// characters -- so grapheme mode is used below; no dictionary, no word mode.
|
|
10
|
+
// Scripts written without word spaces: a closed set of writing systems.
|
|
11
|
+
// Script_Extensions, not Script: the katakana long-vowel mark ー is Script=Common.
|
|
12
|
+
export const UNSPACED_SCRIPTS = '\\p{scx=Han}\\p{scx=Hiragana}\\p{scx=Katakana}\\p{scx=Thai}\\p{scx=Khmer}\\p{scx=Lao}\\p{scx=Myanmar}';
|
|
13
|
+
// A run is script BASE characters with their combining marks attached; a bare mark after a
|
|
14
|
+
// Latin letter (decomposed é) never starts one.
|
|
15
|
+
const RUN = new RegExp(`((?:[${UNSPACED_SCRIPTS}]\\p{M}*)+)`, 'gu');
|
|
16
|
+
const HAS_RUN = new RegExp(`[${UNSPACED_SCRIPTS}]`, 'u');
|
|
17
|
+
// Grapheme clusters, ECMA-402/UAX #29: base char plus its marks, ZWJ sequences, Hangul jamo.
|
|
18
|
+
// Built once (construction cost amortizes) and used for both index and query splitting.
|
|
19
|
+
const GRAPHEME_SEGMENTER = new Intl.Segmenter(undefined, {
|
|
20
|
+
granularity: 'grapheme'
|
|
21
|
+
});
|
|
22
|
+
// unicode61 drops Unicode punctuation as a separator. Some of it (。、「」) has Script_Extensions
|
|
23
|
+
// into an unspaced script, so RUN keeps it -- a grapheme matching this becomes a split point.
|
|
24
|
+
const PUNCTUATION = /\p{P}/u;
|
|
25
|
+
// Token barrier between separate runs (or a punctuation split within one) so their graphemes
|
|
26
|
+
// are never phrase-adjacent. U+A7F7: a letter (so FTS5 keeps it as a token) no one types.
|
|
27
|
+
const BARRIER = 'ꟷ';
|
|
28
|
+
function graphemes(run) {
|
|
29
|
+
return Array.from(GRAPHEME_SEGMENTER.segment(run), (s)=>s.segment);
|
|
30
|
+
}
|
|
31
|
+
// A run's graphemes, cut into punctuation-free groups at every punctuation grapheme (dropped,
|
|
32
|
+
// matching unicode61). The one place punctuation is classified, so index and query agree.
|
|
33
|
+
function splitOnPunctuation(run) {
|
|
34
|
+
const groups = [
|
|
35
|
+
[]
|
|
36
|
+
];
|
|
37
|
+
for (const g of graphemes(run)){
|
|
38
|
+
if (PUNCTUATION.test(g)) groups.push([]);
|
|
39
|
+
else groups[groups.length - 1].push(g);
|
|
40
|
+
}
|
|
41
|
+
return groups.filter((g)=>g.length > 0);
|
|
42
|
+
}
|
|
43
|
+
// Index side: '' when the field has no unspaced-script run (the common case, paid for by
|
|
44
|
+
// nothing). Otherwise each run explodes into its graphemes, barrier-delimited from its
|
|
45
|
+
// neighbors and from a punctuation split within itself.
|
|
46
|
+
export function segmentField(text) {
|
|
47
|
+
if (!HAS_RUN.test(text)) return '';
|
|
48
|
+
const out = text.replace(RUN, (run)=>{
|
|
49
|
+
const body = splitOnPunctuation(run).map((g)=>g.join(' ')).join(` ${BARRIER} `);
|
|
50
|
+
return ` ${BARRIER} ${body} ${BARRIER} `;
|
|
51
|
+
});
|
|
52
|
+
return out.replace(/\s+/g, ' ').trim();
|
|
53
|
+
}
|
|
54
|
+
// A `title:`/`summary:`/`text:` qualifier directly before a run that is about to become a
|
|
55
|
+
// quoted grapheme phrase, so the rewrite can retarget it at the matching `_seg` column.
|
|
56
|
+
const QUALIFIER = /(^|[\s(])(-?)(title|summary|text)\s*:\s*$/;
|
|
57
|
+
// An unqualified run's target: raw title/summary/text drop punctuation as unicode61's token
|
|
58
|
+
// separator, so two adjacent single-grapheme phrase positions match across punctuation the same
|
|
59
|
+
// as across a real gap (`数数` vs `数。数`) -- a false positive only the barriered `_seg` columns
|
|
60
|
+
// are safe from. FTS5's column-set filter, not parens+OR, so the group still composes under
|
|
61
|
+
// AND/OR/NOT/juxtaposition exactly like the single-column qualifier form below.
|
|
62
|
+
const SIDECAR_COLUMNS = '{title_seg summary_seg text_seg}:';
|
|
63
|
+
// A run's punctuation-free groups, each its own quoted phrase (bare token if one grapheme),
|
|
64
|
+
// space-joined -- the same split points segmentField barriers, so query and index agree.
|
|
65
|
+
function runQuery(run, columnPrefix) {
|
|
66
|
+
return splitOnPunctuation(run).map((g)=>`${columnPrefix}${g.length > 1 ? `"${g.join(' ')}"` : g[0]}`).join(' ');
|
|
67
|
+
}
|
|
68
|
+
// Query side: each unspaced run becomes phrases of its graphemes, matching how segmentField
|
|
69
|
+
// indexed it. A qualifier ahead of such a run maps to its `_seg` column; unqualified maps to all
|
|
70
|
+
// three (SIDECAR_COLUMNS), since raw columns cannot express the contract. An author's own quoted
|
|
71
|
+
// phrase is their explicit escape hatch to FTS5's native syntax, and passes through byte-identical.
|
|
72
|
+
export function segmentMatch(terms) {
|
|
73
|
+
if (!HAS_RUN.test(terms)) return terms;
|
|
74
|
+
let out = '';
|
|
75
|
+
let quoted = false;
|
|
76
|
+
const pieces = terms.split(RUN); // split keeps captured runs at odd indices
|
|
77
|
+
for(let i = 0; i < pieces.length; i++){
|
|
78
|
+
if (i % 2 === 0) {
|
|
79
|
+
for (const ch of pieces[i])if (ch === '"') quoted = !quoted;
|
|
80
|
+
out += pieces[i];
|
|
81
|
+
continue;
|
|
82
|
+
}
|
|
83
|
+
if (quoted) {
|
|
84
|
+
out += pieces[i]; // an author's phrase is matched as written
|
|
85
|
+
continue;
|
|
86
|
+
}
|
|
87
|
+
const m = out.match(QUALIFIER);
|
|
88
|
+
if (m) {
|
|
89
|
+
out = `${out.slice(0, m.index)}${m[1]}${runQuery(pieces[i], `${m[2]}${m[3]}_seg:`)}`;
|
|
90
|
+
} else {
|
|
91
|
+
out += runQuery(pieces[i], SIDECAR_COLUMNS);
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
return out;
|
|
95
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/text/segment.ts"],"sourcesContent":["// Word boundaries FTS5's tokenizers cannot find on their own.\n//\n// Segmentation is context-free: seg(query) appears inside seg(document) whenever the query is\n// a substring of the document. That is the contract -- substring semantics, what `grep` and\n// `LIKE '%..%'` give -- and it is what makes index and query agree on every input, always.\n// Intl.Segmenter's word mode is context-dependent (`东京都政府` splits as `东 | 京都 | 政府`\n// while the query `东京` splits as one word) and is rejected for that reason. Its grapheme\n// mode has no such dependency -- UAX #29 cluster boundaries never look past adjacent\n// characters -- so grapheme mode is used below; no dictionary, no word mode.\n\n// Scripts written without word spaces: a closed set of writing systems.\n// Script_Extensions, not Script: the katakana long-vowel mark ー is Script=Common.\nexport const UNSPACED_SCRIPTS = '\\\\p{scx=Han}\\\\p{scx=Hiragana}\\\\p{scx=Katakana}\\\\p{scx=Thai}\\\\p{scx=Khmer}\\\\p{scx=Lao}\\\\p{scx=Myanmar}';\n// A run is script BASE characters with their combining marks attached; a bare mark after a\n// Latin letter (decomposed é) never starts one.\nconst RUN = new RegExp(`((?:[${UNSPACED_SCRIPTS}]\\\\p{M}*)+)`, 'gu');\nconst HAS_RUN = new RegExp(`[${UNSPACED_SCRIPTS}]`, 'u');\n// Grapheme clusters, ECMA-402/UAX #29: base char plus its marks, ZWJ sequences, Hangul jamo.\n// Built once (construction cost amortizes) and used for both index and query splitting.\nconst GRAPHEME_SEGMENTER = new Intl.Segmenter(undefined, { granularity: 'grapheme' });\n// unicode61 drops Unicode punctuation as a separator. Some of it (。、「」) has Script_Extensions\n// into an unspaced script, so RUN keeps it -- a grapheme matching this becomes a split point.\nconst PUNCTUATION = /\\p{P}/u;\n// Token barrier between separate runs (or a punctuation split within one) so their graphemes\n// are never phrase-adjacent. U+A7F7: a letter (so FTS5 keeps it as a token) no one types.\nconst BARRIER = 'ꟷ';\n\nfunction graphemes(run: string): string[] {\n return Array.from(GRAPHEME_SEGMENTER.segment(run), (s) => s.segment);\n}\n\n// A run's graphemes, cut into punctuation-free groups at every punctuation grapheme (dropped,\n// matching unicode61). The one place punctuation is classified, so index and query agree.\nfunction splitOnPunctuation(run: string): string[][] {\n const groups: string[][] = [[]];\n for (const g of graphemes(run)) {\n if (PUNCTUATION.test(g)) groups.push([]);\n else groups[groups.length - 1].push(g);\n }\n return groups.filter((g) => g.length > 0);\n}\n\n// Index side: '' when the field has no unspaced-script run (the common case, paid for by\n// nothing). Otherwise each run explodes into its graphemes, barrier-delimited from its\n// neighbors and from a punctuation split within itself.\nexport function segmentField(text: string): string {\n if (!HAS_RUN.test(text)) return '';\n const out = text.replace(RUN, (run) => {\n const body = splitOnPunctuation(run)\n .map((g) => g.join(' '))\n .join(` ${BARRIER} `);\n return ` ${BARRIER} ${body} ${BARRIER} `;\n });\n return out.replace(/\\s+/g, ' ').trim();\n}\n\n// A `title:`/`summary:`/`text:` qualifier directly before a run that is about to become a\n// quoted grapheme phrase, so the rewrite can retarget it at the matching `_seg` column.\nconst QUALIFIER = /(^|[\\s(])(-?)(title|summary|text)\\s*:\\s*$/;\n\n// An unqualified run's target: raw title/summary/text drop punctuation as unicode61's token\n// separator, so two adjacent single-grapheme phrase positions match across punctuation the same\n// as across a real gap (`数数` vs `数。数`) -- a false positive only the barriered `_seg` columns\n// are safe from. FTS5's column-set filter, not parens+OR, so the group still composes under\n// AND/OR/NOT/juxtaposition exactly like the single-column qualifier form below.\nconst SIDECAR_COLUMNS = '{title_seg summary_seg text_seg}:';\n\n// A run's punctuation-free groups, each its own quoted phrase (bare token if one grapheme),\n// space-joined -- the same split points segmentField barriers, so query and index agree.\nfunction runQuery(run: string, columnPrefix: string): string {\n return splitOnPunctuation(run)\n .map((g) => `${columnPrefix}${g.length > 1 ? `\"${g.join(' ')}\"` : g[0]}`)\n .join(' ');\n}\n\n// Query side: each unspaced run becomes phrases of its graphemes, matching how segmentField\n// indexed it. A qualifier ahead of such a run maps to its `_seg` column; unqualified maps to all\n// three (SIDECAR_COLUMNS), since raw columns cannot express the contract. An author's own quoted\n// phrase is their explicit escape hatch to FTS5's native syntax, and passes through byte-identical.\nexport function segmentMatch(terms: string): string {\n if (!HAS_RUN.test(terms)) return terms;\n let out = '';\n let quoted = false;\n const pieces = terms.split(RUN); // split keeps captured runs at odd indices\n for (let i = 0; i < pieces.length; i++) {\n if (i % 2 === 0) {\n for (const ch of pieces[i]) if (ch === '\"') quoted = !quoted;\n out += pieces[i];\n continue;\n }\n if (quoted) {\n out += pieces[i]; // an author's phrase is matched as written\n continue;\n }\n const m = out.match(QUALIFIER);\n if (m) {\n out = `${out.slice(0, m.index)}${m[1]}${runQuery(pieces[i], `${m[2]}${m[3]}_seg:`)}`;\n } else {\n out += runQuery(pieces[i], SIDECAR_COLUMNS);\n }\n }\n return out;\n}\n"],"names":["UNSPACED_SCRIPTS","RUN","RegExp","HAS_RUN","GRAPHEME_SEGMENTER","Intl","Segmenter","undefined","granularity","PUNCTUATION","BARRIER","graphemes","run","Array","from","segment","s","splitOnPunctuation","groups","g","test","push","length","filter","segmentField","text","out","replace","body","map","join","trim","QUALIFIER","SIDECAR_COLUMNS","runQuery","columnPrefix","segmentMatch","terms","quoted","pieces","split","i","ch","m","match","slice","index"],"mappings":"AAAA,8DAA8D;AAC9D,EAAE;AACF,8FAA8F;AAC9F,4FAA4F;AAC5F,2FAA2F;AAC3F,mFAAmF;AACnF,yFAAyF;AACzF,qFAAqF;AACrF,6EAA6E;AAE7E,wEAAwE;AACxE,kFAAkF;AAClF,OAAO,MAAMA,mBAAmB,wGAAwG;AACxI,2FAA2F;AAC3F,gDAAgD;AAChD,MAAMC,MAAM,IAAIC,OAAO,CAAC,KAAK,EAAEF,iBAAiB,WAAW,CAAC,EAAE;AAC9D,MAAMG,UAAU,IAAID,OAAO,CAAC,CAAC,EAAEF,iBAAiB,CAAC,CAAC,EAAE;AACpD,6FAA6F;AAC7F,wFAAwF;AACxF,MAAMI,qBAAqB,IAAIC,KAAKC,SAAS,CAACC,WAAW;IAAEC,aAAa;AAAW;AACnF,8FAA8F;AAC9F,8FAA8F;AAC9F,MAAMC,cAAc;AACpB,6FAA6F;AAC7F,0FAA0F;AAC1F,MAAMC,UAAU;AAEhB,SAASC,UAAUC,GAAW;IAC5B,OAAOC,MAAMC,IAAI,CAACV,mBAAmBW,OAAO,CAACH,MAAM,CAACI,IAAMA,EAAED,OAAO;AACrE;AAEA,8FAA8F;AAC9F,0FAA0F;AAC1F,SAASE,mBAAmBL,GAAW;IACrC,MAAMM,SAAqB;QAAC,EAAE;KAAC;IAC/B,KAAK,MAAMC,KAAKR,UAAUC,KAAM;QAC9B,IAAIH,YAAYW,IAAI,CAACD,IAAID,OAAOG,IAAI,CAAC,EAAE;aAClCH,MAAM,CAACA,OAAOI,MAAM,GAAG,EAAE,CAACD,IAAI,CAACF;IACtC;IACA,OAAOD,OAAOK,MAAM,CAAC,CAACJ,IAAMA,EAAEG,MAAM,GAAG;AACzC;AAEA,yFAAyF;AACzF,uFAAuF;AACvF,wDAAwD;AACxD,OAAO,SAASE,aAAaC,IAAY;IACvC,IAAI,CAACtB,QAAQiB,IAAI,CAACK,OAAO,OAAO;IAChC,MAAMC,MAAMD,KAAKE,OAAO,CAAC1B,KAAK,CAACW;QAC7B,MAAMgB,OAAOX,mBAAmBL,KAC7BiB,GAAG,CAAC,CAACV,IAAMA,EAAEW,IAAI,CAAC,MAClBA,IAAI,CAAC,CAAC,CAAC,EAAEpB,QAAQ,CAAC,CAAC;QACtB,OAAO,CAAC,CAAC,EAAEA,QAAQ,CAAC,EAAEkB,KAAK,CAAC,EAAElB,QAAQ,CAAC,CAAC;IAC1C;IACA,OAAOgB,IAAIC,OAAO,CAAC,QAAQ,KAAKI,IAAI;AACtC;AAEA,0FAA0F;AAC1F,wFAAwF;AACxF,MAAMC,YAAY;AAElB,4FAA4F;AAC5F,gGAAgG;AAChG,6FAA6F;AAC7F,4FAA4F;AAC5F,gFAAgF;AAChF,MAAMC,kBAAkB;AAExB,4FAA4F;AAC5F,yFAAyF;AACzF,SAASC,SAAStB,GAAW,EAAEuB,YAAoB;IACjD,OAAOlB,mBAAmBL,KACvBiB,GAAG,CAAC,CAACV,IAAM,GAAGgB,eAAehB,EAAEG,MAAM,GAAG,IAAI,CAAC,CAAC,EAAEH,EAAEW,IAAI,CAAC,KAAK,CAAC,CAAC,GAAGX,CAAC,CAAC,EAAE,EAAE,EACvEW,IAAI,CAAC;AACV;AAEA,4FAA4F;AAC5F,iGAAiG;AACjG,iGAAiG;AACjG,oGAAoG;AACpG,OAAO,SAASM,aAAaC,KAAa;IACxC,IAAI,CAAClC,QAAQiB,IAAI,CAACiB,QAAQ,OAAOA;IACjC,IAAIX,MAAM;IACV,IAAIY,SAAS;IACb,MAAMC,SAASF,MAAMG,KAAK,CAACvC,MAAM,2CAA2C;IAC5E,IAAK,IAAIwC,IAAI,GAAGA,IAAIF,OAAOjB,MAAM,EAAEmB,IAAK;QACtC,IAAIA,IAAI,MAAM,GAAG;YACf,KAAK,MAAMC,MAAMH,MAAM,CAACE,EAAE,CAAE,IAAIC,OAAO,KAAKJ,SAAS,CAACA;YACtDZ,OAAOa,MAAM,CAACE,EAAE;YAChB;QACF;QACA,IAAIH,QAAQ;YACVZ,OAAOa,MAAM,CAACE,EAAE,EAAE,2CAA2C;YAC7D;QACF;QACA,MAAME,IAAIjB,IAAIkB,KAAK,CAACZ;QACpB,IAAIW,GAAG;YACLjB,MAAM,GAAGA,IAAImB,KAAK,CAAC,GAAGF,EAAEG,KAAK,IAAIH,CAAC,CAAC,EAAE,GAAGT,SAASK,MAAM,CAACE,EAAE,EAAE,GAAGE,CAAC,CAAC,EAAE,GAAGA,CAAC,CAAC,EAAE,CAAC,KAAK,CAAC,GAAG;QACtF,OAAO;YACLjB,OAAOQ,SAASK,MAAM,CAACE,EAAE,EAAER;QAC7B;IACF;IACA,OAAOP;AACT"}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import { extractText, parse } from '../chunk/index.js';
|
|
2
|
+
import { normalizeText } from '../scan/frontmatter.js';
|
|
3
|
+
// Flat prose from blocks a caller already parsed, via the same extractText the chunker uses
|
|
4
|
+
// (src/chunk) -- words and vectors read one definition of "the prose of this note".
|
|
5
|
+
export function textFromBlocks(blocks) {
|
|
6
|
+
const text = blocks.map((block)=>extractText(block.node)).filter((s)=>s.length > 0).join('\n');
|
|
7
|
+
return normalizeText(text);
|
|
8
|
+
}
|
|
9
|
+
// Thin parse + delegate, for callers with no blocks of their own already.
|
|
10
|
+
export function stripText(value) {
|
|
11
|
+
return textFromBlocks(parse(value));
|
|
12
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["/Users/kevin/Dev/OpenSource/ai/sensemaking/src/text/strip.ts"],"sourcesContent":["import type { Block } from '../chunk/index.ts';\nimport { extractText, parse } from '../chunk/index.ts';\nimport { normalizeText } from '../scan/frontmatter.ts';\n\n// Flat prose from blocks a caller already parsed, via the same extractText the chunker uses\n// (src/chunk) -- words and vectors read one definition of \"the prose of this note\".\nexport function textFromBlocks(blocks: Block[]): string {\n const text = blocks\n .map((block) => extractText(block.node))\n .filter((s) => s.length > 0)\n .join('\\n');\n return normalizeText(text);\n}\n\n// Thin parse + delegate, for callers with no blocks of their own already.\nexport function stripText(value: string): string {\n return textFromBlocks(parse(value));\n}\n"],"names":["extractText","parse","normalizeText","textFromBlocks","blocks","text","map","block","node","filter","s","length","join","stripText","value"],"mappings":"AACA,SAASA,WAAW,EAAEC,KAAK,QAAQ,oBAAoB;AACvD,SAASC,aAAa,QAAQ,yBAAyB;AAEvD,4FAA4F;AAC5F,oFAAoF;AACpF,OAAO,SAASC,eAAeC,MAAe;IAC5C,MAAMC,OAAOD,OACVE,GAAG,CAAC,CAACC,QAAUP,YAAYO,MAAMC,IAAI,GACrCC,MAAM,CAAC,CAACC,IAAMA,EAAEC,MAAM,GAAG,GACzBC,IAAI,CAAC;IACR,OAAOV,cAAcG;AACvB;AAEA,0EAA0E;AAC1E,OAAO,SAASQ,UAAUC,KAAa;IACrC,OAAOX,eAAeF,MAAMa;AAC9B"}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "sensemaking",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.17.0",
|
|
4
4
|
"description": "Query and search your markdown notes with context-aware progressive disclosure: SQL over frontmatter, links, and text, plus semantic search and link-graph ranking. No server, no build step",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"markdown",
|
|
@@ -38,6 +38,8 @@
|
|
|
38
38
|
"agent-memory",
|
|
39
39
|
"memory-consolidation",
|
|
40
40
|
"llm-wiki",
|
|
41
|
+
"second-brain",
|
|
42
|
+
"company-brain",
|
|
41
43
|
"dreaming",
|
|
42
44
|
"cli"
|
|
43
45
|
],
|
|
@@ -77,17 +79,36 @@
|
|
|
77
79
|
},
|
|
78
80
|
"dependencies": {
|
|
79
81
|
"@huggingface/tokenizers": "^0.1.3",
|
|
80
|
-
"
|
|
82
|
+
"franc-min": "^6.2.0",
|
|
83
|
+
"mdast-util-from-markdown": "^2.0.3",
|
|
84
|
+
"mdast-util-gfm-autolink-literal": "^2.0.0",
|
|
85
|
+
"mdast-util-gfm-footnote": "^2.0.0",
|
|
86
|
+
"mdast-util-gfm-strikethrough": "^2.0.0",
|
|
87
|
+
"mdast-util-gfm-table": "^2.0.0",
|
|
88
|
+
"mdast-util-gfm-task-list-item": "^2.0.0",
|
|
89
|
+
"micromark-extension-gfm-autolink-literal": "^2.0.0",
|
|
90
|
+
"micromark-extension-gfm-footnote": "^2.0.0",
|
|
91
|
+
"micromark-extension-gfm-strikethrough": "^2.0.0",
|
|
92
|
+
"micromark-extension-gfm-table": "^2.0.0",
|
|
93
|
+
"micromark-extension-gfm-task-list-item": "^2.0.0",
|
|
81
94
|
"yaml": "^2.9.0"
|
|
82
95
|
},
|
|
83
96
|
"devDependencies": {
|
|
97
|
+
"@types/mdast": "^4.0.4",
|
|
84
98
|
"@types/mocha": "*",
|
|
85
99
|
"@types/node": "*",
|
|
100
|
+
"cr": "^0.1.0",
|
|
86
101
|
"node-version-use": "*",
|
|
87
102
|
"ts-dev-stack": "*",
|
|
88
103
|
"tsds-config": "*"
|
|
89
104
|
},
|
|
90
105
|
"engines": {
|
|
91
106
|
"node": ">=22.20"
|
|
107
|
+
},
|
|
108
|
+
"allowScripts": {
|
|
109
|
+
"node-semvers": true,
|
|
110
|
+
"node-filename-to-dist-paths": true,
|
|
111
|
+
"node-version-use": true,
|
|
112
|
+
"thread-sleep-compat": true
|
|
92
113
|
}
|
|
93
114
|
}
|