@remnic/core 9.62.0 → 9.62.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/access-admin-ops-surface.d.ts +11 -11
- package/dist/access-admin-ops-surface.js +28 -28
- package/dist/access-audit.js +2 -2
- package/dist/access-authorization-probe.d.ts +11 -11
- package/dist/access-authorization-probe.js +29 -29
- package/dist/access-boundary.d.ts +11 -11
- package/dist/access-boundary.js +28 -28
- package/dist/access-cli.js +93 -91
- package/dist/access-cli.js.map +1 -1
- package/dist/access-coding-context-resolution.d.ts +1 -1
- package/dist/access-extraction-force-flush.d.ts +11 -11
- package/dist/access-extraction-force-flush.js +28 -28
- package/dist/access-health-types.d.ts +1 -1
- package/dist/access-http-lcm-compaction.d.ts +11 -11
- package/dist/access-http-lifecycle-flush.d.ts +11 -11
- package/dist/access-http-offline-stream.d.ts +11 -11
- package/dist/access-http-query.js +28 -28
- package/dist/access-http.d.ts +12 -12
- package/dist/access-http.js +38 -38
- package/dist/access-identity-continuity-surface.d.ts +9 -9
- package/dist/access-identity-continuity-surface.js +28 -28
- package/dist/access-lcm-surface.d.ts +11 -11
- package/dist/access-lcm-surface.js +28 -28
- package/dist/access-mcp.d.ts +11 -11
- package/dist/access-mcp.js +33 -33
- package/dist/access-memory-search-fanout.d.ts +2 -2
- package/dist/{access-namespace-preflight-EMvicnrb.d.ts → access-namespace-preflight-yO4LMGFH.d.ts} +1 -1
- package/dist/access-namespace-preflight.d.ts +3 -3
- package/dist/access-namespace-preflight.js +28 -28
- package/dist/access-observe-write-surface.d.ts +11 -11
- package/dist/access-observe-write-surface.js +28 -28
- package/dist/access-offline-manifest.d.ts +9 -9
- package/dist/access-offline-manifest.js +2 -2
- package/dist/access-operations-batch.js +29 -29
- package/dist/access-operations.d.ts +11 -11
- package/dist/access-operations.js +32 -32
- package/dist/access-recall-concurrency.d.ts +11 -11
- package/dist/access-recall-concurrency.js +28 -28
- package/dist/access-recall-response.d.ts +11 -11
- package/dist/access-recall-response.js +28 -28
- package/dist/access-recall-surface.d.ts +11 -11
- package/dist/access-recall-surface.js +28 -28
- package/dist/{access-service-BeDYQQ62.d.ts → access-service-M4LgO141.d.ts} +7 -7
- package/dist/access-service-helpers.d.ts +11 -11
- package/dist/access-service.d.ts +11 -11
- package/dist/access-service.js +28 -28
- package/dist/access-surface-catalog.d.ts +11 -11
- package/dist/access-surface-catalog.js +30 -30
- package/dist/access-wearables-meetings-surface.d.ts +3 -3
- package/dist/action-confidence.d.ts +1 -1
- package/dist/active-memory-bridge.d.ts +1 -1
- package/dist/active-memory-bridge.js +2 -2
- package/dist/active-recall.d.ts +1 -1
- package/dist/active-recall.js +3 -3
- package/dist/ambient-provenance.d.ts +1 -1
- package/dist/artifact-search.d.ts +1 -1
- package/dist/{auto-sync-FMWQQB3J.js → auto-sync-Z52VWMNC.js} +3 -3
- package/dist/behavior-learner.d.ts +1 -1
- package/dist/behavior-learner.js +2 -2
- package/dist/behavior-signals.d.ts +1 -1
- package/dist/bootstrap.d.ts +9 -9
- package/dist/briefing-window.d.ts +2 -2
- package/dist/briefing.d.ts +2 -2
- package/dist/briefing.js +13 -12
- package/dist/buffer-surprise-report.d.ts +1 -1
- package/dist/buffer-turn-helpers.d.ts +1 -1
- package/dist/buffer.d.ts +2 -2
- package/dist/bulk-import/index.d.ts +3 -3
- package/dist/calibration.d.ts +1 -1
- package/dist/capabilities.d.ts +1 -1
- package/dist/{catalog-B2SxZ6LU.d.ts → catalog-CyYTf_ok.d.ts} +1 -1
- package/dist/causal-behavior.d.ts +1 -1
- package/dist/causal-behavior.js +3 -3
- package/dist/causal-chain.js +3 -3
- package/dist/causal-consolidation.d.ts +1 -1
- package/dist/causal-consolidation.js +16 -16
- package/dist/causal-retrieval.js +3 -3
- package/dist/causal-trajectory-graph.d.ts +1 -1
- package/dist/causal-trajectory.js +2 -2
- package/dist/{chunk-B2QHVLJO.js → chunk-256FBCVX.js} +22 -10
- package/dist/chunk-256FBCVX.js.map +1 -0
- package/dist/{chunk-QP2IZTUR.js → chunk-2TYIAKU5.js} +2 -2
- package/dist/{chunk-HQBVQICL.js → chunk-347ZR6D4.js} +2 -2
- package/dist/{chunk-76YNCJ5H.js → chunk-3F3CVVDS.js} +2 -2
- package/dist/{chunk-TE4MKFT2.js → chunk-3MA7JVXN.js} +4 -4
- package/dist/{chunk-KHQ57QLJ.js → chunk-46WM6JX2.js} +7 -3
- package/dist/chunk-46WM6JX2.js.map +1 -0
- package/dist/chunk-47DU4SGJ.js +92 -0
- package/dist/chunk-47DU4SGJ.js.map +1 -0
- package/dist/{chunk-AWRJBAIH.js → chunk-4FFROSFG.js} +37 -14
- package/dist/{chunk-AWRJBAIH.js.map → chunk-4FFROSFG.js.map} +1 -1
- package/dist/{chunk-N4OYPGZS.js → chunk-4HO3542O.js} +2 -2
- package/dist/{chunk-HEEEMJ26.js → chunk-4IZ2IT7D.js} +3 -3
- package/dist/{chunk-4XAVQLBR.js → chunk-4Y2RHVCV.js} +2 -2
- package/dist/{chunk-33JBK2XP.js → chunk-5F6HMNMZ.js} +2 -2
- package/dist/{chunk-5WQMORU5.js → chunk-5PFGMO33.js} +2 -2
- package/dist/{chunk-6HEM6HTQ.js → chunk-6GGTATZF.js} +2 -2
- package/dist/{chunk-W3Y6FYRK.js → chunk-7BJYTX6Q.js} +2 -2
- package/dist/{chunk-2GRLJMKJ.js → chunk-7DRSWPUV.js} +3 -3
- package/dist/{chunk-TVVEYCNW.js → chunk-7K5Q6COX.js} +4 -4
- package/dist/{chunk-HXXGZ4KV.js → chunk-ALNZEWGN.js} +2 -2
- package/dist/{chunk-5WVUO6QT.js → chunk-AR253JBM.js} +1 -1
- package/dist/chunk-AR253JBM.js.map +1 -0
- package/dist/{chunk-S3B3UIQ3.js → chunk-AYIHDC5S.js} +2 -2
- package/dist/{chunk-2RIDW23Y.js → chunk-BBHFAVMP.js} +8 -8
- package/dist/{chunk-MJ4UGUR6.js → chunk-BHO6RHE5.js} +3 -3
- package/dist/{chunk-URN7XWGC.js → chunk-BP6WMYLP.js} +3 -3
- package/dist/{chunk-YUIBXMZY.js → chunk-CALURYWR.js} +284 -249
- package/dist/chunk-CALURYWR.js.map +1 -0
- package/dist/{chunk-D5HDJWNN.js → chunk-CH627RRV.js} +4 -4
- package/dist/{chunk-C3R732UY.js → chunk-CLJ5QS6J.js} +257 -60
- package/dist/chunk-CLJ5QS6J.js.map +1 -0
- package/dist/{chunk-YLHM5BQQ.js → chunk-DCUMFQYM.js} +171 -8
- package/dist/chunk-DCUMFQYM.js.map +1 -0
- package/dist/{chunk-ZBJMUXZH.js → chunk-DJ7HMNSL.js} +15 -8
- package/dist/chunk-DJ7HMNSL.js.map +1 -0
- package/dist/{chunk-XDK6ZFLL.js → chunk-DROAPP6N.js} +3 -3
- package/dist/{chunk-MU3TPZOV.js → chunk-DTLF2SLZ.js} +2 -2
- package/dist/{chunk-I67HKRBF.js → chunk-EFWHCTYJ.js} +2 -2
- package/dist/{chunk-XD4VGLDL.js → chunk-FF6E6HOT.js} +56 -52
- package/dist/chunk-FF6E6HOT.js.map +1 -0
- package/dist/{chunk-NP7FLVZW.js → chunk-G2FCSILA.js} +2 -2
- package/dist/{chunk-3GPBFDXH.js → chunk-H23GIXJC.js} +2 -2
- package/dist/{chunk-WT6UL3UA.js → chunk-H7PRHUBD.js} +2 -2
- package/dist/{chunk-4JGSG7QN.js → chunk-HSBQJRV2.js} +2 -2
- package/dist/{chunk-L33N3BCA.js → chunk-HVKMJAEF.js} +2 -2
- package/dist/{chunk-CSOYVTRX.js → chunk-IIPJKCZB.js} +3 -3
- package/dist/{chunk-E57E5SCS.js → chunk-IRLH4NM2.js} +3 -3
- package/dist/{chunk-H32ANIFP.js → chunk-IZRNVNPV.js} +2 -2
- package/dist/{chunk-XBR5BAKW.js → chunk-JCJGYNIS.js} +3 -3
- package/dist/{chunk-EIQFIJDM.js → chunk-JDI7G6EA.js} +2 -2
- package/dist/{chunk-QWPAQLYV.js → chunk-JFLKSD5V.js} +2 -2
- package/dist/{chunk-KQS72GCR.js → chunk-LJOIZPJH.js} +2 -2
- package/dist/{chunk-N4EPIKSY.js → chunk-LLGDVG52.js} +3 -3
- package/dist/{chunk-7N7DB4L5.js → chunk-LTJTP5YQ.js} +10 -9
- package/dist/chunk-LTJTP5YQ.js.map +1 -0
- package/dist/{chunk-VOELBUS4.js → chunk-NXH34IPL.js} +2 -2
- package/dist/{chunk-XL3TATMJ.js → chunk-OQIRBGUU.js} +21 -12
- package/dist/chunk-OQIRBGUU.js.map +1 -0
- package/dist/chunk-PJKTFXE2.js +119 -0
- package/dist/chunk-PJKTFXE2.js.map +1 -0
- package/dist/{chunk-JGW2EUCQ.js → chunk-PMRUMGK6.js} +2 -2
- package/dist/{chunk-KAB7TZN3.js → chunk-PSNN25DW.js} +2 -2
- package/dist/{chunk-LRGSLV7M.js → chunk-PYHOAXTI.js} +4 -4
- package/dist/{chunk-PA5OPZX2.js → chunk-QCI73Z7C.js} +2 -2
- package/dist/{chunk-EHMMRZOZ.js → chunk-QICOU4RY.js} +2 -2
- package/dist/{chunk-FW73HN33.js → chunk-RFH7U5LT.js} +2 -2
- package/dist/{chunk-4SGDYYWV.js → chunk-RPKJHYVE.js} +2 -2
- package/dist/{chunk-DSA47WBN.js → chunk-RYST72N5.js} +13 -2
- package/dist/chunk-RYST72N5.js.map +1 -0
- package/dist/{chunk-L4DLLKPN.js → chunk-SZ4HFZKB.js} +2 -2
- package/dist/chunk-T3GIETKV.js +152 -0
- package/dist/chunk-T3GIETKV.js.map +1 -0
- package/dist/{chunk-MYZ4NA43.js → chunk-TGSMU3F2.js} +3 -3
- package/dist/{chunk-CM4SZZBN.js → chunk-TIRJV2LW.js} +2 -2
- package/dist/{chunk-EGW4O32Z.js → chunk-UDTME32W.js} +7 -4
- package/dist/chunk-UDTME32W.js.map +1 -0
- package/dist/{chunk-CIA6ZUBA.js → chunk-WMOJIOIG.js} +2 -2
- package/dist/{chunk-W4Q54HM3.js → chunk-WR2YKUIW.js} +2 -2
- package/dist/{chunk-VWDTV6SH.js → chunk-X6E7TT3A.js} +2 -2
- package/dist/{chunk-D4OCJSLV.js → chunk-YDPAE2RD.js} +2 -2
- package/dist/{chunk-G3YBWFUC.js → chunk-YIIC5IDU.js} +35 -18
- package/dist/chunk-YIIC5IDU.js.map +1 -0
- package/dist/{chunk-QRKYABS6.js → chunk-YSR3DXGC.js} +2 -2
- package/dist/{chunk-LBJBNWS2.js → chunk-YTHQHH5U.js} +2 -2
- package/dist/{chunk-GN7HO2LA.js → chunk-ZPGHA7EO.js} +12 -12
- package/dist/{chunk-VUDALPI5.js → chunk-ZQKXRKMR.js} +19 -5
- package/dist/chunk-ZQKXRKMR.js.map +1 -0
- package/dist/{chunk-FQSDU634.js → chunk-ZS4JADYY.js} +3 -3
- package/dist/{cli-DB_uakTc.d.ts → cli-Bw9z-KqX.d.ts} +5 -5
- package/dist/cli.d.ts +13 -13
- package/dist/cli.js +62 -62
- package/dist/coding/pre-action-gate.d.ts +1 -1
- package/dist/coding/pre-action-gate.js +3 -3
- package/dist/compounding/engine.d.ts +2 -2
- package/dist/compounding/engine.js +13 -12
- package/dist/compounding/preference-consolidator.d.ts +1 -1
- package/dist/compression-optimizer.d.ts +1 -1
- package/dist/config.d.ts +1 -1
- package/dist/config.js +3 -3
- package/dist/connectors/codex-materialize-runner.d.ts +1 -1
- package/dist/connectors/codex-materialize-runner.js +13 -12
- package/dist/connectors/codex-materialize.d.ts +1 -1
- package/dist/connectors/index.d.ts +1 -1
- package/dist/connectors/index.js +13 -12
- package/dist/consolidation-provenance-check.d.ts +2 -2
- package/dist/consolidation-undo.d.ts +2 -2
- package/dist/contradiction/index.d.ts +2 -2
- package/dist/converge-config.d.ts +1 -1
- package/dist/convergence-refresh.d.ts +2 -2
- package/dist/convergence-refresh.js +15 -14
- package/dist/conversation-index/backend.d.ts +1 -1
- package/dist/conversation-index/chunker.d.ts +1 -1
- package/dist/conversation-index/faiss-adapter.d.ts +1 -1
- package/dist/conversation-index/indexer.d.ts +1 -1
- package/dist/conversation-index/search.d.ts +1 -1
- package/dist/corpus-watermark.d.ts +1 -1
- package/dist/corpus-watermark.js +15 -14
- package/dist/day-summary.d.ts +1 -1
- package/dist/delinearize.d.ts +1 -1
- package/dist/dependency-propagation-config.d.ts +1 -1
- package/dist/{dependency-propagation-delivery-DcOVpK3y.d.ts → dependency-propagation-delivery-BxhvAppj.d.ts} +1 -1
- package/dist/direct-answer-wiring.d.ts +1 -1
- package/dist/direct-answer-wiring.js +3 -3
- package/dist/direct-answer.d.ts +1 -1
- package/dist/direct-answer.js +2 -2
- package/dist/embedding-fallback.d.ts +1 -1
- package/dist/enrichment/index.d.ts +1 -1
- package/dist/entity-origin-fields.d.ts +1 -1
- package/dist/entity-retrieval.d.ts +2 -2
- package/dist/entity-retrieval.js +13 -12
- package/dist/entity-schema.d.ts +1 -1
- package/dist/episodic-context.d.ts +92 -0
- package/dist/episodic-context.js +16 -0
- package/dist/episodic-context.js.map +1 -0
- package/dist/explicit-capture.d.ts +9 -9
- package/dist/external-wiki-access.d.ts +11 -11
- package/dist/external-wiki-access.js +29 -29
- package/dist/external-wiki-collection-registration.d.ts +1 -1
- package/dist/external-wiki-collection.d.ts +1 -1
- package/dist/external-wiki-mcp-tools.d.ts +11 -11
- package/dist/extraction-error-classification.d.ts +1 -1
- package/dist/extraction-faithfulness.d.ts +1 -1
- package/dist/extraction-judge-telemetry.d.ts +1 -1
- package/dist/extraction-judge-training.d.ts +1 -1
- package/dist/extraction-judge.d.ts +1 -1
- package/dist/extraction-liveness.d.ts +1 -1
- package/dist/extraction-normalization.d.ts +1 -1
- package/dist/extraction-normalization.js +2 -2
- package/dist/extraction-prompt.d.ts +1 -1
- package/dist/extraction-source-grounding-rules.d.ts +1 -1
- package/dist/extraction-source-grounding.d.ts +1 -1
- package/dist/extraction-source-grounding.js +3 -3
- package/dist/extraction.d.ts +1 -1
- package/dist/extraction.js +18 -18
- package/dist/fallback-llm.d.ts +1 -1
- package/dist/{forget-5TWK7CZT.js → forget-SZ2BXDGF.js} +2 -2
- package/dist/graph-dashboard-diff.d.ts +1 -1
- package/dist/graph-dashboard-key.d.ts +1 -1
- package/dist/graph-dashboard-parser.d.ts +1 -1
- package/dist/graph-edge-reinforcement.d.ts +1 -1
- package/dist/graph-path-reconstruction.d.ts +1 -1
- package/dist/graph-path-scoring.d.ts +1 -1
- package/dist/graph-snapshot.d.ts +1 -1
- package/dist/graph.d.ts +1 -1
- package/dist/harmonic-construction.js +2 -2
- package/dist/harmonic-retrieval.js +2 -2
- package/dist/importance.d.ts +1 -1
- package/dist/importers/index.d.ts +1 -1
- package/dist/in-flight-reads.d.ts +1 -1
- package/dist/index.d.ts +22 -21
- package/dist/index.js +118 -103
- package/dist/intent.d.ts +1 -1
- package/dist/lcm/engine.d.ts +8 -2
- package/dist/lcm/engine.js +4 -4
- package/dist/lcm/index.d.ts +1 -1
- package/dist/lcm/index.js +9 -9
- package/dist/lcm/tools.d.ts +1 -1
- package/dist/lifecycle.d.ts +1 -1
- package/dist/live-connectors-runner.d.ts +1 -1
- package/dist/local-llm.d.ts +1 -1
- package/dist/local-model-endpoint.d.ts +1 -1
- package/dist/maintenance/memory-governance.d.ts +1 -1
- package/dist/maintenance/memory-governance.js +13 -12
- package/dist/maintenance/rebuild-memory-lifecycle-ledger.d.ts +2 -2
- package/dist/maintenance/rebuild-memory-lifecycle-ledger.js +13 -12
- package/dist/maintenance/rebuild-memory-projection.d.ts +2 -2
- package/dist/maintenance/rebuild-memory-projection.js +14 -13
- package/dist/{maintenance-5c1olVtB.d.ts → maintenance-uJb74j64.d.ts} +3 -3
- package/dist/mcp-memory-inspector-app.d.ts +11 -11
- package/dist/memory-action-policy.d.ts +1 -1
- package/dist/memory-cache.d.ts +1 -1
- package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
- package/dist/memory-projection-store.d.ts +1 -1
- package/dist/memory-provenance.d.ts +1 -1
- package/dist/memory-snapshot.d.ts +1 -1
- package/dist/memory-worth-outcomes.d.ts +2 -2
- package/dist/models-json.d.ts +1 -1
- package/dist/namespaces/migrate.d.ts +3 -3
- package/dist/namespaces/migrate.js +14 -13
- package/dist/namespaces/principal.d.ts +1 -1
- package/dist/namespaces/search.d.ts +1 -1
- package/dist/namespaces/search.js +6 -5
- package/dist/namespaces/storage.d.ts +3 -3
- package/dist/namespaces/storage.js +13 -12
- package/dist/native-knowledge.d.ts +1 -1
- package/dist/offline-sync-impression-drain.d.ts +1 -1
- package/dist/operator-doctor-corpus.d.ts +1 -1
- package/dist/operator-doctor-corpus.js +16 -15
- package/dist/operator-doctor-replica.d.ts +1 -1
- package/dist/operator-toolkit.d.ts +3 -3
- package/dist/operator-toolkit.js +21 -20
- package/dist/orchestration/compression-guideline-coordinator.d.ts +2 -2
- package/dist/orchestration/maintenance.d.ts +4 -4
- package/dist/orchestration/maintenance.js +18 -17
- package/dist/{orchestrator-HPuY6I4i.d.ts → orchestrator-CFi-5JSR.d.ts} +23 -8
- package/dist/orchestrator.d.ts +9 -9
- package/dist/orchestrator.js +93 -91
- package/dist/patterns-cli.d.ts +1 -1
- package/dist/{pipeline-CvVe1Q2_.d.ts → pipeline-BdF3ferY.d.ts} +1 -1
- package/dist/policy-runtime.d.ts +1 -1
- package/dist/policy-runtime.js +3 -3
- package/dist/proactive-contention.d.ts +1 -1
- package/dist/provenance.d.ts +1 -1
- package/dist/provenance.js +2 -2
- package/dist/{public-http-D0JKFkyf.d.ts → public-http-DK-QwaBu.d.ts} +1 -1
- package/dist/{qmd-Bd0Tp76_.d.ts → qmd-D_62bYNl.d.ts} +1 -1
- package/dist/qmd-preflight.d.ts +2 -2
- package/dist/qmd-recall-cache.d.ts +1 -1
- package/dist/qmd.d.ts +2 -2
- package/dist/recall-concurrency-config.d.ts +1 -1
- package/dist/recall-disclosure-escalation.d.ts +1 -1
- package/dist/recall-explain-renderer.d.ts +1 -1
- package/dist/recall-explain-renderer.js +3 -3
- package/dist/recall-memory-map.d.ts +1 -1
- package/dist/recall-planner-llm.d.ts +1 -1
- package/dist/recall-query-policy.js +2 -2
- package/dist/recall-state.d.ts +1 -1
- package/dist/recall-tag-filter.d.ts +1 -1
- package/dist/recall-timings.d.ts +1 -1
- package/dist/recall-tokenization.d.ts +3 -1
- package/dist/recall-tokenization.js +5 -1
- package/dist/recall-xray-cli.d.ts +1 -1
- package/dist/recall-xray-cli.js +4 -4
- package/dist/recall-xray-renderer.d.ts +1 -1
- package/dist/recall-xray-renderer.js +3 -3
- package/dist/recall-xray.d.ts +1 -1
- package/dist/recall-xray.js +2 -2
- package/dist/reconcile/cursor.d.ts +1 -1
- package/dist/reconcile/manifest.d.ts +10 -1
- package/dist/reconcile/manifest.js +1 -1
- package/dist/reconcile/plan.d.ts +1 -1
- package/dist/replay/normalizers/chatgpt.d.ts +1 -1
- package/dist/replay/normalizers/claude.d.ts +1 -1
- package/dist/replay/normalizers/openclaw.d.ts +1 -1
- package/dist/replay/normalizers/shared.d.ts +1 -1
- package/dist/replay/runner.d.ts +1 -1
- package/dist/replay/types.d.ts +1 -1
- package/dist/replica-divergence.d.ts +1 -1
- package/dist/replica-peers-config.d.ts +1 -1
- package/dist/resolve-auth-token.d.ts +1 -1
- package/dist/resume-bundles.js +6 -6
- package/dist/retrieval-agents.d.ts +2 -2
- package/dist/retrieval-tiers.d.ts +1 -1
- package/dist/routing/engine.d.ts +1 -1
- package/dist/routing/store.d.ts +1 -1
- package/dist/salvage-envelope.d.ts +1 -1
- package/dist/{scope-profiles-ClLtm9gK.d.ts → scope-profiles-CeMv7pP0.d.ts} +1 -1
- package/dist/search/embed-helper.d.ts +1 -1
- package/dist/search/factory.d.ts +1 -1
- package/dist/search/factory.js +5 -4
- package/dist/search/index.d.ts +1 -1
- package/dist/search/index.js +9 -8
- package/dist/search/lancedb-backend.d.ts +14 -1
- package/dist/search/lancedb-backend.js +1 -1
- package/dist/search/meilisearch-backend.d.ts +1 -1
- package/dist/search/noop-backend.d.ts +1 -1
- package/dist/search/orama-backend.d.ts +22 -1
- package/dist/search/orama-backend.js +2 -1
- package/dist/search/port.d.ts +1 -1
- package/dist/search/remote-backend.d.ts +1 -1
- package/dist/{semantic-consolidation-B1EkeEDj.d.ts → semantic-consolidation-B-85brYj.d.ts} +1 -1
- package/dist/semantic-consolidation.d.ts +2 -2
- package/dist/semantic-consolidation.js +14 -14
- package/dist/semantic-rule-promotion.js +13 -12
- package/dist/semantic-rule-verifier.d.ts +1 -1
- package/dist/semantic-rule-verifier.js +13 -13
- package/dist/{service-Cf3ANQj9.d.ts → service-CoO0HA2E.d.ts} +2 -2
- package/dist/session-observer-bands.d.ts +1 -1
- package/dist/session-observer-state.d.ts +1 -1
- package/dist/shared-context/manager.d.ts +1 -1
- package/dist/signal.d.ts +1 -1
- package/dist/source-agent-qualifier.d.ts +1 -1
- package/dist/source-agent-qualifier.js +15 -14
- package/dist/{storage-DS2O_6Qt.d.ts → storage-BVAVitkO.d.ts} +1 -1
- package/dist/storage.d.ts +2 -2
- package/dist/storage.js +12 -11
- package/dist/summarizer.d.ts +1 -1
- package/dist/summary-snapshot.d.ts +1 -1
- package/dist/support-passport/index.d.ts +14 -14
- package/dist/support-passport/index.js +31 -31
- package/dist/temporal-supersession.d.ts +2 -2
- package/dist/temporal-timeline-recall.d.ts +1 -1
- package/dist/temporal-validity.d.ts +1 -1
- package/dist/threading.d.ts +1 -1
- package/dist/tier-migration.d.ts +2 -2
- package/dist/tier-routing.d.ts +1 -1
- package/dist/topics.d.ts +1 -1
- package/dist/transcript.d.ts +1 -1
- package/dist/transfer/import-sqlite.js +2 -2
- package/dist/trust-score-stage.d.ts +1 -1
- package/dist/trust-score.d.ts +1 -1
- package/dist/trust-zones.js +2 -2
- package/dist/{types-BOANGhCJ.d.ts → types-DWHOxlZP.d.ts} +3 -0
- package/dist/types.d.ts +1 -1
- package/dist/types.js +1 -1
- package/dist/user-message-cleaning.d.ts +23 -0
- package/dist/user-message-cleaning.js +10 -0
- package/dist/user-message-cleaning.js.map +1 -0
- package/dist/utility-runtime.d.ts +1 -1
- package/dist/verified-recall.js +13 -13
- package/dist/whitespace.d.ts +26 -1
- package/dist/whitespace.js +11 -3
- package/dist/work-product-ledger.js +2 -2
- package/dist/write-envelope.d.ts +1 -1
- package/package.json +12 -2
- package/src/access-service-offline-file-content.test.ts +1 -0
- package/src/cli.ts +14 -13
- package/src/config.ts +11 -0
- package/src/entity-retrieval.ts +4 -7
- package/src/episodic-context-recall.integration.test.ts +169 -0
- package/src/episodic-context.test.ts +266 -0
- package/src/episodic-context.ts +274 -0
- package/src/harmonic-construction.test.ts +187 -2
- package/src/harmonic-retrieval.ts +54 -32
- package/src/index.ts +3 -4
- package/src/lcm/engine.ts +18 -1
- package/src/lifecycle/tombstones.test.ts +134 -0
- package/src/lifecycle/tombstones.ts +29 -2
- package/src/orchestration/episodic-context-section.ts +255 -0
- package/src/orchestration/recall-internal-deps.ts +2 -0
- package/src/orchestration/recall-internal.ts +40 -25
- package/src/orchestration/recall-result-formatter.ts +37 -0
- package/src/orchestration/recall-section-coordinator.ts +20 -14
- package/src/recall-tokenization.ts +14 -8
- package/src/reconcile/manifest.test.ts +133 -4
- package/src/reconcile/manifest.ts +43 -7
- package/src/search/factory.ts +1 -0
- package/src/search/lancedb-backend.ts +24 -16
- package/src/search/orama-backend.test.ts +208 -1
- package/src/search/orama-backend.ts +133 -18
- package/src/search/orama-cjk-tokenizer.test.ts +108 -0
- package/src/search/orama-cjk-tokenizer.ts +143 -0
- package/src/storage/tombstone-migration-sources.test.ts +116 -0
- package/src/storage/tombstone-migration-sources.ts +90 -0
- package/src/storage.ts +13 -34
- package/src/types.ts +3 -3
- package/src/user-message-cleaning.ts +139 -0
- package/src/whitespace.test.ts +178 -0
- package/src/whitespace.ts +119 -0
- package/dist/chunk-5WVUO6QT.js.map +0 -1
- package/dist/chunk-6MKAMLQL.js +0 -16
- package/dist/chunk-6MKAMLQL.js.map +0 -1
- package/dist/chunk-7N7DB4L5.js.map +0 -1
- package/dist/chunk-B2QHVLJO.js.map +0 -1
- package/dist/chunk-C3R732UY.js.map +0 -1
- package/dist/chunk-DSA47WBN.js.map +0 -1
- package/dist/chunk-EGW4O32Z.js.map +0 -1
- package/dist/chunk-G3YBWFUC.js.map +0 -1
- package/dist/chunk-KHQ57QLJ.js.map +0 -1
- package/dist/chunk-VUDALPI5.js.map +0 -1
- package/dist/chunk-XD4VGLDL.js.map +0 -1
- package/dist/chunk-XL3TATMJ.js.map +0 -1
- package/dist/chunk-YLHM5BQQ.js.map +0 -1
- package/dist/chunk-YUIBXMZY.js.map +0 -1
- package/dist/chunk-ZBJMUXZH.js.map +0 -1
- /package/dist/{auto-sync-FMWQQB3J.js.map → auto-sync-Z52VWMNC.js.map} +0 -0
- /package/dist/{chunk-QP2IZTUR.js.map → chunk-2TYIAKU5.js.map} +0 -0
- /package/dist/{chunk-HQBVQICL.js.map → chunk-347ZR6D4.js.map} +0 -0
- /package/dist/{chunk-76YNCJ5H.js.map → chunk-3F3CVVDS.js.map} +0 -0
- /package/dist/{chunk-TE4MKFT2.js.map → chunk-3MA7JVXN.js.map} +0 -0
- /package/dist/{chunk-N4OYPGZS.js.map → chunk-4HO3542O.js.map} +0 -0
- /package/dist/{chunk-HEEEMJ26.js.map → chunk-4IZ2IT7D.js.map} +0 -0
- /package/dist/{chunk-4XAVQLBR.js.map → chunk-4Y2RHVCV.js.map} +0 -0
- /package/dist/{chunk-33JBK2XP.js.map → chunk-5F6HMNMZ.js.map} +0 -0
- /package/dist/{chunk-5WQMORU5.js.map → chunk-5PFGMO33.js.map} +0 -0
- /package/dist/{chunk-6HEM6HTQ.js.map → chunk-6GGTATZF.js.map} +0 -0
- /package/dist/{chunk-W3Y6FYRK.js.map → chunk-7BJYTX6Q.js.map} +0 -0
- /package/dist/{chunk-2GRLJMKJ.js.map → chunk-7DRSWPUV.js.map} +0 -0
- /package/dist/{chunk-TVVEYCNW.js.map → chunk-7K5Q6COX.js.map} +0 -0
- /package/dist/{chunk-HXXGZ4KV.js.map → chunk-ALNZEWGN.js.map} +0 -0
- /package/dist/{chunk-S3B3UIQ3.js.map → chunk-AYIHDC5S.js.map} +0 -0
- /package/dist/{chunk-2RIDW23Y.js.map → chunk-BBHFAVMP.js.map} +0 -0
- /package/dist/{chunk-MJ4UGUR6.js.map → chunk-BHO6RHE5.js.map} +0 -0
- /package/dist/{chunk-URN7XWGC.js.map → chunk-BP6WMYLP.js.map} +0 -0
- /package/dist/{chunk-D5HDJWNN.js.map → chunk-CH627RRV.js.map} +0 -0
- /package/dist/{chunk-XDK6ZFLL.js.map → chunk-DROAPP6N.js.map} +0 -0
- /package/dist/{chunk-MU3TPZOV.js.map → chunk-DTLF2SLZ.js.map} +0 -0
- /package/dist/{chunk-I67HKRBF.js.map → chunk-EFWHCTYJ.js.map} +0 -0
- /package/dist/{chunk-NP7FLVZW.js.map → chunk-G2FCSILA.js.map} +0 -0
- /package/dist/{chunk-3GPBFDXH.js.map → chunk-H23GIXJC.js.map} +0 -0
- /package/dist/{chunk-WT6UL3UA.js.map → chunk-H7PRHUBD.js.map} +0 -0
- /package/dist/{chunk-4JGSG7QN.js.map → chunk-HSBQJRV2.js.map} +0 -0
- /package/dist/{chunk-L33N3BCA.js.map → chunk-HVKMJAEF.js.map} +0 -0
- /package/dist/{chunk-CSOYVTRX.js.map → chunk-IIPJKCZB.js.map} +0 -0
- /package/dist/{chunk-E57E5SCS.js.map → chunk-IRLH4NM2.js.map} +0 -0
- /package/dist/{chunk-H32ANIFP.js.map → chunk-IZRNVNPV.js.map} +0 -0
- /package/dist/{chunk-XBR5BAKW.js.map → chunk-JCJGYNIS.js.map} +0 -0
- /package/dist/{chunk-EIQFIJDM.js.map → chunk-JDI7G6EA.js.map} +0 -0
- /package/dist/{chunk-QWPAQLYV.js.map → chunk-JFLKSD5V.js.map} +0 -0
- /package/dist/{chunk-KQS72GCR.js.map → chunk-LJOIZPJH.js.map} +0 -0
- /package/dist/{chunk-N4EPIKSY.js.map → chunk-LLGDVG52.js.map} +0 -0
- /package/dist/{chunk-VOELBUS4.js.map → chunk-NXH34IPL.js.map} +0 -0
- /package/dist/{chunk-JGW2EUCQ.js.map → chunk-PMRUMGK6.js.map} +0 -0
- /package/dist/{chunk-KAB7TZN3.js.map → chunk-PSNN25DW.js.map} +0 -0
- /package/dist/{chunk-LRGSLV7M.js.map → chunk-PYHOAXTI.js.map} +0 -0
- /package/dist/{chunk-PA5OPZX2.js.map → chunk-QCI73Z7C.js.map} +0 -0
- /package/dist/{chunk-EHMMRZOZ.js.map → chunk-QICOU4RY.js.map} +0 -0
- /package/dist/{chunk-FW73HN33.js.map → chunk-RFH7U5LT.js.map} +0 -0
- /package/dist/{chunk-4SGDYYWV.js.map → chunk-RPKJHYVE.js.map} +0 -0
- /package/dist/{chunk-L4DLLKPN.js.map → chunk-SZ4HFZKB.js.map} +0 -0
- /package/dist/{chunk-MYZ4NA43.js.map → chunk-TGSMU3F2.js.map} +0 -0
- /package/dist/{chunk-CM4SZZBN.js.map → chunk-TIRJV2LW.js.map} +0 -0
- /package/dist/{chunk-CIA6ZUBA.js.map → chunk-WMOJIOIG.js.map} +0 -0
- /package/dist/{chunk-W4Q54HM3.js.map → chunk-WR2YKUIW.js.map} +0 -0
- /package/dist/{chunk-VWDTV6SH.js.map → chunk-X6E7TT3A.js.map} +0 -0
- /package/dist/{chunk-D4OCJSLV.js.map → chunk-YDPAE2RD.js.map} +0 -0
- /package/dist/{chunk-QRKYABS6.js.map → chunk-YSR3DXGC.js.map} +0 -0
- /package/dist/{chunk-LBJBNWS2.js.map → chunk-YTHQHH5U.js.map} +0 -0
- /package/dist/{chunk-GN7HO2LA.js.map → chunk-ZPGHA7EO.js.map} +0 -0
- /package/dist/{chunk-FQSDU634.js.map → chunk-ZS4JADYY.js.map} +0 -0
- /package/dist/{forget-5TWK7CZT.js.map → forget-SZ2BXDGF.js.map} +0 -0
|
@@ -1,8 +1,46 @@
|
|
|
1
1
|
import assert from "node:assert/strict";
|
|
2
|
+
import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
|
|
3
|
+
import os from "node:os";
|
|
2
4
|
import path from "node:path";
|
|
3
5
|
import test from "node:test";
|
|
4
6
|
|
|
5
|
-
import {
|
|
7
|
+
import { EmbedHelper } from "./embed-helper.js";
|
|
8
|
+
import { OramaBackend, resolveOramaCollectionDbFilePath } from "./orama-backend.js";
|
|
9
|
+
import type { PluginConfig } from "../types.js";
|
|
10
|
+
|
|
11
|
+
function noEmbedHelper(): EmbedHelper {
|
|
12
|
+
return new EmbedHelper({} as PluginConfig);
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
async function writeMemoryFact(
|
|
16
|
+
memoryDir: string,
|
|
17
|
+
fileName: string,
|
|
18
|
+
content: string,
|
|
19
|
+
): Promise<void> {
|
|
20
|
+
await mkdir(path.join(memoryDir, "facts"), { recursive: true });
|
|
21
|
+
await writeFile(path.join(memoryDir, "facts", fileName), content, "utf8");
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
async function createBackend(
|
|
25
|
+
memoryDir: string,
|
|
26
|
+
dbPath: string,
|
|
27
|
+
cjkSegmentationEnabled?: boolean,
|
|
28
|
+
options: { update?: boolean } = {},
|
|
29
|
+
): Promise<OramaBackend> {
|
|
30
|
+
const backend = new OramaBackend({
|
|
31
|
+
dbPath,
|
|
32
|
+
collection: "openclaw-engram",
|
|
33
|
+
embedHelper: noEmbedHelper(),
|
|
34
|
+
memoryDir,
|
|
35
|
+
embeddingDimension: 4,
|
|
36
|
+
cjkSegmentationEnabled,
|
|
37
|
+
});
|
|
38
|
+
assert.equal(await backend.probe(), true);
|
|
39
|
+
if (options.update !== false) {
|
|
40
|
+
await backend.update();
|
|
41
|
+
}
|
|
42
|
+
return backend;
|
|
43
|
+
}
|
|
6
44
|
|
|
7
45
|
test("Orama collection filenames cannot escape dbPath", () => {
|
|
8
46
|
const dbPath = path.join("/tmp", "remnic-orama-db");
|
|
@@ -25,3 +63,172 @@ test("Orama collection filenames cannot escape dbPath", () => {
|
|
|
25
63
|
);
|
|
26
64
|
}
|
|
27
65
|
});
|
|
66
|
+
|
|
67
|
+
test("Orama lexical search finds a Japanese fact with no embeddings, at parity with English (issue #2187)", async () => {
|
|
68
|
+
const root = await mkdtemp(path.join(os.tmpdir(), "remnic-orama-cjk-"));
|
|
69
|
+
try {
|
|
70
|
+
await writeMemoryFact(
|
|
71
|
+
root,
|
|
72
|
+
"japanese-fact.md",
|
|
73
|
+
"---\nid: mem-japanese-fact\n---\n東京都庁の所在地は新宿区にある。最寄り駅は都庁前駅だ。",
|
|
74
|
+
);
|
|
75
|
+
await writeMemoryFact(
|
|
76
|
+
root,
|
|
77
|
+
"english-fact.md",
|
|
78
|
+
"---\nid: mem-english-fact\n---\nThe Tokyo Metropolitan Government Office is located in Shinjuku. The nearest station is Tochomae.",
|
|
79
|
+
);
|
|
80
|
+
|
|
81
|
+
const backend = await createBackend(root, path.join(root, "orama-db"));
|
|
82
|
+
|
|
83
|
+
const japanese = await backend.bm25Search("東京都庁の所在地");
|
|
84
|
+
const english = await backend.bm25Search("Tokyo Metropolitan Government Office location");
|
|
85
|
+
|
|
86
|
+
assert.equal(japanese.some((r) => r.docid === "mem-japanese-fact"), true);
|
|
87
|
+
assert.equal(english.some((r) => r.docid === "mem-english-fact"), true);
|
|
88
|
+
} finally {
|
|
89
|
+
await rm(root, { recursive: true, force: true });
|
|
90
|
+
}
|
|
91
|
+
});
|
|
92
|
+
|
|
93
|
+
test("Orama CJK tokenizer rebuilds stale pre-CJK indexes on first open (issue #2187)", async () => {
|
|
94
|
+
const root = await mkdtemp(path.join(os.tmpdir(), "remnic-orama-rebuild-"));
|
|
95
|
+
try {
|
|
96
|
+
await writeMemoryFact(
|
|
97
|
+
root,
|
|
98
|
+
"japanese-fact.md",
|
|
99
|
+
"---\nid: mem-japanese-fact\n---\n東京都庁の所在地は新宿区にある。",
|
|
100
|
+
);
|
|
101
|
+
await writeMemoryFact(
|
|
102
|
+
root,
|
|
103
|
+
"russian-fact.md",
|
|
104
|
+
"---\nid: mem-russian-fact\n---\nПравительство Токио находится в Синдзюку.",
|
|
105
|
+
);
|
|
106
|
+
await writeMemoryFact(
|
|
107
|
+
root,
|
|
108
|
+
"english-fact.md",
|
|
109
|
+
"---\nid: mem-english-fact\n---\nNebula-472 deploy finished in production.",
|
|
110
|
+
);
|
|
111
|
+
|
|
112
|
+
// Phase 1: index with the pre-CJK stock tokenizer (marker "english").
|
|
113
|
+
const legacyBackend = await createBackend(root, path.join(root, "orama-db"), false);
|
|
114
|
+
const legacyJapanese = await legacyBackend.bm25Search("東京都庁の所在地");
|
|
115
|
+
assert.equal(
|
|
116
|
+
legacyJapanese.some((r) => r.docid === "mem-japanese-fact"),
|
|
117
|
+
false,
|
|
118
|
+
"stock tokenizer must not match Japanese phrases",
|
|
119
|
+
);
|
|
120
|
+
const legacyEnglish = await legacyBackend.bm25Search("Nebula-472 deploy");
|
|
121
|
+
assert.equal(legacyEnglish.some((r) => r.docid === "mem-english-fact"), true);
|
|
122
|
+
|
|
123
|
+
// Phase 2: reopen with CJK segmentation on and NO update() call — the
|
|
124
|
+
// stale marker alone must trigger the in-place rebuild during probe().
|
|
125
|
+
const rebuiltBackend = await createBackend(root, path.join(root, "orama-db"), true, {
|
|
126
|
+
update: false,
|
|
127
|
+
});
|
|
128
|
+
const rebuiltJapanese = await rebuiltBackend.bm25Search("東京都庁の所在地");
|
|
129
|
+
assert.equal(rebuiltJapanese.some((r) => r.docid === "mem-japanese-fact"), true);
|
|
130
|
+
|
|
131
|
+
// Non-CJK non-Latin scripts are re-indexed too (whole-word terms), not
|
|
132
|
+
// just CJK/Thai.
|
|
133
|
+
const rebuiltRussian = await rebuiltBackend.bm25Search("Правительство Токио");
|
|
134
|
+
assert.equal(rebuiltRussian.some((r) => r.docid === "mem-russian-fact"), true);
|
|
135
|
+
|
|
136
|
+
// English-corpus behavior is unchanged by the rebuild.
|
|
137
|
+
const rebuiltEnglish = await rebuiltBackend.bm25Search("Nebula-472 deploy");
|
|
138
|
+
assert.equal(rebuiltEnglish.some((r) => r.docid === "mem-english-fact"), true);
|
|
139
|
+
} finally {
|
|
140
|
+
await rm(root, { recursive: true, force: true });
|
|
141
|
+
}
|
|
142
|
+
});
|
|
143
|
+
|
|
144
|
+
test("Orama English-only corpora are not re-indexed when the tokenizer version advances (issue #2187)", async () => {
|
|
145
|
+
const root = await mkdtemp(path.join(os.tmpdir(), "remnic-orama-english-"));
|
|
146
|
+
try {
|
|
147
|
+
await writeMemoryFact(
|
|
148
|
+
root,
|
|
149
|
+
"english-fact.md",
|
|
150
|
+
"---\nid: mem-english-fact\n---\nNebula-472 deploy finished in production.",
|
|
151
|
+
);
|
|
152
|
+
|
|
153
|
+
const legacyBackend = await createBackend(root, path.join(root, "orama-db"), false);
|
|
154
|
+
assert.equal(
|
|
155
|
+
(await legacyBackend.bm25Search("Nebula-472 deploy")).some((r) => r.docid === "mem-english-fact"),
|
|
156
|
+
true,
|
|
157
|
+
);
|
|
158
|
+
|
|
159
|
+
const upgradedBackend = await createBackend(root, path.join(root, "orama-db"), true);
|
|
160
|
+
assert.equal(
|
|
161
|
+
(await upgradedBackend.bm25Search("Nebula-472 deploy")).some((r) => r.docid === "mem-english-fact"),
|
|
162
|
+
true,
|
|
163
|
+
"English recall must survive the tokenizer upgrade",
|
|
164
|
+
);
|
|
165
|
+
} finally {
|
|
166
|
+
await rm(root, { recursive: true, force: true });
|
|
167
|
+
}
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
test("Orama downgrades persisted CJK indexes when segmentation is disabled (issue #2187)", async () => {
|
|
171
|
+
const root = await mkdtemp(path.join(os.tmpdir(), "remnic-orama-downgrade-"));
|
|
172
|
+
try {
|
|
173
|
+
await writeMemoryFact(
|
|
174
|
+
root,
|
|
175
|
+
"japanese-fact.md",
|
|
176
|
+
"---\nid: mem-japanese-fact\n---\n東京都庁の所在地は新宿区にある。",
|
|
177
|
+
);
|
|
178
|
+
|
|
179
|
+
// Phase 1: index with CJK segmentation on (marker "english+cjk-v1").
|
|
180
|
+
const cjkBackend = await createBackend(root, path.join(root, "orama-db"), true);
|
|
181
|
+
assert.equal(
|
|
182
|
+
(await cjkBackend.bm25Search("東京都庁の所在地")).some((r) => r.docid === "mem-japanese-fact"),
|
|
183
|
+
true,
|
|
184
|
+
"CJK tokenizer must match Japanese phrases before the downgrade",
|
|
185
|
+
);
|
|
186
|
+
|
|
187
|
+
// Phase 2: reopen with segmentation disabled and NO update() call — the
|
|
188
|
+
// persisted CJK marker alone must trigger the stock re-index.
|
|
189
|
+
const downgraded = await createBackend(root, path.join(root, "orama-db"), false, {
|
|
190
|
+
update: false,
|
|
191
|
+
});
|
|
192
|
+
assert.equal(
|
|
193
|
+
(await downgraded.bm25Search("東京都庁の所在地")).some((r) => r.docid === "mem-japanese-fact"),
|
|
194
|
+
false,
|
|
195
|
+
"stock tokenizer must not match Japanese phrases after the downgrade",
|
|
196
|
+
);
|
|
197
|
+
} finally {
|
|
198
|
+
await rm(root, { recursive: true, force: true });
|
|
199
|
+
}
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
test("Orama rebuild gate inspects every indexed string field, not only content (issue #2187)", async () => {
|
|
203
|
+
const root = await mkdtemp(path.join(os.tmpdir(), "remnic-orama-pathfield-"));
|
|
204
|
+
try {
|
|
205
|
+
// The only non-Latin text is the FILENAME, which lands in the indexed
|
|
206
|
+
// path field; the body is pure English.
|
|
207
|
+
await writeMemoryFact(
|
|
208
|
+
root,
|
|
209
|
+
"東京-fact.md",
|
|
210
|
+
"---\nid: mem-path-only\n---\nAn English body with no non-Latin characters.",
|
|
211
|
+
);
|
|
212
|
+
|
|
213
|
+
// Phase 1: pre-CJK stock index — the CJK filename never becomes a term.
|
|
214
|
+
const legacyBackend = await createBackend(root, path.join(root, "orama-db"), false);
|
|
215
|
+
assert.equal(
|
|
216
|
+
(await legacyBackend.bm25Search("東京")).some((r) => r.docid === "mem-path-only"),
|
|
217
|
+
false,
|
|
218
|
+
"stock tokenizer must not match the CJK filename",
|
|
219
|
+
);
|
|
220
|
+
|
|
221
|
+
// Phase 2: reopen with segmentation on and NO update() call — the
|
|
222
|
+
// path field alone must gate the in-place re-index.
|
|
223
|
+
const rebuilt = await createBackend(root, path.join(root, "orama-db"), true, {
|
|
224
|
+
update: false,
|
|
225
|
+
});
|
|
226
|
+
assert.equal(
|
|
227
|
+
(await rebuilt.bm25Search("東京")).some((r) => r.docid === "mem-path-only"),
|
|
228
|
+
true,
|
|
229
|
+
"non-Latin text in the path field must gate the rebuild",
|
|
230
|
+
);
|
|
231
|
+
} finally {
|
|
232
|
+
await rm(root, { recursive: true, force: true });
|
|
233
|
+
}
|
|
234
|
+
});
|
|
@@ -9,17 +9,28 @@ import {
|
|
|
9
9
|
type SearchQueryOptions,
|
|
10
10
|
type SearchResult,
|
|
11
11
|
} from "./port.js";
|
|
12
|
-
import type { EmbedHelper, EmbedProviderIdentity, EmbedWithProviderResult } from "./embed-helper.js";
|
|
13
|
-
import { scanMemoryDir } from "./document-scanner.js";
|
|
14
|
-
import { isSearchAborted, throwIfSearchAborted } from "./abort.js";
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
12
|
+
import type { EmbedHelper, EmbedProviderIdentity, EmbedWithProviderResult } from "./embed-helper.js";
|
|
13
|
+
import { scanMemoryDir } from "./document-scanner.js";
|
|
14
|
+
import { isSearchAborted, throwIfSearchAborted } from "./abort.js";
|
|
15
|
+
import {
|
|
16
|
+
containsNonLegacyTokenizerChars,
|
|
17
|
+
createCjkCapableTokenizer,
|
|
18
|
+
ORAMA_CJK_TOKENIZER_LANGUAGE,
|
|
19
|
+
} from "./orama-cjk-tokenizer.js";
|
|
20
|
+
|
|
21
|
+
export interface OramaBackendOptions {
|
|
22
|
+
dbPath: string;
|
|
23
|
+
collection: string;
|
|
24
|
+
embedHelper: EmbedHelper;
|
|
25
|
+
memoryDir: string;
|
|
26
|
+
embeddingDimension: number;
|
|
27
|
+
/**
|
|
28
|
+
* Segment space-free scripts (CJK/Thai) in the lexical index. Default
|
|
29
|
+
* `true`; `false` restores the stock English-only tokenizer (issue #2187).
|
|
30
|
+
*/
|
|
31
|
+
cjkSegmentationEnabled?: boolean;
|
|
32
|
+
}
|
|
33
|
+
|
|
23
34
|
|
|
24
35
|
const ORAMA_COLLECTION_FILENAME_PATTERN = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
|
|
25
36
|
|
|
@@ -57,6 +68,7 @@ export class OramaBackend implements SearchBackend {
|
|
|
57
68
|
private readonly embedHelper: EmbedHelper;
|
|
58
69
|
private readonly memoryDir: string;
|
|
59
70
|
private readonly embeddingDimension: number;
|
|
71
|
+
private readonly cjkSegmentationEnabled: boolean;
|
|
60
72
|
private available = false;
|
|
61
73
|
private db: any = null;
|
|
62
74
|
private oramaModule: any = null;
|
|
@@ -72,6 +84,7 @@ export class OramaBackend implements SearchBackend {
|
|
|
72
84
|
this.embedHelper = opts.embedHelper;
|
|
73
85
|
this.memoryDir = opts.memoryDir;
|
|
74
86
|
this.embeddingDimension = opts.embeddingDimension;
|
|
87
|
+
this.cjkSegmentationEnabled = opts.cjkSegmentationEnabled !== false;
|
|
75
88
|
}
|
|
76
89
|
|
|
77
90
|
async probe(): Promise<boolean> {
|
|
@@ -127,6 +140,11 @@ export class OramaBackend implements SearchBackend {
|
|
|
127
140
|
}
|
|
128
141
|
}
|
|
129
142
|
|
|
143
|
+
/**
|
|
144
|
+
* Full-text (FTS) search only. With the stock tokenizer this cannot match
|
|
145
|
+
* space-free CJK/Thai text lexically — hybrid or vector search provides
|
|
146
|
+
* the fallback for those scripts (issue #2187).
|
|
147
|
+
*/
|
|
130
148
|
async bm25Search(query: string, collection?: string, maxResults?: number, execution?: SearchExecutionOptions): Promise<SearchResult[]> {
|
|
131
149
|
if (isSearchAborted(execution)) return [];
|
|
132
150
|
const db = await this.ensureDbForCollection(collection ?? this.collection);
|
|
@@ -403,8 +421,11 @@ export class OramaBackend implements SearchBackend {
|
|
|
403
421
|
return this.db;
|
|
404
422
|
}
|
|
405
423
|
|
|
406
|
-
this.db = await this.
|
|
407
|
-
await this.
|
|
424
|
+
this.db = await this.applyTokenizerVersion(
|
|
425
|
+
await this.migrateLegacyVectorProviderSchema(
|
|
426
|
+
await this.persistModule.restore("json", raw),
|
|
427
|
+
this.collection,
|
|
428
|
+
),
|
|
408
429
|
this.collection,
|
|
409
430
|
);
|
|
410
431
|
return this.db;
|
|
@@ -426,8 +447,11 @@ export class OramaBackend implements SearchBackend {
|
|
|
426
447
|
return await this.createDb();
|
|
427
448
|
}
|
|
428
449
|
|
|
429
|
-
return await this.
|
|
430
|
-
await this.
|
|
450
|
+
return await this.applyTokenizerVersion(
|
|
451
|
+
await this.migrateLegacyVectorProviderSchema(
|
|
452
|
+
await this.persistModule.restore("json", raw),
|
|
453
|
+
collection,
|
|
454
|
+
),
|
|
431
455
|
collection,
|
|
432
456
|
);
|
|
433
457
|
}
|
|
@@ -442,7 +466,95 @@ export class OramaBackend implements SearchBackend {
|
|
|
442
466
|
vectorProvider: "string",
|
|
443
467
|
vector: `vector[${this.embeddingDimension}]`,
|
|
444
468
|
};
|
|
445
|
-
|
|
469
|
+
if (!this.cjkSegmentationEnabled) {
|
|
470
|
+
return await create({ schema });
|
|
471
|
+
}
|
|
472
|
+
return await create({
|
|
473
|
+
schema,
|
|
474
|
+
components: { tokenizer: createCjkCapableTokenizer(this.oramaModule) },
|
|
475
|
+
});
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
/**
|
|
479
|
+
* Attach the CJK-capable tokenizer to a restored index and rebuild the
|
|
480
|
+
* full-text index when it predates CJK tokenization. Orama persists
|
|
481
|
+
* `tokenizer.language` with the index, so a marker other than
|
|
482
|
+
* `ORAMA_CJK_TOKENIZER_LANGUAGE` identifies an index whose terms were
|
|
483
|
+
* produced by the stock English tokenizer. Corpora whose content the stock
|
|
484
|
+
* tokenizer and this tokenizer treat identically (pure legacy Latin) need
|
|
485
|
+
* no re-indexing — only the version marker is rewritten for them.
|
|
486
|
+
*/
|
|
487
|
+
private async applyTokenizerVersion(db: any, collection: string): Promise<any> {
|
|
488
|
+
const persistedLanguage =
|
|
489
|
+
db?.tokenizer?.language && typeof db.tokenizer.language === "string"
|
|
490
|
+
? db.tokenizer.language
|
|
491
|
+
: "";
|
|
492
|
+
if (this.cjkSegmentationEnabled) {
|
|
493
|
+
db.tokenizer = createCjkCapableTokenizer(this.oramaModule);
|
|
494
|
+
if (persistedLanguage === ORAMA_CJK_TOKENIZER_LANGUAGE) return db;
|
|
495
|
+
} else {
|
|
496
|
+
// Segmentation disabled: a database persisted under the CJK tokenizer
|
|
497
|
+
// keeps its CJK terms unless it is re-indexed under the stock one, so
|
|
498
|
+
// downgrade it instead of returning it unchanged (issue #2187 round 2).
|
|
499
|
+
if (persistedLanguage !== ORAMA_CJK_TOKENIZER_LANGUAGE) return db;
|
|
500
|
+
db.tokenizer = this.oramaModule.components.tokenizer.createTokenizer({ language: "english" });
|
|
501
|
+
}
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
try {
|
|
505
|
+
const { search: oramaSearch, count, update: oramaUpdate } = this.oramaModule;
|
|
506
|
+
const existingCount = await count(db);
|
|
507
|
+
if (existingCount === 0) {
|
|
508
|
+
await this.persistDbForCollection(db, collection);
|
|
509
|
+
return db;
|
|
510
|
+
}
|
|
511
|
+
const allHits = await oramaSearch(db, { term: "", limit: existingCount + 100 });
|
|
512
|
+
const hits = allHits.hits ?? [];
|
|
513
|
+
const hasNonLegacyContent = hits.some((hit: any) => {
|
|
514
|
+
const doc = this.getStoredDocument(db, hit);
|
|
515
|
+
// Orama full-text queries every string field by default, so every
|
|
516
|
+
// indexed string field gates the rebuild — non-Latin text that only
|
|
517
|
+
// rides in snippet/path/id would otherwise keep stale terms
|
|
518
|
+
// (issue #2187 review round 2).
|
|
519
|
+
for (const field of [doc.content, doc.snippet, doc.path, doc.id, doc.vectorProvider]) {
|
|
520
|
+
if (typeof field === "string" && containsNonLegacyTokenizerChars(field)) return true;
|
|
521
|
+
}
|
|
522
|
+
return false;
|
|
523
|
+
});
|
|
524
|
+
if (hasNonLegacyContent) {
|
|
525
|
+
// Re-index every document in place (Orama update is remove+insert,
|
|
526
|
+
// so content is re-tokenized; vectors are carried through). A single
|
|
527
|
+
// failing document must not block the rest of the corpus: it keeps
|
|
528
|
+
// its pre-upgrade terms, which is no worse than before the upgrade.
|
|
529
|
+
for (const hit of hits) {
|
|
530
|
+
try {
|
|
531
|
+
const doc = this.getStoredDocument(db, hit);
|
|
532
|
+
const vector = this.getStoredVector(db, hit, doc);
|
|
533
|
+
await oramaUpdate(db, hit.id, {
|
|
534
|
+
id: typeof doc.id === "string" && doc.id.length > 0 ? doc.id : String(hit.id),
|
|
535
|
+
path: typeof doc.path === "string" ? doc.path : "",
|
|
536
|
+
content: typeof doc.content === "string" ? doc.content : "",
|
|
537
|
+
snippet:
|
|
538
|
+
typeof doc.snippet === "string"
|
|
539
|
+
? doc.snippet
|
|
540
|
+
: typeof doc.content === "string"
|
|
541
|
+
? doc.content.slice(0, 200)
|
|
542
|
+
: "",
|
|
543
|
+
vectorProvider: typeof doc.vectorProvider === "string" ? doc.vectorProvider : "",
|
|
544
|
+
vector: vector ?? this.zeroVector(),
|
|
545
|
+
});
|
|
546
|
+
} catch (docErr) {
|
|
547
|
+
log.debug(
|
|
548
|
+
`OramaBackend tokenizer-version re-index skipped document ${hit.id} in ${collection}: ${docErr}`,
|
|
549
|
+
);
|
|
550
|
+
}
|
|
551
|
+
}
|
|
552
|
+
}
|
|
553
|
+
await this.persistDbForCollection(db, collection);
|
|
554
|
+
} catch (err) {
|
|
555
|
+
log.debug(`OramaBackend tokenizer-version rebuild for ${collection} failed: ${err}`);
|
|
556
|
+
}
|
|
557
|
+
return db;
|
|
446
558
|
}
|
|
447
559
|
|
|
448
560
|
private async migrateLegacyVectorProviderSchema(db: any, collection: string): Promise<any> {
|
|
@@ -526,8 +638,11 @@ export class OramaBackend implements SearchBackend {
|
|
|
526
638
|
await this.ensureModules();
|
|
527
639
|
const raw = await readFile(filePath, "utf-8");
|
|
528
640
|
const collection = path.basename(filePath, ".msp");
|
|
529
|
-
return await this.
|
|
530
|
-
await this.
|
|
641
|
+
return await this.applyTokenizerVersion(
|
|
642
|
+
await this.migrateLegacyVectorProviderSchema(
|
|
643
|
+
await this.persistModule.restore("json", raw),
|
|
644
|
+
collection,
|
|
645
|
+
),
|
|
531
646
|
collection,
|
|
532
647
|
);
|
|
533
648
|
} catch (err) {
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import test from "node:test";
|
|
3
|
+
|
|
4
|
+
import * as orama from "@orama/orama";
|
|
5
|
+
|
|
6
|
+
import {
|
|
7
|
+
containsNonLegacyTokenizerChars,
|
|
8
|
+
createCjkCapableTokenizer,
|
|
9
|
+
isSpaceFreeScriptChar,
|
|
10
|
+
ORAMA_CJK_TOKENIZER_LANGUAGE,
|
|
11
|
+
} from "./orama-cjk-tokenizer.js";
|
|
12
|
+
|
|
13
|
+
const stockTokenizer = orama.components.tokenizer.createTokenizer({ language: "english" });
|
|
14
|
+
const tokenizer = createCjkCapableTokenizer(orama);
|
|
15
|
+
|
|
16
|
+
test("CJK tokenizer matches the stock English tokenizer for legacy Latin input", () => {
|
|
17
|
+
for (const input of [
|
|
18
|
+
"The quick brown fox jumps over the lazy dog",
|
|
19
|
+
"Nebula-472 deploy finished; it's live (prod)!",
|
|
20
|
+
"snake_case camelCase 42.7%",
|
|
21
|
+
"",
|
|
22
|
+
]) {
|
|
23
|
+
assert.deepEqual(tokenizer.tokenize(input), stockTokenizer.tokenize(input), input);
|
|
24
|
+
}
|
|
25
|
+
});
|
|
26
|
+
|
|
27
|
+
test("CJK tokenizer expands Japanese runs into n-grams", () => {
|
|
28
|
+
const tokens = tokenizer.tokenize("東京都庁の所在地");
|
|
29
|
+
|
|
30
|
+
assert.equal(tokens.includes("東京都庁"), true);
|
|
31
|
+
assert.equal(tokens.includes("所在"), true);
|
|
32
|
+
assert.equal(tokens.includes("地"), true);
|
|
33
|
+
assert.equal(tokens.includes("東京都庁の所在地"), true);
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
test("CJK tokenizer reuses the recall segmentation n-gram sizes", () => {
|
|
37
|
+
// A 5-char run yields chars + 2/3/4-grams + the whole run.
|
|
38
|
+
const tokens = tokenizer.tokenize("用户喜欢深");
|
|
39
|
+
const expected = new Set(["用", "户", "喜", "欢", "深"]);
|
|
40
|
+
for (const size of [2, 3, 4]) {
|
|
41
|
+
for (let i = 0; i <= 5 - size; i++) {
|
|
42
|
+
expected.add("用户喜欢深".slice(i, i + size));
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
expected.add("用户喜欢深");
|
|
46
|
+
for (const token of expected) {
|
|
47
|
+
assert.equal(tokens.includes(token), true, token);
|
|
48
|
+
}
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
test("CJK tokenizer keeps Latin tokens intact inside mixed-script input", () => {
|
|
52
|
+
const tokens = tokenizer.tokenize("Nebula-472 東京都庁");
|
|
53
|
+
|
|
54
|
+
assert.equal(tokens.includes("nebula-472"), true);
|
|
55
|
+
assert.equal(tokens.includes("東京都庁"), true);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test("CJK tokenizer n-grams Thai runs", () => {
|
|
59
|
+
const tokens = tokenizer.tokenize("กรุงเทพมหานคร");
|
|
60
|
+
|
|
61
|
+
assert.equal(tokens.includes("กรุง"), true);
|
|
62
|
+
assert.equal(tokens.includes("งเทพ"), true);
|
|
63
|
+
assert.equal(tokens.includes("กรุงเทพมหานคร"), true);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
test("CJK tokenizer keeps Hangul and other alphabetic scripts as whole words", () => {
|
|
67
|
+
const tokens = tokenizer.tokenize("사용자 설정 Привет мир");
|
|
68
|
+
|
|
69
|
+
assert.equal(tokens.includes("사용자"), true);
|
|
70
|
+
assert.equal(tokens.includes("설정"), true);
|
|
71
|
+
assert.equal(tokens.includes("사"), false);
|
|
72
|
+
assert.equal(tokens.includes("привет"), true);
|
|
73
|
+
assert.equal(tokens.includes("мир"), true);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test("CJK tokenizer persists a versioned language marker", () => {
|
|
77
|
+
assert.equal(tokenizer.language, ORAMA_CJK_TOKENIZER_LANGUAGE);
|
|
78
|
+
assert.equal(typeof tokenizer.tokenize, "function");
|
|
79
|
+
assert.ok(tokenizer.normalizationCache instanceof Map);
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
test("isSpaceFreeScriptChar covers CJK and Thai only", () => {
|
|
83
|
+
assert.equal(isSpaceFreeScriptChar("東"), true);
|
|
84
|
+
assert.equal(isSpaceFreeScriptChar("あ"), true);
|
|
85
|
+
assert.equal(isSpaceFreeScriptChar("ア"), true);
|
|
86
|
+
assert.equal(isSpaceFreeScriptChar("ก"), true);
|
|
87
|
+
assert.equal(isSpaceFreeScriptChar("사"), false);
|
|
88
|
+
assert.equal(isSpaceFreeScriptChar("a"), false);
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
test("CJK tokenizer returns no tokens for non-string input", () => {
|
|
92
|
+
assert.deepEqual(tokenizer.tokenize(undefined as unknown as string), []);
|
|
93
|
+
assert.deepEqual(tokenizer.tokenize(null as unknown as string), []);
|
|
94
|
+
assert.deepEqual(tokenizer.tokenize(42 as unknown as string), []);
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
test("rebuild gate flags only tokenization-changing characters", () => {
|
|
98
|
+
// Word material outside the stock keep-set changes tokens (whole words,
|
|
99
|
+
// n-grams, attached marks) — the gate must flag it.
|
|
100
|
+
assert.equal(containsNonLegacyTokenizerChars("naïve"), true);
|
|
101
|
+
assert.equal(containsNonLegacyTokenizerChars("東京都庁"), true);
|
|
102
|
+
assert.equal(containsNonLegacyTokenizerChars("사용자"), true);
|
|
103
|
+
assert.equal(containsNonLegacyTokenizerChars("cafe\u0301"), true);
|
|
104
|
+
// Separators in BOTH tokenizers — punctuation and symbols — leave the
|
|
105
|
+
// stock token stream unchanged, so no re-index is warranted.
|
|
106
|
+
assert.equal(containsNonLegacyTokenizerChars("plain — text „quoted” · done"), false);
|
|
107
|
+
assert.equal(containsNonLegacyTokenizerChars("Nebula-472 deploy (prod)"), false);
|
|
108
|
+
});
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
import { expandUnsegmentableRecallNGrams, isUnsegmentableRecallChar } from "../recall-tokenization.js";
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Language marker persisted inside Orama index files (`.msp`). Orama persists
|
|
5
|
+
* `tokenizer.language` with the index, so this doubles as the tokenization
|
|
6
|
+
* version: `OramaBackend` detects any other marker on restore and rebuilds
|
|
7
|
+
* the full-text index (issue #2187). Bump the suffix whenever tokenization
|
|
8
|
+
* changes.
|
|
9
|
+
*/
|
|
10
|
+
export const ORAMA_CJK_TOKENIZER_LANGUAGE = "english+cjk-v1";
|
|
11
|
+
|
|
12
|
+
/** Minimal structural surface of `@orama/orama` this module consumes. */
|
|
13
|
+
export interface OramaTokenizerComponents {
|
|
14
|
+
components: {
|
|
15
|
+
tokenizer: {
|
|
16
|
+
createTokenizer: (config: { language: string }) => OramaTokenizer;
|
|
17
|
+
};
|
|
18
|
+
};
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export interface OramaTokenizer {
|
|
22
|
+
language: string;
|
|
23
|
+
normalizationCache: Map<string, string>;
|
|
24
|
+
tokenize: (raw: string, language?: string, prop?: string, withCache?: boolean) => string[];
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
const THAI_SCRIPT_CHAR = /[\p{Script=Thai}]/u;
|
|
28
|
+
const WORD_CHAR = /[\p{L}\p{N}\p{M}]/u;
|
|
29
|
+
const COMBINING_MARK = /\p{M}/u;
|
|
30
|
+
const WHITESPACE_CHAR = /\s/u;
|
|
31
|
+
/**
|
|
32
|
+
* Chars the stock English tokenizer keeps inside a token. Input made only of
|
|
33
|
+
* these (plus whitespace) is delegated to the stock tokenizer byte-for-byte so
|
|
34
|
+
* existing English indexes stay term-compatible.
|
|
35
|
+
*/
|
|
36
|
+
const LEGACY_TOKENIZER_CHAR = /[A-Za-zàèéìòóù0-9_'-]/u;
|
|
37
|
+
|
|
38
|
+
/**
|
|
39
|
+
* Space-free scripts: CJK per the shared recall segmentation strategy, plus
|
|
40
|
+
* Thai. Runs in these scripts are indexed as character n-grams so phrase
|
|
41
|
+
* queries match without word boundaries.
|
|
42
|
+
*/
|
|
43
|
+
export function isSpaceFreeScriptChar(char: string): boolean {
|
|
44
|
+
return isUnsegmentableRecallChar(char) || THAI_SCRIPT_CHAR.test(char);
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* True when tokenizing the value differs from the stock English tokenizer —
|
|
49
|
+
* i.e. the value contains characters outside the stock keep-set that this
|
|
50
|
+
* tokenizer still indexes as token material: word characters (whole-word
|
|
51
|
+
* non-ASCII scripts), combining marks (attached to runs), and space-free
|
|
52
|
+
* scripts (CJK/Thai n-grams). Symbols and punctuation that are separators in
|
|
53
|
+
* BOTH tokenizers do not count. `OramaBackend` uses this to decide whether a
|
|
54
|
+
* stale pre-CJK index needs re-indexing on upgrade.
|
|
55
|
+
*/
|
|
56
|
+
export function containsNonLegacyTokenizerChars(value: string): boolean {
|
|
57
|
+
for (const ch of value) {
|
|
58
|
+
if (LEGACY_TOKENIZER_CHAR.test(ch) || WHITESPACE_CHAR.test(ch)) continue;
|
|
59
|
+
if (WORD_CHAR.test(ch) || COMBINING_MARK.test(ch) || isSpaceFreeScriptChar(ch)) {
|
|
60
|
+
return true;
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
return false;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Orama tokenizer component that segments space-free scripts (CJK, Thai).
|
|
68
|
+
*
|
|
69
|
+
* - CJK/Thai runs expand to the same n-gram set the recall query-side
|
|
70
|
+
* tokenizer (`recall-tokenization.ts`) produces, so index-side and
|
|
71
|
+
* query-side tokens agree.
|
|
72
|
+
* - Other non-Latin scripts (Hangul, Cyrillic, Greek, Arabic, ...) are kept
|
|
73
|
+
* as whole words instead of being dropped by the English splitter.
|
|
74
|
+
* - Legacy Latin content tokenizes exactly like the stock English tokenizer;
|
|
75
|
+
* content with other characters keeps every stock token and additionally
|
|
76
|
+
* indexes non-ASCII words whole (e.g. "über" adds "über" beside "ber").
|
|
77
|
+
*/
|
|
78
|
+
export function createCjkCapableTokenizer(oramaModule: OramaTokenizerComponents): OramaTokenizer {
|
|
79
|
+
const base = oramaModule.components.tokenizer.createTokenizer({ language: "english" });
|
|
80
|
+
|
|
81
|
+
const tokenize = (raw: string, _language?: string, prop?: string, withCache?: boolean): string[] => {
|
|
82
|
+
if (typeof raw !== "string") return [];
|
|
83
|
+
const normalized = raw.normalize("NFC");
|
|
84
|
+
|
|
85
|
+
if (!containsNonLegacyTokenizerChars(normalized)) {
|
|
86
|
+
return base.tokenize(normalized, "english", prop, withCache);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// The stock tokenizer treats every non-legacy char as a separator, so its
|
|
90
|
+
// output over the raw input already carries the legacy-Latin tokens
|
|
91
|
+
// (e.g. the "nebula-472" in "Nebula-472 東京都庁").
|
|
92
|
+
const tokens = base.tokenize(normalized, "english", prop, withCache);
|
|
93
|
+
const seen = new Set(tokens);
|
|
94
|
+
|
|
95
|
+
const pushToken = (token: string) => {
|
|
96
|
+
if (token && !seen.has(token)) {
|
|
97
|
+
seen.add(token);
|
|
98
|
+
tokens.push(token);
|
|
99
|
+
}
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
let spaceFreeRun = "";
|
|
103
|
+
let wordRun = "";
|
|
104
|
+
const flushSpaceFreeRun = () => {
|
|
105
|
+
if (!spaceFreeRun) return;
|
|
106
|
+
for (const token of expandUnsegmentableRecallNGrams(spaceFreeRun)) {
|
|
107
|
+
pushToken(token);
|
|
108
|
+
}
|
|
109
|
+
spaceFreeRun = "";
|
|
110
|
+
};
|
|
111
|
+
const flushWordRun = () => {
|
|
112
|
+
if (!wordRun) return;
|
|
113
|
+
pushToken(wordRun.toLowerCase());
|
|
114
|
+
wordRun = "";
|
|
115
|
+
};
|
|
116
|
+
|
|
117
|
+
for (const ch of normalized) {
|
|
118
|
+
if (isSpaceFreeScriptChar(ch)) {
|
|
119
|
+
flushWordRun();
|
|
120
|
+
spaceFreeRun += ch;
|
|
121
|
+
} else if (COMBINING_MARK.test(ch)) {
|
|
122
|
+
if (spaceFreeRun) spaceFreeRun += ch;
|
|
123
|
+
else if (wordRun) wordRun += ch;
|
|
124
|
+
} else if (WORD_CHAR.test(ch) && !LEGACY_TOKENIZER_CHAR.test(ch)) {
|
|
125
|
+
flushSpaceFreeRun();
|
|
126
|
+
wordRun += ch;
|
|
127
|
+
} else {
|
|
128
|
+
flushSpaceFreeRun();
|
|
129
|
+
flushWordRun();
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
flushSpaceFreeRun();
|
|
133
|
+
flushWordRun();
|
|
134
|
+
|
|
135
|
+
return tokens;
|
|
136
|
+
};
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
language: ORAMA_CJK_TOKENIZER_LANGUAGE,
|
|
140
|
+
normalizationCache: base.normalizationCache,
|
|
141
|
+
tokenize,
|
|
142
|
+
};
|
|
143
|
+
}
|