@remnic/core 9.5.0 → 9.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/access-admin-ops-surface.d.ts +6 -6
- package/dist/access-admin-ops-surface.js +12 -12
- package/dist/access-boundary.d.ts +6 -6
- package/dist/access-boundary.js +13 -13
- package/dist/access-cli.js +36 -36
- package/dist/access-http.d.ts +6 -6
- package/dist/access-http.js +17 -17
- package/dist/access-identity-continuity-surface.d.ts +5 -5
- package/dist/access-identity-continuity-surface.js +12 -12
- package/dist/access-lcm-surface.d.ts +6 -6
- package/dist/access-lcm-surface.js +12 -12
- package/dist/access-mcp.d.ts +6 -6
- package/dist/access-mcp.js +16 -16
- package/dist/access-observe-write-surface.d.ts +6 -6
- package/dist/access-observe-write-surface.js +12 -12
- package/dist/access-operations-batch.js +14 -14
- package/dist/access-operations.d.ts +7 -7
- package/dist/access-operations.js +15 -15
- package/dist/access-recall-surface.d.ts +6 -6
- package/dist/access-recall-surface.js +12 -12
- package/dist/access-schema.d.ts +4 -4
- package/dist/{access-service-DZX0sYAv.d.ts → access-service-CXJaql3A.d.ts} +4 -4
- package/dist/access-service.d.ts +6 -6
- package/dist/access-service.js +12 -12
- package/dist/access-surface-catalog.d.ts +6 -6
- package/dist/action-confidence.d.ts +1 -1
- package/dist/active-memory-bridge.d.ts +1 -1
- package/dist/active-recall.d.ts +1 -1
- package/dist/active-recall.js +1 -1
- package/dist/{auto-sync-G6IU7L6C.js → auto-sync-4A5YBWYO.js} +3 -3
- package/dist/behavior-learner.d.ts +1 -1
- package/dist/behavior-signals.d.ts +1 -1
- package/dist/bootstrap.d.ts +5 -5
- package/dist/briefing.d.ts +2 -2
- package/dist/briefing.js +4 -4
- package/dist/buffer-surprise-report.d.ts +1 -1
- package/dist/buffer.d.ts +2 -2
- package/dist/calibration.d.ts +1 -1
- package/dist/capabilities.d.ts +1 -1
- package/dist/{catalog-DoH2b0OR.d.ts → catalog-C9cu67Ig.d.ts} +1 -1
- package/dist/causal-behavior.d.ts +1 -1
- package/dist/causal-consolidation.d.ts +1 -1
- package/dist/causal-consolidation.js +5 -5
- package/dist/causal-trajectory-graph.d.ts +1 -1
- package/dist/{chunk-H5PFR5ZT.js → chunk-23JFRB73.js} +2 -2
- package/dist/{chunk-BGMSRN6I.js → chunk-3XIY7MDQ.js} +3 -3
- package/dist/{chunk-L2QJQSHL.js → chunk-4GNEDXDI.js} +2 -2
- package/dist/{chunk-MFDJ63U5.js → chunk-4LV23CHK.js} +2 -2
- package/dist/{chunk-AER6MT24.js → chunk-6SXVCD7W.js} +2 -2
- package/dist/{chunk-2SFEFR5I.js → chunk-72WG5QKN.js} +2 -2
- package/dist/{chunk-7BAJAXGB.js → chunk-7MW3CVLD.js} +2 -2
- package/dist/{chunk-SNNIUDRS.js → chunk-AYZJID4S.js} +2 -2
- package/dist/{chunk-M7XQSUBB.js → chunk-B3ABVQ36.js} +116 -9
- package/dist/chunk-B3ABVQ36.js.map +1 -0
- package/dist/{chunk-LNHJ32NK.js → chunk-BEM4D2QG.js} +3 -3
- package/dist/{chunk-AOYJVCFN.js → chunk-CYWA2CXR.js} +2 -2
- package/dist/{chunk-2JFDHI2K.js → chunk-DEG5ULFJ.js} +2 -2
- package/dist/{chunk-L6V3WJ55.js → chunk-DS7MM2ES.js} +3 -3
- package/dist/{chunk-CXEL3Y22.js → chunk-FKCRGLRD.js} +184 -29
- package/dist/{chunk-CXEL3Y22.js.map → chunk-FKCRGLRD.js.map} +1 -1
- package/dist/{chunk-TI7IHDBQ.js → chunk-H2DXDQCQ.js} +5 -5
- package/dist/{chunk-KMF67QDQ.js → chunk-HLN7RROI.js} +4 -4
- package/dist/{chunk-D2XYYU5Q.js → chunk-J3XHPFEU.js} +2 -2
- package/dist/{chunk-ODSYJXDU.js → chunk-JSNVWNGX.js} +2 -2
- package/dist/{chunk-7Z2DZNWN.js → chunk-KL4UAXUY.js} +43 -2
- package/dist/{chunk-7Z2DZNWN.js.map → chunk-KL4UAXUY.js.map} +1 -1
- package/dist/{chunk-4WNLSSHL.js → chunk-LADFWYHY.js} +4 -4
- package/dist/{chunk-UBGAOYBP.js → chunk-LR5DZ56F.js} +2 -2
- package/dist/{chunk-5K5VEDFY.js → chunk-N3JDIRBW.js} +2 -2
- package/dist/{chunk-2B7DP7AP.js → chunk-P454KXD6.js} +93 -15
- package/dist/chunk-P454KXD6.js.map +1 -0
- package/dist/{chunk-VWEWK2BM.js → chunk-PCLIWBGD.js} +3 -3
- package/dist/{chunk-TV7C3HCB.js → chunk-PNLFN2PZ.js} +2 -2
- package/dist/{chunk-STCCFQU5.js → chunk-PVUJ6WVV.js} +6 -6
- package/dist/{chunk-ITWTKT2R.js → chunk-RGK27ELN.js} +2 -2
- package/dist/{chunk-II6WHCCW.js → chunk-SIRPLF54.js} +2 -2
- package/dist/{chunk-UDCDGP6A.js → chunk-SVOHGUMI.js} +1188 -112
- package/dist/chunk-SVOHGUMI.js.map +1 -0
- package/dist/{chunk-NFTABV56.js → chunk-TLUDFPZL.js} +2 -2
- package/dist/{chunk-I74ZEBMA.js → chunk-UFUKHGZD.js} +4 -4
- package/dist/{chunk-L4UZBJNM.js → chunk-UTRJFDP2.js} +2 -2
- package/dist/{chunk-5GPPACXK.js → chunk-VSMRDBFT.js} +2 -1
- package/dist/{chunk-SNBEVJBX.js → chunk-XCL57M6Q.js} +3 -3
- package/dist/{chunk-DZSAXE3T.js → chunk-Y7UNNAZD.js} +2 -2
- package/dist/{chunk-XN3TUUX2.js → chunk-YQB32INI.js} +2 -2
- package/dist/{cli-1dghPCO2.d.ts → cli-CqdPA6z9.d.ts} +3 -3
- package/dist/cli.d.ts +7 -7
- package/dist/cli.js +29 -29
- package/dist/compounding/engine.d.ts +2 -2
- package/dist/compounding/engine.js +4 -4
- package/dist/compounding/preference-consolidator.d.ts +1 -1
- package/dist/compression-optimizer.d.ts +1 -1
- package/dist/config.d.ts +1 -1
- package/dist/config.js +1 -1
- package/dist/connectors/codex-materialize-runner.d.ts +1 -1
- package/dist/connectors/codex-materialize-runner.js +4 -4
- package/dist/connectors/codex-materialize.d.ts +1 -1
- package/dist/connectors/index.d.ts +1 -1
- package/dist/connectors/index.js +4 -4
- package/dist/consolidation-provenance-check.d.ts +2 -2
- package/dist/consolidation-undo.d.ts +2 -2
- package/dist/contradiction/index.d.ts +2 -2
- package/dist/conversation-index/backend.d.ts +1 -1
- package/dist/conversation-index/chunker.d.ts +1 -1
- package/dist/conversation-index/faiss-adapter.d.ts +1 -1
- package/dist/conversation-index/indexer.d.ts +1 -1
- package/dist/conversation-index/search.d.ts +1 -1
- package/dist/day-summary.d.ts +1 -1
- package/dist/delinearize.d.ts +1 -1
- package/dist/direct-answer-wiring.d.ts +1 -1
- package/dist/direct-answer.d.ts +1 -1
- package/dist/embedding-fallback.d.ts +1 -1
- package/dist/enrichment/index.d.ts +1 -1
- package/dist/entity-retrieval.d.ts +2 -2
- package/dist/entity-retrieval.js +4 -4
- package/dist/entity-schema.d.ts +1 -1
- package/dist/explicit-capture.d.ts +5 -5
- package/dist/extraction-faithfulness.d.ts +1 -1
- package/dist/extraction-judge-telemetry.d.ts +1 -1
- package/dist/extraction-judge-training.d.ts +1 -1
- package/dist/extraction-judge.d.ts +1 -1
- package/dist/extraction.d.ts +1 -1
- package/dist/fallback-llm.d.ts +1 -1
- package/dist/graph-dashboard-diff.d.ts +1 -1
- package/dist/graph-dashboard-key.d.ts +1 -1
- package/dist/graph-dashboard-parser.d.ts +1 -1
- package/dist/graph-edge-reinforcement.d.ts +1 -1
- package/dist/graph-snapshot.d.ts +1 -1
- package/dist/graph.d.ts +1 -1
- package/dist/identity-continuity.d.ts +1 -1
- package/dist/importance.d.ts +1 -1
- package/dist/index.d.ts +268 -17
- package/dist/index.js +80 -38
- package/dist/intent.d.ts +1 -1
- package/dist/lcm/engine.d.ts +1 -1
- package/dist/lcm/index.d.ts +1 -1
- package/dist/lcm/tools.d.ts +1 -1
- package/dist/lifecycle.d.ts +1 -1
- package/dist/live-connectors-runner.d.ts +1 -1
- package/dist/local-llm.d.ts +1 -1
- package/dist/local-model-endpoint.d.ts +1 -1
- package/dist/maintenance/memory-governance.d.ts +1 -1
- package/dist/maintenance/memory-governance.js +4 -4
- package/dist/maintenance/rebuild-memory-lifecycle-ledger.js +4 -4
- package/dist/maintenance/rebuild-memory-projection.js +5 -5
- package/dist/mcp-memory-inspector-app.d.ts +6 -6
- package/dist/memory-action-policy.d.ts +1 -1
- package/dist/memory-cache.d.ts +1 -1
- package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
- package/dist/memory-projection-store.d.ts +1 -1
- package/dist/memory-provenance.d.ts +1 -1
- package/dist/memory-worth-outcomes.d.ts +2 -2
- package/dist/models-json.d.ts +1 -1
- package/dist/namespaces/migrate.d.ts +3 -3
- package/dist/namespaces/migrate.js +11 -11
- package/dist/namespaces/principal.d.ts +1 -1
- package/dist/namespaces/search.d.ts +1 -1
- package/dist/namespaces/search.js +7 -7
- package/dist/namespaces/storage.d.ts +3 -3
- package/dist/namespaces/storage.js +4 -4
- package/dist/native-knowledge.d.ts +1 -1
- package/dist/operator-toolkit.d.ts +2 -2
- package/dist/operator-toolkit.js +15 -15
- package/dist/orchestration/compression-guideline-coordinator.d.ts +2 -2
- package/dist/orchestration/maintenance.d.ts +2 -2
- package/dist/orchestration/maintenance.js +6 -6
- package/dist/{orchestrator-Dh3WSW_O.d.ts → orchestrator-CWFrT8AF.d.ts} +91 -60
- package/dist/orchestrator.d.ts +5 -5
- package/dist/orchestrator.js +36 -36
- package/dist/patterns-cli.d.ts +1 -1
- package/dist/policy-runtime.d.ts +1 -1
- package/dist/provenance.d.ts +1 -1
- package/dist/{qmd-DoUVxqND.d.ts → qmd--fnPMK60.d.ts} +1 -1
- package/dist/qmd-preflight.d.ts +2 -2
- package/dist/qmd-recall-cache.d.ts +1 -1
- package/dist/qmd.d.ts +2 -2
- package/dist/recall-disclosure-escalation.d.ts +1 -1
- package/dist/recall-explain-renderer.d.ts +1 -1
- package/dist/recall-planner-llm.d.ts +1 -1
- package/dist/recall-state.d.ts +1 -1
- package/dist/recall-tag-filter.d.ts +1 -1
- package/dist/recall-timings.d.ts +1 -1
- package/dist/recall-xray-cli.d.ts +1 -1
- package/dist/recall-xray-renderer.d.ts +1 -1
- package/dist/recall-xray.d.ts +1 -1
- package/dist/resolve-auth-token.d.ts +1 -1
- package/dist/resume-bundles.js +2 -2
- package/dist/retrieval-agents.d.ts +2 -2
- package/dist/retrieval-tiers.d.ts +1 -1
- package/dist/routing/engine.d.ts +1 -1
- package/dist/routing/store.d.ts +1 -1
- package/dist/schemas.d.ts +22 -22
- package/dist/search/document-scanner.js +2 -2
- package/dist/search/embed-helper.d.ts +1 -1
- package/dist/search/factory.d.ts +1 -1
- package/dist/search/factory.js +6 -6
- package/dist/search/index.d.ts +1 -1
- package/dist/search/index.js +6 -6
- package/dist/search/lancedb-backend.d.ts +1 -1
- package/dist/search/lancedb-backend.js +3 -3
- package/dist/search/meilisearch-backend.d.ts +1 -1
- package/dist/search/meilisearch-backend.js +3 -3
- package/dist/search/noop-backend.d.ts +1 -1
- package/dist/search/orama-backend.d.ts +1 -1
- package/dist/search/orama-backend.js +3 -3
- package/dist/search/port.d.ts +1 -1
- package/dist/search/remote-backend.d.ts +1 -1
- package/dist/{semantic-consolidation-Dof1865o.d.ts → semantic-consolidation-O5lMZ4LG.d.ts} +1 -1
- package/dist/semantic-consolidation.d.ts +2 -2
- package/dist/semantic-consolidation.js +5 -5
- package/dist/semantic-rule-promotion.js +4 -4
- package/dist/semantic-rule-verifier.d.ts +1 -1
- package/dist/semantic-rule-verifier.js +4 -4
- package/dist/session-observer-bands.d.ts +1 -1
- package/dist/session-observer-state.d.ts +1 -1
- package/dist/shared-context/manager.d.ts +1 -1
- package/dist/signal.d.ts +1 -1
- package/dist/storage-C1maCNbc.d.ts +1591 -0
- package/dist/storage.d.ts +5 -1239
- package/dist/storage.js +3 -3
- package/dist/summarizer.d.ts +1 -1
- package/dist/summary-snapshot.d.ts +1 -1
- package/dist/temporal-supersession.d.ts +2 -2
- package/dist/temporal-validity.d.ts +1 -1
- package/dist/threading.d.ts +1 -1
- package/dist/tier-migration.d.ts +2 -2
- package/dist/tier-routing.d.ts +1 -1
- package/dist/topics.d.ts +1 -1
- package/dist/transcript.d.ts +1 -1
- package/dist/transfer/types.d.ts +12 -12
- package/dist/trust-score-stage.d.ts +1 -1
- package/dist/trust-score.d.ts +1 -1
- package/dist/{types-C7HUgOuA.d.ts → types-CGgcNWCG.d.ts} +31 -1
- package/dist/types.d.ts +1 -1
- package/dist/utility-runtime.d.ts +1 -1
- package/dist/verified-recall.js +4 -4
- package/package.json +2 -2
- package/src/cli-wearables-fuse.test.ts +197 -0
- package/src/cli.ts +25 -0
- package/src/storage.ts +16 -0
- package/src/wearables/cli.test.ts +102 -0
- package/src/wearables/cli.ts +65 -0
- package/src/wearables/config.test.ts +32 -0
- package/src/wearables/config.ts +52 -0
- package/src/wearables/day-store.test.ts +246 -0
- package/src/wearables/day-store.ts +272 -3
- package/src/wearables/fusion/cluster.ts +230 -0
- package/src/wearables/fusion/fuse.ts +218 -0
- package/src/wearables/fusion/fusion.test.ts +2575 -0
- package/src/wearables/fusion/index.ts +50 -0
- package/src/wearables/fusion/reconcile.ts +831 -0
- package/src/wearables/fusion/reconstruct.ts +251 -0
- package/src/wearables/fusion/store.ts +480 -0
- package/src/wearables/fusion/types.ts +228 -0
- package/src/wearables/index.ts +41 -0
- package/src/wearables/service.test.ts +1030 -2
- package/src/wearables/service.ts +258 -16
- package/src/wearables/speakers.test.ts +128 -0
- package/src/wearables/speakers.ts +58 -4
- package/src/wearables/storage-io.test.ts +146 -0
- package/src/wearables/types.ts +31 -0
- package/dist/chunk-2B7DP7AP.js.map +0 -1
- package/dist/chunk-M7XQSUBB.js.map +0 -1
- package/dist/chunk-UDCDGP6A.js.map +0 -1
- /package/dist/{auto-sync-G6IU7L6C.js.map → auto-sync-4A5YBWYO.js.map} +0 -0
- /package/dist/{chunk-H5PFR5ZT.js.map → chunk-23JFRB73.js.map} +0 -0
- /package/dist/{chunk-BGMSRN6I.js.map → chunk-3XIY7MDQ.js.map} +0 -0
- /package/dist/{chunk-L2QJQSHL.js.map → chunk-4GNEDXDI.js.map} +0 -0
- /package/dist/{chunk-MFDJ63U5.js.map → chunk-4LV23CHK.js.map} +0 -0
- /package/dist/{chunk-AER6MT24.js.map → chunk-6SXVCD7W.js.map} +0 -0
- /package/dist/{chunk-2SFEFR5I.js.map → chunk-72WG5QKN.js.map} +0 -0
- /package/dist/{chunk-7BAJAXGB.js.map → chunk-7MW3CVLD.js.map} +0 -0
- /package/dist/{chunk-SNNIUDRS.js.map → chunk-AYZJID4S.js.map} +0 -0
- /package/dist/{chunk-LNHJ32NK.js.map → chunk-BEM4D2QG.js.map} +0 -0
- /package/dist/{chunk-AOYJVCFN.js.map → chunk-CYWA2CXR.js.map} +0 -0
- /package/dist/{chunk-2JFDHI2K.js.map → chunk-DEG5ULFJ.js.map} +0 -0
- /package/dist/{chunk-L6V3WJ55.js.map → chunk-DS7MM2ES.js.map} +0 -0
- /package/dist/{chunk-TI7IHDBQ.js.map → chunk-H2DXDQCQ.js.map} +0 -0
- /package/dist/{chunk-KMF67QDQ.js.map → chunk-HLN7RROI.js.map} +0 -0
- /package/dist/{chunk-D2XYYU5Q.js.map → chunk-J3XHPFEU.js.map} +0 -0
- /package/dist/{chunk-ODSYJXDU.js.map → chunk-JSNVWNGX.js.map} +0 -0
- /package/dist/{chunk-4WNLSSHL.js.map → chunk-LADFWYHY.js.map} +0 -0
- /package/dist/{chunk-UBGAOYBP.js.map → chunk-LR5DZ56F.js.map} +0 -0
- /package/dist/{chunk-5K5VEDFY.js.map → chunk-N3JDIRBW.js.map} +0 -0
- /package/dist/{chunk-VWEWK2BM.js.map → chunk-PCLIWBGD.js.map} +0 -0
- /package/dist/{chunk-TV7C3HCB.js.map → chunk-PNLFN2PZ.js.map} +0 -0
- /package/dist/{chunk-STCCFQU5.js.map → chunk-PVUJ6WVV.js.map} +0 -0
- /package/dist/{chunk-ITWTKT2R.js.map → chunk-RGK27ELN.js.map} +0 -0
- /package/dist/{chunk-II6WHCCW.js.map → chunk-SIRPLF54.js.map} +0 -0
- /package/dist/{chunk-NFTABV56.js.map → chunk-TLUDFPZL.js.map} +0 -0
- /package/dist/{chunk-I74ZEBMA.js.map → chunk-UFUKHGZD.js.map} +0 -0
- /package/dist/{chunk-L4UZBJNM.js.map → chunk-UTRJFDP2.js.map} +0 -0
- /package/dist/{chunk-5GPPACXK.js.map → chunk-VSMRDBFT.js.map} +0 -0
- /package/dist/{chunk-SNBEVJBX.js.map → chunk-XCL57M6Q.js.map} +0 -0
- /package/dist/{chunk-DZSAXE3T.js.map → chunk-Y7UNNAZD.js.map} +0 -0
- /package/dist/{chunk-XN3TUUX2.js.map → chunk-YQB32INI.js.map} +0 -0
|
@@ -34,6 +34,127 @@ export const WEARABLES_DIR_NAME = "wearables";
|
|
|
34
34
|
|
|
35
35
|
const DATE_PATTERN = /^\d{4}-\d{2}-\d{2}$/;
|
|
36
36
|
|
|
37
|
+
/**
|
|
38
|
+
* Day-transcript body serialization format version. Folded into
|
|
39
|
+
* `hashTranscriptBody` so a version bump invalidates files written by an
|
|
40
|
+
* older serializer and forces an idempotent rewrite. Decoders gate on this:
|
|
41
|
+
* only bodies whose parsed meta carries >= this version have escape
|
|
42
|
+
* sequences decoded; legacy bodies (absent / older) are left byte-for-byte
|
|
43
|
+
* unchanged so a literal two-character `\n`/`\r` in a pre-escaper
|
|
44
|
+
* transcript is never altered (issue #1849).
|
|
45
|
+
*/
|
|
46
|
+
export const TRANSCRIPT_FORMAT_VERSION = 2;
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Result of parsing a rendered transcript segment line.
|
|
50
|
+
*/
|
|
51
|
+
export interface TranscriptSegmentMatch {
|
|
52
|
+
/** Speaker label text between the `**` delimiters. */
|
|
53
|
+
label: string;
|
|
54
|
+
/** Clock text between the `[` `]` delimiters. */
|
|
55
|
+
clock: string;
|
|
56
|
+
/** Segment text after the colon. */
|
|
57
|
+
text: string;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Test a single character code point against the JavaScript regex `\s`
|
|
62
|
+
* whitespace class without invoking a regex. Used by the linear segment
|
|
63
|
+
* line parser so no quantified pattern is ever evaluated on untrusted
|
|
64
|
+
* transcript content (CodeQL polynomial-redos, issue #1849).
|
|
65
|
+
*/
|
|
66
|
+
function isRegexWhitespace(code: number): boolean {
|
|
67
|
+
return (
|
|
68
|
+
(code >= 0x09 && code <= 0x0d) ||
|
|
69
|
+
code === 0x20 ||
|
|
70
|
+
code === 0xa0 ||
|
|
71
|
+
code === 0x1680 ||
|
|
72
|
+
(code >= 0x2000 && code <= 0x200a) ||
|
|
73
|
+
code === 0x2028 ||
|
|
74
|
+
code === 0x2029 ||
|
|
75
|
+
code === 0x202f ||
|
|
76
|
+
code === 0x205f ||
|
|
77
|
+
code === 0x3000 ||
|
|
78
|
+
code === 0xfeff
|
|
79
|
+
);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Parse a rendered transcript segment line (`**label** [clock]: text`)
|
|
84
|
+
* into its three components using a linear, bounded scan — a CodeQL-safe
|
|
85
|
+
* replacement for the polynomial-risk regex that previously matched this
|
|
86
|
+
* format. Returns null when the line is not a segment line.
|
|
87
|
+
*
|
|
88
|
+
* The scan finds the FIRST occurrence of each structural delimiter
|
|
89
|
+
* (`**` after the label, `]` after the clock) exactly as the original
|
|
90
|
+
* non-greedy regex did, and verifies the fixed separator characters
|
|
91
|
+
* (`\s`, `[`, `:`, `\s`) by single-character code-point comparison.
|
|
92
|
+
* No quantified regex is evaluated on the line (#1849).
|
|
93
|
+
*
|
|
94
|
+
* Escaped labels (formatVersion >= 2) never contain an unescaped `**`,
|
|
95
|
+
* `[`, or `]`, and rendered clocks are always `HH:MM` / `--:--`, so the
|
|
96
|
+
* first delimiter is always the correct one for every well-formed
|
|
97
|
+
* transcript. Legacy unescaped labels are used verbatim by the caller
|
|
98
|
+
* and parsed identically here.
|
|
99
|
+
*/
|
|
100
|
+
export function parseTranscriptSegmentLine(
|
|
101
|
+
line: string,
|
|
102
|
+
): TranscriptSegmentMatch | null {
|
|
103
|
+
// Must start with '**'.
|
|
104
|
+
if (!line.startsWith("**")) return null;
|
|
105
|
+
|
|
106
|
+
// Find the closing '**' of the label: the first '**' at index >= 3
|
|
107
|
+
// (label is at least 1 char) that is immediately followed by \s and
|
|
108
|
+
// then '['. This mirrors the non-greedy regex: the engine tries the
|
|
109
|
+
// shortest label first and extends only when the subsequent fixed
|
|
110
|
+
// delimiters do not line up. An escaped label ending with '\*' is
|
|
111
|
+
// correctly handled because the spurious '**' (escape-star + close-
|
|
112
|
+
// star) is not followed by \s and the scan continues.
|
|
113
|
+
let labelEnd = -1;
|
|
114
|
+
for (let i = 3; i < line.length; i++) {
|
|
115
|
+
if (
|
|
116
|
+
line.charCodeAt(i) !== 0x2a /* '*' */ ||
|
|
117
|
+
i + 3 >= line.length ||
|
|
118
|
+
line.charCodeAt(i + 1) !== 0x2a /* '*' */ ||
|
|
119
|
+
!isRegexWhitespace(line.charCodeAt(i + 2)) ||
|
|
120
|
+
line.charCodeAt(i + 3) !== 0x5b /* '[' */
|
|
121
|
+
) {
|
|
122
|
+
continue;
|
|
123
|
+
}
|
|
124
|
+
labelEnd = i;
|
|
125
|
+
break;
|
|
126
|
+
}
|
|
127
|
+
if (labelEnd === -1) return null;
|
|
128
|
+
|
|
129
|
+
const label = line.slice(2, labelEnd);
|
|
130
|
+
// Clock content starts after the closing '**', the \s, and the '['.
|
|
131
|
+
const clockStart = labelEnd + 4;
|
|
132
|
+
|
|
133
|
+
// Find the closing ']' of the clock: the first ']' at index >=
|
|
134
|
+
// clockStart + 1 (clock is at least 1 char) that is immediately
|
|
135
|
+
// followed by ':' and \s — matching the non-greedy regex.
|
|
136
|
+
let clockEnd = -1;
|
|
137
|
+
for (let i = clockStart + 1; i < line.length; i++) {
|
|
138
|
+
if (
|
|
139
|
+
line.charCodeAt(i) !== 0x5d /* ']' */ ||
|
|
140
|
+
i + 2 >= line.length ||
|
|
141
|
+
line.charCodeAt(i + 1) !== 0x3a /* ':' */ ||
|
|
142
|
+
!isRegexWhitespace(line.charCodeAt(i + 2))
|
|
143
|
+
) {
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
clockEnd = i;
|
|
147
|
+
break;
|
|
148
|
+
}
|
|
149
|
+
if (clockEnd === -1) return null;
|
|
150
|
+
|
|
151
|
+
return {
|
|
152
|
+
label,
|
|
153
|
+
clock: line.slice(clockStart, clockEnd),
|
|
154
|
+
text: line.slice(clockEnd + 3),
|
|
155
|
+
};
|
|
156
|
+
}
|
|
157
|
+
|
|
37
158
|
export function isValidTranscriptDate(date: string): boolean {
|
|
38
159
|
if (!DATE_PATTERN.test(date)) return false;
|
|
39
160
|
const parsed = new Date(`${date}T00:00:00Z`);
|
|
@@ -41,7 +162,14 @@ export function isValidTranscriptDate(date: string): boolean {
|
|
|
41
162
|
}
|
|
42
163
|
|
|
43
164
|
export function hashTranscriptBody(body: string): string {
|
|
44
|
-
|
|
165
|
+
// The version prefix makes the hash an idempotency key for the body AS
|
|
166
|
+
// SERIALIZED under the current format: when the escape encoding changes
|
|
167
|
+
// (version bump) every existing file's stored hash no longer matches and
|
|
168
|
+
// the pipeline rewrites it with the new marker — no separate migration
|
|
169
|
+
// pass needed (issue #1849).
|
|
170
|
+
return createHash("sha256")
|
|
171
|
+
.update(`v${TRANSCRIPT_FORMAT_VERSION}\n${body}`, "utf-8")
|
|
172
|
+
.digest("hex");
|
|
45
173
|
}
|
|
46
174
|
|
|
47
175
|
function formatClockTime(iso: string | undefined, timezone: string): string {
|
|
@@ -69,6 +197,135 @@ function conversationDurationMinutes(conversation: WearableConversation): number
|
|
|
69
197
|
return (end - start) / 60_000;
|
|
70
198
|
}
|
|
71
199
|
|
|
200
|
+
/**
|
|
201
|
+
* Escape segment text for the line-based markdown format so that embedded
|
|
202
|
+
* newlines, carriage returns, and backslashes survive the serialize →
|
|
203
|
+
* reconstruct round-trip losslessly. Reversed by `unescapeSegmentText`
|
|
204
|
+
* (this module — the single decode primitive shared with the fusion
|
|
205
|
+
* reconstruct path and every view/search/index surface).
|
|
206
|
+
*
|
|
207
|
+
* Without this, a segment whose text contains a newline is split across
|
|
208
|
+
* multiple physical lines; the reconstruct path treats each line
|
|
209
|
+
* independently and the continuation is silently dropped (or worse, a
|
|
210
|
+
* continuation line that looks like a heading/clock is mis-parsed).
|
|
211
|
+
*/
|
|
212
|
+
export function escapeSegmentText(text: string): string {
|
|
213
|
+
return text
|
|
214
|
+
.replace(/\\/g, "\\\\")
|
|
215
|
+
.replace(/\n/g, "\\n")
|
|
216
|
+
.replace(/\r/g, "\\r");
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* Reverse `escapeSegmentText`. Unknown escape sequences (a lone
|
|
221
|
+
* backslash followed by a character the escaper never emits) are passed
|
|
222
|
+
* through literally so legacy transcripts that never went through the
|
|
223
|
+
* escaper still round-trip their original text. This is the SINGLE
|
|
224
|
+
* decode primitive: the fusion reconstruct path and every user-facing
|
|
225
|
+
* view/search/index surface call it so escaped storage never leaks to
|
|
226
|
+
* display (#1849).
|
|
227
|
+
*/
|
|
228
|
+
export function unescapeSegmentText(text: string): string {
|
|
229
|
+
return text.replace(/\\(.)/g, (_match, ch: string) => {
|
|
230
|
+
if (ch === "n") return "\n";
|
|
231
|
+
if (ch === "r") return "\r";
|
|
232
|
+
if (ch === "\\") return "\\";
|
|
233
|
+
return "\\" + ch;
|
|
234
|
+
});
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
/**
|
|
238
|
+
* Escape a speaker label for the line-based markdown format so that
|
|
239
|
+
* markdown-delimiter characters in an arbitrary user/provider label
|
|
240
|
+
* (`**`, `[`, `]`), the escape character (`\`), and embedded newlines/
|
|
241
|
+
* carriage returns cannot break `parseTranscriptSegmentLine` parsing.
|
|
242
|
+
* Without this, a label containing `**` (or `** [clock]:`) can make the
|
|
243
|
+
* non-greedy delimiter match land on the wrong `**` and mis-parse the
|
|
244
|
+
* clock/text — or fail to parse the line at all. Reversed by
|
|
245
|
+
* `unescapeSpeakerLabel` (this module). Legacy transcripts whose labels
|
|
246
|
+
* were never escaped still parse: `unescapeSpeakerLabel` is a no-op on
|
|
247
|
+
* any label without a backslash escape sequence (#1849).
|
|
248
|
+
*/
|
|
249
|
+
export function escapeSpeakerLabel(label: string): string {
|
|
250
|
+
return label
|
|
251
|
+
.replace(/\\/g, "\\\\")
|
|
252
|
+
.replace(/\*/g, "\\*")
|
|
253
|
+
.replace(/\[/g, "\\[")
|
|
254
|
+
.replace(/\]/g, "\\]")
|
|
255
|
+
.replace(/\n/g, "\\n")
|
|
256
|
+
.replace(/\r/g, "\\r");
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Reverse `escapeSpeakerLabel`. Unknown escape sequences (a backslash
|
|
261
|
+
* followed by a character the escaper never emits) are passed through
|
|
262
|
+
* literally so legacy transcripts with raw labels still round-trip
|
|
263
|
+
* their original text, mirroring `unescapeSegmentText` for segment text.
|
|
264
|
+
*/
|
|
265
|
+
export function unescapeSpeakerLabel(label: string): string {
|
|
266
|
+
return label.replace(/\\(.)/g, (_match, ch: string) => {
|
|
267
|
+
if (ch === "\\") return "\\";
|
|
268
|
+
if (ch === "*") return "*";
|
|
269
|
+
if (ch === "[") return "[";
|
|
270
|
+
if (ch === "]") return "]";
|
|
271
|
+
if (ch === "n") return "\n";
|
|
272
|
+
if (ch === "r") return "\r";
|
|
273
|
+
return "\\" + ch;
|
|
274
|
+
});
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
/**
|
|
278
|
+
* Whether a parsed day-transcript's body was written by the escape-aware
|
|
279
|
+
* serializer (meta carries `formatVersion` >= `TRANSCRIPT_FORMAT_VERSION`).
|
|
280
|
+
* Legacy bodies (no marker, or an older version) must NOT be decoded: their
|
|
281
|
+
* literal two-character `\n`, `\r`, and lone backslashes are original
|
|
282
|
+
* content, not escape sequences (issue #1849).
|
|
283
|
+
*/
|
|
284
|
+
export function bodyIsEscaped(
|
|
285
|
+
meta: { formatVersion?: number } | null | undefined,
|
|
286
|
+
): boolean {
|
|
287
|
+
return (
|
|
288
|
+
meta !== null &&
|
|
289
|
+
meta !== undefined &&
|
|
290
|
+
typeof meta.formatVersion === "number" &&
|
|
291
|
+
meta.formatVersion >= TRANSCRIPT_FORMAT_VERSION
|
|
292
|
+
);
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Decode the escaped segment text AND speaker label of a stored
|
|
297
|
+
* transcript body into the ORIGINAL forms for user-facing
|
|
298
|
+
* view/search/index surfaces. Only segment lines are touched: the text
|
|
299
|
+
* is decoded via `unescapeSegmentText` and the label via
|
|
300
|
+
* `unescapeSpeakerLabel`; headings, locations, and the H1 header are
|
|
301
|
+
* NOT escaped on write and pass through unchanged, so a title or
|
|
302
|
+
* location containing a literal backslash is never altered. The fusion
|
|
303
|
+
* reconstruct path does NOT use this — it decodes each segment once
|
|
304
|
+
* during parse — so callers feeding bodies back into reconstruct must
|
|
305
|
+
* pass the RAW stored body, not a decoded one, to avoid double-decoding.
|
|
306
|
+
*
|
|
307
|
+
* FORMAT-AWARE (issue #1849): when `escaped` is falsy (the default) the
|
|
308
|
+
* body is returned byte-for-byte unchanged. Only bodies written by the
|
|
309
|
+
* escape-aware serializer (`escaped = true`, derived from
|
|
310
|
+
* `bodyIsEscaped(meta)`) are decoded, so a legacy transcript's literal
|
|
311
|
+
* two-character `\n`/`\r` or lone backslash is never altered. Callers
|
|
312
|
+
* that have the parsed meta should pass `bodyIsEscaped(meta)`.
|
|
313
|
+
*/
|
|
314
|
+
export function decodeTranscriptBody(body: string, escaped = false): string {
|
|
315
|
+
if (!escaped) return body;
|
|
316
|
+
const lines = body.split("\n");
|
|
317
|
+
for (let i = 0; i < lines.length; i++) {
|
|
318
|
+
const match = parseTranscriptSegmentLine(lines[i]);
|
|
319
|
+
if (match === null) continue;
|
|
320
|
+
const { label, clock, text: rawText } = match;
|
|
321
|
+
const decodedText = unescapeSegmentText(rawText);
|
|
322
|
+
const decodedLabel = unescapeSpeakerLabel(label);
|
|
323
|
+
if (decodedText === rawText && decodedLabel === label) continue;
|
|
324
|
+
lines[i] = `**${decodedLabel}** [${clock}]: ${decodedText}`;
|
|
325
|
+
}
|
|
326
|
+
return lines.join("\n");
|
|
327
|
+
}
|
|
328
|
+
|
|
72
329
|
/** Compose the markdown body (no frontmatter) for one source/day. */
|
|
73
330
|
export function composeDayTranscriptBody(
|
|
74
331
|
sourceId: string,
|
|
@@ -101,7 +358,9 @@ export function composeDayTranscriptBody(
|
|
|
101
358
|
for (const segment of conversation.segments) {
|
|
102
359
|
const { label } = resolveSpeaker(sourceId, segment, registry);
|
|
103
360
|
const at = formatClockTime(segment.startIso, timezone);
|
|
104
|
-
lines.push(
|
|
361
|
+
lines.push(
|
|
362
|
+
`**${escapeSpeakerLabel(label)}** [${at}]: ${escapeSegmentText(segment.text)}`,
|
|
363
|
+
);
|
|
105
364
|
}
|
|
106
365
|
lines.push("");
|
|
107
366
|
}
|
|
@@ -132,6 +391,7 @@ export function composeDayTranscriptMeta(
|
|
|
132
391
|
durationMinutes,
|
|
133
392
|
contentHash: hashTranscriptBody(body),
|
|
134
393
|
syncedAt,
|
|
394
|
+
formatVersion: TRANSCRIPT_FORMAT_VERSION,
|
|
135
395
|
};
|
|
136
396
|
}
|
|
137
397
|
|
|
@@ -142,6 +402,9 @@ export function serializeDayTranscript(
|
|
|
142
402
|
): string {
|
|
143
403
|
const lines: string[] = ["---"];
|
|
144
404
|
lines.push(`kind: ${meta.kind}`);
|
|
405
|
+
if (typeof meta.formatVersion === "number") {
|
|
406
|
+
lines.push(`formatVersion: ${meta.formatVersion}`);
|
|
407
|
+
}
|
|
145
408
|
lines.push(`source: ${JSON.stringify(meta.source)}`);
|
|
146
409
|
lines.push(`date: ${JSON.stringify(meta.date)}`);
|
|
147
410
|
lines.push(`timezone: ${JSON.stringify(meta.timezone)}`);
|
|
@@ -204,9 +467,15 @@ export function parseDayTranscript(raw: string): WearableDayTranscript | null {
|
|
|
204
467
|
|
|
205
468
|
const meta: WearableDayTranscriptMeta = {
|
|
206
469
|
kind: "wearable-transcript",
|
|
470
|
+
...(scalars.has("formatVersion")
|
|
471
|
+
? { formatVersion: parseNonNegativeInt(scalars.get("formatVersion")) }
|
|
472
|
+
: {}),
|
|
207
473
|
source,
|
|
208
474
|
date,
|
|
209
|
-
timezone
|
|
475
|
+
// Preserve a missing timezone field as "" rather than coercing to a
|
|
476
|
+
// default: the fusion identity guard must see "no resolvable tz id"
|
|
477
|
+
// and skip, never silently match a known zone.
|
|
478
|
+
timezone: scalars.get("timezone") ?? "",
|
|
210
479
|
conversationCount: parseNonNegativeInt(scalars.get("conversationCount")),
|
|
211
480
|
segmentCount: parseNonNegativeInt(scalars.get("segmentCount")),
|
|
212
481
|
speakers,
|
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Wearable cross-source fusion — deterministic conversation clustering.
|
|
3
|
+
*
|
|
4
|
+
* Conversations from multiple sources recording the same real-world event
|
|
5
|
+
* overlap or sit within a proximity gap. This module groups them into
|
|
6
|
+
* clusters that are fully deterministic: the only inputs are start/end
|
|
7
|
+
* times, source ids, and conversation ids (no randomness, no LLM).
|
|
8
|
+
*
|
|
9
|
+
* Cross-source time proximity BRIDGES the same real-world conversation
|
|
10
|
+
* recorded by DIFFERENT sources. A single source's own distinct
|
|
11
|
+
* conversations are never merged by time proximity alone — that would
|
|
12
|
+
* collapse the source's conversation boundaries. Two same-source
|
|
13
|
+
* conversations may end up in one cluster only when a different-source
|
|
14
|
+
* conversation within the gap corroborates that they are part of the same
|
|
15
|
+
* real-world event (a cross-source chain/bridge).
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import type { FusionConversationInput } from "./types.js";
|
|
19
|
+
|
|
20
|
+
/** Default max gap (ms) between two conversations to merge into one cluster. */
|
|
21
|
+
export const DEFAULT_PROXIMITY_GAP_MS = 5 * 60_000;
|
|
22
|
+
|
|
23
|
+
interface Interval {
|
|
24
|
+
start: number;
|
|
25
|
+
end: number;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** The latest extent of a set of segments as an epoch-ms value + its ISO.
|
|
29
|
+
* Each segment contributes its OWN extent — its `endIso` when known,
|
|
30
|
+
* otherwise its `startIso` — and the result is the maximum. Reconstructed
|
|
31
|
+
* `--:--` transcripts rebuild segments with only a start, so this reaches
|
|
32
|
+
* the last segment start; segment ISOs are already cross-midnight-rolled by
|
|
33
|
+
* the reconstruct layer, so the derived end is consistent with the segment
|
|
34
|
+
* timeline. Returns undefined when no segment carries a parseable extent.
|
|
35
|
+
*
|
|
36
|
+
* This is the SINGLE shared primitive behind every end/interval derivation
|
|
37
|
+
* in the fusion pipeline (cluster interval, fused conversation end): when a
|
|
38
|
+
* conversation-level end is missing, the window falls back to the latest
|
|
39
|
+
* segment extent instead of collapsing to a zero-length point at the start.
|
|
40
|
+
*/
|
|
41
|
+
export interface SegmentExtent {
|
|
42
|
+
/** Epoch ms of the latest extent. */
|
|
43
|
+
ms: number;
|
|
44
|
+
/** ISO string that produced `ms` (a segment endIso, else its startIso). */
|
|
45
|
+
iso: string;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function maxSegmentExtent(
|
|
49
|
+
segments: readonly { endIso?: string; startIso?: string }[],
|
|
50
|
+
): SegmentExtent | undefined {
|
|
51
|
+
let best: SegmentExtent | undefined;
|
|
52
|
+
for (const seg of segments) {
|
|
53
|
+
const iso = seg.endIso ?? seg.startIso;
|
|
54
|
+
if (iso === undefined) continue;
|
|
55
|
+
const ms = Date.parse(iso);
|
|
56
|
+
if (!Number.isFinite(ms)) continue;
|
|
57
|
+
if (best === undefined || ms > best.ms) {
|
|
58
|
+
best = { ms, iso };
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
return best;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Resolve a conversation's effective [start, end] window in epoch ms.
|
|
66
|
+
*
|
|
67
|
+
* When the conversation carries no explicit end (`endIso` undefined — the
|
|
68
|
+
* reconstructed form of a stored transcript whose heading end renders as
|
|
69
|
+
* `--:--`), the window is derived from the segments so it spans the actual
|
|
70
|
+
* utterances instead of collapsing to a zero-length point at the start:
|
|
71
|
+
*
|
|
72
|
+
* 1. conversation `endIso`, when known;
|
|
73
|
+
* 2. the latest segment END (`segment.endIso`);
|
|
74
|
+
* 3. the latest segment START (`segment.startIso`) — reconstructed
|
|
75
|
+
* transcripts rebuild segments with only a start, so the window must
|
|
76
|
+
* reach the last segment start;
|
|
77
|
+
* 4. the conversation start itself (no segments at all).
|
|
78
|
+
*
|
|
79
|
+
* Each segment contributes its own extent (end when known, otherwise its
|
|
80
|
+
* start); the window end is the maximum extent. Segment ISOs are already
|
|
81
|
+
* rolled for cross-midnight by the reconstruct layer, so the derived end is
|
|
82
|
+
* consistent with the segment timeline. The end is clamped to `>= start` so
|
|
83
|
+
* the window is always valid regardless of input anomalies.
|
|
84
|
+
*/
|
|
85
|
+
function effectiveInterval(conversation: FusionConversationInput): Interval {
|
|
86
|
+
const start = Date.parse(conversation.startIso);
|
|
87
|
+
const endRaw =
|
|
88
|
+
conversation.endIso !== undefined ? Date.parse(conversation.endIso) : NaN;
|
|
89
|
+
// No conversation-level end ("--:--"): derive the window from the latest
|
|
90
|
+
// SEGMENT extent via the shared primitive, so every end/interval derivation
|
|
91
|
+
// in the pipeline uses ONE coherent implementation. Falling back only to
|
|
92
|
+
// `start` would produce a zero/negative-length window that drops or
|
|
93
|
+
// mis-clusters the conversation (issue #1810).
|
|
94
|
+
const end = Number.isFinite(endRaw)
|
|
95
|
+
? (endRaw as number)
|
|
96
|
+
: (maxSegmentExtent(conversation.segments)?.ms ?? start);
|
|
97
|
+
// Clamp universally so the window is always non-negative. An explicit
|
|
98
|
+
// conversation end that precedes the start (malformed/cross-day input)
|
|
99
|
+
// would otherwise yield a negative-length window that shrinks the merge
|
|
100
|
+
// horizon and can split a within-gap cluster.
|
|
101
|
+
return { start, end: Math.max(end, start) };
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
function compareConversation(
|
|
105
|
+
a: FusionConversationInput,
|
|
106
|
+
b: FusionConversationInput,
|
|
107
|
+
): number {
|
|
108
|
+
if (a.source !== b.source) return a.source < b.source ? -1 : 1;
|
|
109
|
+
if (a.conversationId !== b.conversationId) {
|
|
110
|
+
return a.conversationId < b.conversationId ? -1 : 1;
|
|
111
|
+
}
|
|
112
|
+
return 0;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Cluster conversations across sources for one day. Returns clusters in
|
|
117
|
+
* chronological order; each cluster is sorted by (start, source, id).
|
|
118
|
+
* Conversations whose start time is unparseable are emitted as
|
|
119
|
+
* single-element clusters at the tail (sorted by source then id) so no
|
|
120
|
+
* data is lost.
|
|
121
|
+
*
|
|
122
|
+
* ## Cross-source bridging rule (issue #1849)
|
|
123
|
+
*
|
|
124
|
+
* Time proximity is meant to bridge the SAME real-world conversation
|
|
125
|
+
* recorded by DIFFERENT sources — not to merge one source's own distinct
|
|
126
|
+
* conversations. Two conversations from the SAME source are therefore
|
|
127
|
+
* never directly joined by time proximity alone; they may end up in the
|
|
128
|
+
* same cluster only through a cross-source chain where a different-source
|
|
129
|
+
* conversation within the gap corroborates the bridge. This preserves the
|
|
130
|
+
* source's own conversation boundaries when no other source corroborates
|
|
131
|
+
* a merge.
|
|
132
|
+
*
|
|
133
|
+
* Concretely, a union-find pass links every pair of conversations that
|
|
134
|
+
* (a) come from DIFFERENT sources and (b) sit within the proximity gap
|
|
135
|
+
* (the later start ≤ the earlier end + gap). The transitive closure then
|
|
136
|
+
* groups same-source conversations that are bridged by at least one
|
|
137
|
+
* intervening different-source conversation. A single source with two
|
|
138
|
+
* near-back-to-back conversations and no other source recording the same
|
|
139
|
+
* window stays in two separate clusters.
|
|
140
|
+
*/
|
|
141
|
+
export function clusterConversations(
|
|
142
|
+
inputs: readonly FusionConversationInput[],
|
|
143
|
+
proximityGapMs: number = DEFAULT_PROXIMITY_GAP_MS,
|
|
144
|
+
): FusionConversationInput[][] {
|
|
145
|
+
const finite: Array<{ conv: FusionConversationInput; interval: Interval }> = [];
|
|
146
|
+
const noStart: FusionConversationInput[] = [];
|
|
147
|
+
for (const conv of inputs) {
|
|
148
|
+
const interval = effectiveInterval(conv);
|
|
149
|
+
if (Number.isFinite(interval.start)) {
|
|
150
|
+
finite.push({ conv, interval });
|
|
151
|
+
} else {
|
|
152
|
+
noStart.push(conv);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// Deterministic processing + output ordering: (start, source, id).
|
|
157
|
+
finite.sort((a, b) => {
|
|
158
|
+
if (a.interval.start !== b.interval.start) {
|
|
159
|
+
return a.interval.start - b.interval.start;
|
|
160
|
+
}
|
|
161
|
+
return compareConversation(a.conv, b.conv);
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
const n = finite.length;
|
|
165
|
+
|
|
166
|
+
// Union-find so that same-source pairs are never directly joined by time
|
|
167
|
+
// proximity alone; they may share a cluster only through a cross-source
|
|
168
|
+
// bridge (transitive closure over different-source proximity edges).
|
|
169
|
+
const parent = new Array<number>(n);
|
|
170
|
+
for (let i = 0; i < n; i++) parent[i] = i;
|
|
171
|
+
|
|
172
|
+
const find = (x: number): number => {
|
|
173
|
+
let root = x;
|
|
174
|
+
while (parent[root] !== root) root = parent[root];
|
|
175
|
+
// Path compression.
|
|
176
|
+
let cur = x;
|
|
177
|
+
while (parent[cur] !== root) {
|
|
178
|
+
const next = parent[cur]!;
|
|
179
|
+
parent[cur] = root;
|
|
180
|
+
cur = next;
|
|
181
|
+
}
|
|
182
|
+
return root;
|
|
183
|
+
};
|
|
184
|
+
|
|
185
|
+
const union = (a: number, b: number): void => {
|
|
186
|
+
const ra = find(a);
|
|
187
|
+
const rb = find(b);
|
|
188
|
+
if (ra !== rb) parent[ra] = rb;
|
|
189
|
+
};
|
|
190
|
+
|
|
191
|
+
// Link every cross-source pair within the proximity gap. `finite` is
|
|
192
|
+
// sorted by start, so once j's start exceeds i's end + gap, every later
|
|
193
|
+
// j is also out of reach — break early.
|
|
194
|
+
for (let i = 0; i < n; i++) {
|
|
195
|
+
const endGap = finite[i]!.interval.end + proximityGapMs;
|
|
196
|
+
for (let j = i + 1; j < n; j++) {
|
|
197
|
+
if (finite[j]!.interval.start > endGap) break;
|
|
198
|
+
// Only DIFFERENT sources may be directly joined by proximity.
|
|
199
|
+
if (finite[i]!.conv.source === finite[j]!.conv.source) continue;
|
|
200
|
+
union(i, j);
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
// Group conversations by union-find root. Because `finite` is sorted and
|
|
205
|
+
// we iterate in index order, each group's conversations stay in (start,
|
|
206
|
+
// source, id) order, and groups are discovered in chronological order
|
|
207
|
+
// (the first root seen belongs to the earliest-starting conversation).
|
|
208
|
+
const clusters: FusionConversationInput[][] = [];
|
|
209
|
+
const rootToCluster = new Map<number, number>();
|
|
210
|
+
for (let i = 0; i < n; i++) {
|
|
211
|
+
const root = find(i);
|
|
212
|
+
let idx = rootToCluster.get(root);
|
|
213
|
+
if (idx === undefined) {
|
|
214
|
+
idx = clusters.length;
|
|
215
|
+
rootToCluster.set(root, idx);
|
|
216
|
+
clusters.push([]);
|
|
217
|
+
}
|
|
218
|
+
clusters[idx]!.push(finite[i]!.conv);
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
// Conversations with no parseable start can't be ordered in time, so
|
|
222
|
+
// each becomes its own cluster, appended after the time-ordered ones.
|
|
223
|
+
// Sort for determinism (source, id).
|
|
224
|
+
noStart.sort(compareConversation);
|
|
225
|
+
for (const conv of noStart) {
|
|
226
|
+
clusters.push([conv]);
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
return clusters;
|
|
230
|
+
}
|