@remnic/core 9.53.0 → 9.54.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (254) hide show
  1. package/dist/access-admin-ops-surface.d.ts +11 -11
  2. package/dist/access-authorization-probe.d.ts +11 -11
  3. package/dist/access-boundary.d.ts +11 -11
  4. package/dist/access-cli.js +7 -7
  5. package/dist/access-coding-context-resolution.d.ts +1 -1
  6. package/dist/access-extraction-force-flush.d.ts +11 -11
  7. package/dist/access-health-types.d.ts +1 -1
  8. package/dist/access-http-lcm-compaction.d.ts +11 -11
  9. package/dist/access-http-lifecycle-flush.d.ts +11 -11
  10. package/dist/access-http-offline-stream.d.ts +11 -11
  11. package/dist/access-http.d.ts +11 -11
  12. package/dist/access-identity-continuity-surface.d.ts +9 -9
  13. package/dist/access-lcm-surface.d.ts +11 -11
  14. package/dist/access-mcp.d.ts +11 -11
  15. package/dist/access-memory-search-fanout.d.ts +2 -2
  16. package/dist/{access-namespace-preflight-Ds6CGlCP.d.ts → access-namespace-preflight-C7n4tgWC.d.ts} +1 -1
  17. package/dist/access-namespace-preflight.d.ts +3 -3
  18. package/dist/access-observe-write-surface.d.ts +11 -11
  19. package/dist/access-offline-manifest.d.ts +9 -9
  20. package/dist/access-operations.d.ts +11 -11
  21. package/dist/access-recall-concurrency.d.ts +11 -11
  22. package/dist/access-recall-response.d.ts +11 -11
  23. package/dist/access-recall-surface.d.ts +11 -11
  24. package/dist/{access-service-CgCjvbQu.d.ts → access-service-CXvbxLK-.d.ts} +7 -7
  25. package/dist/access-service-helpers.d.ts +11 -11
  26. package/dist/access-service.d.ts +11 -11
  27. package/dist/access-surface-catalog.d.ts +11 -11
  28. package/dist/access-wearables-meetings-surface.d.ts +3 -3
  29. package/dist/action-confidence.d.ts +1 -1
  30. package/dist/active-memory-bridge.d.ts +1 -1
  31. package/dist/active-recall.d.ts +1 -1
  32. package/dist/active-recall.js +2 -2
  33. package/dist/ambient-provenance.d.ts +1 -1
  34. package/dist/artifact-search.d.ts +1 -1
  35. package/dist/{auto-sync-JXEW7444.js → auto-sync-TDHLZPVP.js} +3 -3
  36. package/dist/behavior-learner.d.ts +1 -1
  37. package/dist/behavior-signals.d.ts +1 -1
  38. package/dist/bootstrap.d.ts +9 -9
  39. package/dist/briefing.d.ts +2 -2
  40. package/dist/buffer-surprise-report.d.ts +1 -1
  41. package/dist/buffer-turn-helpers.d.ts +1 -1
  42. package/dist/buffer.d.ts +2 -2
  43. package/dist/bulk-import/index.d.ts +3 -3
  44. package/dist/calibration.d.ts +1 -1
  45. package/dist/capabilities.d.ts +1 -1
  46. package/dist/{catalog-D3B4RDy1.d.ts → catalog-BOOxl-I2.d.ts} +1 -1
  47. package/dist/causal-behavior.d.ts +1 -1
  48. package/dist/causal-consolidation.d.ts +1 -1
  49. package/dist/causal-trajectory-graph.d.ts +1 -1
  50. package/dist/{chunk-QAVH6X2A.js → chunk-47P3RYZP.js} +69 -7
  51. package/dist/{chunk-QAVH6X2A.js.map → chunk-47P3RYZP.js.map} +1 -1
  52. package/dist/{chunk-VLL43QEQ.js → chunk-4IIFL7GC.js} +2 -2
  53. package/dist/{chunk-AY43LBE4.js → chunk-CYF6WVE7.js} +7 -7
  54. package/dist/{chunk-P7XI2AQ3.js → chunk-HOIRB77A.js} +98 -19
  55. package/dist/chunk-HOIRB77A.js.map +1 -0
  56. package/dist/{chunk-NDAH7BJ5.js → chunk-HVP33QVW.js} +103 -6
  57. package/dist/chunk-HVP33QVW.js.map +1 -0
  58. package/dist/{chunk-3DWTDC54.js → chunk-JV5YUBIZ.js} +4 -4
  59. package/dist/{chunk-FHFERRYE.js → chunk-UU7TRFNE.js} +2 -2
  60. package/dist/{cli-D-HiEYhl.d.ts → cli-DVUFIa5E.d.ts} +5 -5
  61. package/dist/cli.d.ts +13 -13
  62. package/dist/cli.js +5 -5
  63. package/dist/coding/pre-action-gate.d.ts +1 -1
  64. package/dist/compounding/engine.d.ts +2 -2
  65. package/dist/compounding/preference-consolidator.d.ts +1 -1
  66. package/dist/compression-optimizer.d.ts +1 -1
  67. package/dist/config.d.ts +1 -1
  68. package/dist/config.js +2 -2
  69. package/dist/connectors/codex-materialize-runner.d.ts +1 -1
  70. package/dist/connectors/codex-materialize.d.ts +1 -1
  71. package/dist/connectors/index.d.ts +1 -1
  72. package/dist/consolidation-provenance-check.d.ts +2 -2
  73. package/dist/consolidation-undo.d.ts +2 -2
  74. package/dist/contradiction/index.d.ts +2 -2
  75. package/dist/converge-config.d.ts +1 -1
  76. package/dist/convergence-refresh.d.ts +2 -2
  77. package/dist/conversation-index/backend.d.ts +1 -1
  78. package/dist/conversation-index/chunker.d.ts +1 -1
  79. package/dist/conversation-index/faiss-adapter.d.ts +1 -1
  80. package/dist/conversation-index/indexer.d.ts +1 -1
  81. package/dist/conversation-index/search.d.ts +1 -1
  82. package/dist/corpus-watermark.d.ts +1 -1
  83. package/dist/day-summary.d.ts +1 -1
  84. package/dist/delinearize.d.ts +1 -1
  85. package/dist/dependency-propagation-config.d.ts +1 -1
  86. package/dist/{dependency-propagation-delivery-D63r24pH.d.ts → dependency-propagation-delivery-XzF76C9n.d.ts} +1 -1
  87. package/dist/direct-answer-wiring.d.ts +1 -1
  88. package/dist/direct-answer.d.ts +1 -1
  89. package/dist/embedding-fallback.d.ts +1 -1
  90. package/dist/enrichment/index.d.ts +1 -1
  91. package/dist/entity-retrieval.d.ts +2 -2
  92. package/dist/entity-schema.d.ts +1 -1
  93. package/dist/explicit-capture.d.ts +9 -9
  94. package/dist/external-wiki-access.d.ts +11 -11
  95. package/dist/external-wiki-collection-registration.d.ts +1 -1
  96. package/dist/external-wiki-collection.d.ts +1 -1
  97. package/dist/external-wiki-mcp-tools.d.ts +11 -11
  98. package/dist/extraction-error-classification.d.ts +1 -1
  99. package/dist/extraction-faithfulness.d.ts +1 -1
  100. package/dist/extraction-judge-telemetry.d.ts +1 -1
  101. package/dist/extraction-judge-training.d.ts +1 -1
  102. package/dist/extraction-judge.d.ts +1 -1
  103. package/dist/extraction-liveness.d.ts +1 -1
  104. package/dist/extraction-normalization.d.ts +1 -1
  105. package/dist/extraction-prompt.d.ts +1 -1
  106. package/dist/extraction-source-grounding-rules.d.ts +1 -1
  107. package/dist/extraction-source-grounding.d.ts +1 -1
  108. package/dist/extraction.d.ts +1 -1
  109. package/dist/fallback-llm.d.ts +1 -1
  110. package/dist/graph-dashboard-diff.d.ts +1 -1
  111. package/dist/graph-dashboard-key.d.ts +1 -1
  112. package/dist/graph-dashboard-parser.d.ts +1 -1
  113. package/dist/graph-edge-reinforcement.d.ts +1 -1
  114. package/dist/graph-path-reconstruction.d.ts +1 -1
  115. package/dist/graph-path-scoring.d.ts +1 -1
  116. package/dist/graph-snapshot.d.ts +1 -1
  117. package/dist/graph.d.ts +1 -1
  118. package/dist/importance.d.ts +1 -1
  119. package/dist/importers/index.d.ts +1 -1
  120. package/dist/in-flight-reads.d.ts +1 -1
  121. package/dist/index.d.ts +103 -31
  122. package/dist/index.js +17 -7
  123. package/dist/intent.d.ts +1 -1
  124. package/dist/lcm/engine.d.ts +1 -1
  125. package/dist/lcm/index.d.ts +1 -1
  126. package/dist/lcm/tools.d.ts +1 -1
  127. package/dist/lifecycle.d.ts +1 -1
  128. package/dist/live-connectors-runner.d.ts +1 -1
  129. package/dist/local-llm.d.ts +1 -1
  130. package/dist/local-model-endpoint.d.ts +1 -1
  131. package/dist/maintenance/memory-governance.d.ts +1 -1
  132. package/dist/maintenance/rebuild-memory-lifecycle-ledger.d.ts +2 -2
  133. package/dist/maintenance/rebuild-memory-projection.d.ts +2 -2
  134. package/dist/{maintenance-CN_Ct-Hr.d.ts → maintenance-BJzNIKBE.d.ts} +3 -3
  135. package/dist/mcp-memory-inspector-app.d.ts +11 -11
  136. package/dist/memory-action-policy.d.ts +1 -1
  137. package/dist/memory-cache.d.ts +1 -1
  138. package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
  139. package/dist/memory-projection-store.d.ts +1 -1
  140. package/dist/memory-provenance.d.ts +1 -1
  141. package/dist/memory-snapshot.d.ts +1 -1
  142. package/dist/memory-worth-outcomes.d.ts +2 -2
  143. package/dist/models-json.d.ts +1 -1
  144. package/dist/namespaces/migrate.d.ts +3 -3
  145. package/dist/namespaces/principal.d.ts +1 -1
  146. package/dist/namespaces/search.d.ts +1 -1
  147. package/dist/namespaces/storage.d.ts +3 -3
  148. package/dist/native-knowledge.d.ts +1 -1
  149. package/dist/offline-sync-impression-drain.d.ts +1 -1
  150. package/dist/operator-doctor-corpus.d.ts +1 -1
  151. package/dist/operator-doctor-replica.d.ts +1 -1
  152. package/dist/operator-toolkit.d.ts +3 -3
  153. package/dist/operator-toolkit.js +3 -3
  154. package/dist/orchestration/compression-guideline-coordinator.d.ts +2 -2
  155. package/dist/orchestration/maintenance.d.ts +4 -4
  156. package/dist/{orchestrator-B3Xe8qu0.d.ts → orchestrator-BiXgBJBQ.d.ts} +8 -8
  157. package/dist/orchestrator.d.ts +9 -9
  158. package/dist/orchestrator.js +7 -7
  159. package/dist/patterns-cli.d.ts +1 -1
  160. package/dist/{pipeline-D21Rs6gW.d.ts → pipeline-sz_dcuLg.d.ts} +1 -1
  161. package/dist/policy-runtime.d.ts +1 -1
  162. package/dist/proactive-contention.d.ts +1 -1
  163. package/dist/provenance.d.ts +1 -1
  164. package/dist/{qmd-Dhle1P4X.d.ts → qmd-eAWObK4M.d.ts} +1 -1
  165. package/dist/qmd-preflight.d.ts +2 -2
  166. package/dist/qmd-recall-cache.d.ts +1 -1
  167. package/dist/qmd.d.ts +2 -2
  168. package/dist/recall-concurrency-config.d.ts +1 -1
  169. package/dist/recall-disclosure-escalation.d.ts +1 -1
  170. package/dist/recall-explain-renderer.d.ts +1 -1
  171. package/dist/recall-memory-map.d.ts +1 -1
  172. package/dist/recall-planner-llm.d.ts +1 -1
  173. package/dist/recall-state.d.ts +1 -1
  174. package/dist/recall-tag-filter.d.ts +1 -1
  175. package/dist/recall-timings.d.ts +1 -1
  176. package/dist/recall-xray-cli.d.ts +1 -1
  177. package/dist/recall-xray-renderer.d.ts +1 -1
  178. package/dist/recall-xray.d.ts +1 -1
  179. package/dist/reconcile/cursor.d.ts +1 -1
  180. package/dist/reconcile/manifest.d.ts +1 -1
  181. package/dist/reconcile/plan.d.ts +1 -1
  182. package/dist/replay/normalizers/chatgpt.d.ts +1 -1
  183. package/dist/replay/normalizers/claude.d.ts +1 -1
  184. package/dist/replay/normalizers/openclaw.d.ts +1 -1
  185. package/dist/replay/normalizers/shared.d.ts +1 -1
  186. package/dist/replay/runner.d.ts +1 -1
  187. package/dist/replay/types.d.ts +1 -1
  188. package/dist/replica-divergence.d.ts +1 -1
  189. package/dist/replica-peers-config.d.ts +1 -1
  190. package/dist/resolve-auth-token.d.ts +1 -1
  191. package/dist/resume-bundles.js +3 -3
  192. package/dist/retrieval-agents.d.ts +2 -2
  193. package/dist/retrieval-tiers.d.ts +1 -1
  194. package/dist/routing/engine.d.ts +1 -1
  195. package/dist/routing/store.d.ts +1 -1
  196. package/dist/salvage-envelope.d.ts +1 -1
  197. package/dist/schemas.d.ts +38 -38
  198. package/dist/{scope-profiles-NzGPDXLn.d.ts → scope-profiles-BBQsXuyj.d.ts} +1 -1
  199. package/dist/search/embed-helper.d.ts +1 -1
  200. package/dist/search/factory.d.ts +1 -1
  201. package/dist/search/index.d.ts +1 -1
  202. package/dist/search/lancedb-backend.d.ts +1 -1
  203. package/dist/search/meilisearch-backend.d.ts +1 -1
  204. package/dist/search/noop-backend.d.ts +1 -1
  205. package/dist/search/orama-backend.d.ts +1 -1
  206. package/dist/search/port.d.ts +1 -1
  207. package/dist/search/remote-backend.d.ts +1 -1
  208. package/dist/{semantic-consolidation-DLDKvyy5.d.ts → semantic-consolidation-HD8RkzVA.d.ts} +1 -1
  209. package/dist/semantic-consolidation.d.ts +2 -2
  210. package/dist/semantic-rule-verifier.d.ts +1 -1
  211. package/dist/{service-CQGEqL1g.d.ts → service-DIrtHHEg.d.ts} +2 -2
  212. package/dist/session-observer-bands.d.ts +1 -1
  213. package/dist/session-observer-state.d.ts +1 -1
  214. package/dist/shared-context/manager.d.ts +1 -1
  215. package/dist/signal.d.ts +1 -1
  216. package/dist/source-agent-qualifier.d.ts +1 -1
  217. package/dist/{storage-DmkCjo0l.d.ts → storage-DqCzXi56.d.ts} +1 -1
  218. package/dist/storage.d.ts +2 -2
  219. package/dist/summarizer.d.ts +1 -1
  220. package/dist/summary-snapshot.d.ts +1 -1
  221. package/dist/temporal-supersession.d.ts +2 -2
  222. package/dist/temporal-timeline-recall.d.ts +1 -1
  223. package/dist/temporal-validity.d.ts +1 -1
  224. package/dist/threading.d.ts +1 -1
  225. package/dist/tier-migration.d.ts +2 -2
  226. package/dist/tier-routing.d.ts +1 -1
  227. package/dist/topics.d.ts +1 -1
  228. package/dist/transcript.d.ts +1 -1
  229. package/dist/transfer/types.d.ts +12 -12
  230. package/dist/trust-score-stage.d.ts +1 -1
  231. package/dist/trust-score.d.ts +1 -1
  232. package/dist/{types-CkiDUorT.d.ts → types-d_j4vC6N.d.ts} +26 -1
  233. package/dist/types.d.ts +1 -1
  234. package/dist/utility-runtime.d.ts +1 -1
  235. package/dist/write-envelope.d.ts +1 -1
  236. package/package.json +2 -2
  237. package/src/wearables/cleanup.test.ts +71 -0
  238. package/src/wearables/cleanup.ts +163 -16
  239. package/src/wearables/config.test.ts +62 -0
  240. package/src/wearables/config.ts +99 -5
  241. package/src/wearables/index.ts +10 -0
  242. package/src/wearables/pipeline.test.ts +36 -0
  243. package/src/wearables/pipeline.ts +18 -3
  244. package/src/wearables/redaction.test.ts +156 -0
  245. package/src/wearables/redaction.ts +102 -10
  246. package/src/wearables/text-language.ts +146 -0
  247. package/src/wearables/types.ts +26 -0
  248. package/dist/chunk-NDAH7BJ5.js.map +0 -1
  249. package/dist/chunk-P7XI2AQ3.js.map +0 -1
  250. /package/dist/{auto-sync-JXEW7444.js.map → auto-sync-TDHLZPVP.js.map} +0 -0
  251. /package/dist/{chunk-VLL43QEQ.js.map → chunk-4IIFL7GC.js.map} +0 -0
  252. /package/dist/{chunk-AY43LBE4.js.map → chunk-CYF6WVE7.js.map} +0 -0
  253. /package/dist/{chunk-3DWTDC54.js.map → chunk-JV5YUBIZ.js.map} +0 -0
  254. /package/dist/{chunk-FHFERRYE.js.map → chunk-UU7TRFNE.js.map} +0 -0
@@ -7,6 +7,7 @@
7
7
  * everything here is conservative and reversible by re-syncing.
8
8
  */
9
9
 
10
+ import { detectScriptHints, type ScriptHint } from "./text-language.js";
10
11
  import type {
11
12
  WearableCleanupSettings,
12
13
  WearableConversation,
@@ -33,24 +34,155 @@ const MERGE_GAP_MS = 30_000;
33
34
  const GENERIC_SPEAKER_PATTERN = /^unknown$/i;
34
35
 
35
36
  /**
36
- * Standalone filler tokens stripped when `stripFillers` is on. Matched
37
- * case-insensitively on word boundaries, only as whole tokens — "um"
38
- * inside "umbrella" is never touched. Deliberately short, low-risk
39
- * list; meaning-bearing hedges ("like", "well") are NOT stripped.
37
+ * Standalone filler tokens stripped when `stripFillers` is on, for
38
+ * scripts that separate words with spaces. Matched case-insensitively
39
+ * as whole tokens — "um" inside "umbrella" is never touched.
40
+ * Deliberately short and low-risk: meaning-bearing hedges ("like",
41
+ * "well", Spanish "este", Arabic "يعني") are NOT stripped, and neither
42
+ * is "mm", which is also a unit of length.
40
43
  */
41
- const FILLER_TOKENS = ["um", "uh", "uhm", "umm", "uhh", "erm", "hmm", "mhm"];
44
+ const TOKEN_FILLERS: Readonly<Partial<Record<ScriptHint, readonly string[]>>> = {
45
+ latin: [
46
+ "um",
47
+ "uh",
48
+ "uhm",
49
+ "umm",
50
+ "uhh",
51
+ "erm",
52
+ "hmm",
53
+ "mhm",
54
+ "ähm",
55
+ "äh",
56
+ "ehm",
57
+ "euh",
58
+ "mmm",
59
+ ],
60
+ korean: ["음", "으음", "엄", "어어"],
61
+ arabic: ["امم", "اممم", "اه"],
62
+ cyrillic: ["эм", "ээ", "ммм"],
63
+ };
42
64
 
43
- const FILLER_PATTERN = new RegExp(
44
- // Leading/trailing punctuation around the filler collapses with it so
45
- // "Um, so we should" -> "so we should" rather than ", so we should".
46
- `(?:^|\\s)(?:${FILLER_TOKENS.join("|")})[,.]?(?=\\s|$)`,
47
- "gi",
48
- );
65
+ /**
66
+ * Filler tokens for scripts without word spaces. These are removed
67
+ * inline, so each entry must be unambiguous on its own: the elongated
68
+ * Japanese hesitation forms and the two Chinese hesitation particles
69
+ * qualify, while a bare 「あの」 ("that") does not (issue #2196).
70
+ */
71
+ const INLINE_FILLERS: Readonly<Partial<Record<ScriptHint, readonly string[]>>> = {
72
+ japanese: ["えーと", "えっと", "ええと", "あのー", "あのう", "うーん"],
73
+ han: ["呃", "嗯"],
74
+ };
75
+
76
+ /**
77
+ * Punctuation that trails a filler and collapses with it, so
78
+ * 「えーと、では」 becomes 「では」 rather than 「、では」. Arabic sentence
79
+ * punctuation is included: an Arabic filler is normally followed by
80
+ * `،`, and without it the token survives the pass.
81
+ */
82
+ const TRAILING_FILLER_PUNCTUATION = "[,.、。،؛؟!?]?";
83
+
84
+ interface FillerMatchers {
85
+ token: RegExp | null;
86
+ inline: RegExp | null;
87
+ }
88
+
89
+ /**
90
+ * Compiled matchers are cached per (script hints + extra tokens) key.
91
+ * A day of transcripts is thousands of segments over a handful of
92
+ * scripts, so this compiles a few regexes instead of one per segment.
93
+ */
94
+ const FILLER_MATCHER_CACHE = new Map<string, FillerMatchers>();
95
+
96
+ function escapeRegExp(value: string): string {
97
+ return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
98
+ }
99
+
100
+ /**
101
+ * Build one alternation from `tokens`, or `null` when none remain.
102
+ *
103
+ * Each token contributes BOTH its composed and its decomposed spelling,
104
+ * so `ähm` also matches an ASR that emits `a\u0308hm`. Matching both
105
+ * forms keeps the surrounding transcript byte-identical, which
106
+ * normalizing the text before stripping would not.
107
+ *
108
+ * JavaScript alternation is first-match, not longest-match, so the
109
+ * tokens are sorted longest-first. Without that a built-in prefix such
110
+ * as 「あのー」 would match ahead of a longer operator token that starts
111
+ * with it and leave the tail in the transcript. Ties break on the token
112
+ * itself so the pattern is stable across runs.
113
+ */
114
+ function buildAlternation(
115
+ tokens: readonly string[],
116
+ shape: (pattern: string) => { source: string; flags: string },
117
+ ): RegExp | null {
118
+ const forms = new Set<string>();
119
+ for (const token of tokens) {
120
+ forms.add(token.normalize("NFC"));
121
+ forms.add(token.normalize("NFD"));
122
+ }
123
+ const unique = [...forms].sort(
124
+ (left, right) => right.length - left.length || (left < right ? -1 : left > right ? 1 : 0),
125
+ );
126
+ if (unique.length === 0) return null;
127
+ const { source, flags } = shape(unique.map((token) => escapeRegExp(token)).join("|"));
128
+ return new RegExp(source, flags);
129
+ }
130
+
131
+ function buildFillerMatchers(
132
+ hints: readonly ScriptHint[],
133
+ extraTokens: readonly string[],
134
+ ): FillerMatchers {
135
+ // Structural key: comma-joining collapses `["a,b"]` and `["a","b"]`
136
+ // onto one entry, and the first caller's regex would then clean the
137
+ // second caller's transcripts.
138
+ const key = JSON.stringify([hints, extraTokens]);
139
+ const cached = FILLER_MATCHER_CACHE.get(key);
140
+ if (cached) return cached;
141
+
142
+ const tokens: string[] = [];
143
+ const inline: string[] = [];
144
+ for (const hint of hints) {
145
+ tokens.push(...(TOKEN_FILLERS[hint] ?? []));
146
+ inline.push(...(INLINE_FILLERS[hint] ?? []));
147
+ }
148
+ // Operator tokens apply to every script: the operator knows their own
149
+ // language, and routing them by script hint would silently drop them.
150
+ for (const extra of extraTokens) {
151
+ const trimmed = extra.trim();
152
+ if (trimmed.length === 0) continue;
153
+ if (SPACE_DELIMITED_TOKEN.test(trimmed)) tokens.push(trimmed);
154
+ else inline.push(trimmed);
155
+ }
156
+
157
+ const matchers: FillerMatchers = {
158
+ token: buildAlternation(tokens, (pattern) => ({
159
+ source: `(?:^|\\s)(?:${pattern})${TRAILING_FILLER_PUNCTUATION}(?=\\s|$)`,
160
+ flags: "giu",
161
+ })),
162
+ inline: buildAlternation(inline, (pattern) => ({
163
+ source: `(?:${pattern})${TRAILING_FILLER_PUNCTUATION}`,
164
+ flags: "gu",
165
+ })),
166
+ };
167
+ FILLER_MATCHER_CACHE.set(key, matchers);
168
+ return matchers;
169
+ }
170
+
171
+ /**
172
+ * A token whose first character belongs to a space-delimited script.
173
+ *
174
+ * Anything outside this set is removed INLINE, so a script that does
175
+ * space its words must appear here. Hebrew `אה` routed to the inline
176
+ * matcher would strip the suffix of `נראה` and corrupt the transcript.
177
+ */
178
+ const SPACE_DELIMITED_TOKEN =
179
+ /^[\p{Script=Latin}\p{Script=Cyrillic}\p{Script=Greek}\p{Script=Hangul}\p{Script=Arabic}\p{Script=Hebrew}\p{N}]/u;
49
180
 
50
181
  /** Apply configured cleanup passes to one conversation. */
51
182
  export function cleanConversation(
52
183
  conversation: WearableConversation,
53
184
  settings: WearableCleanupSettings,
185
+ extraFillerTokens: readonly string[] = [],
54
186
  ): CleanupResult {
55
187
  let segments = conversation.segments.map((segment) => ({ ...segment }));
56
188
  let droppedSegments = 0;
@@ -61,7 +193,7 @@ export function cleanConversation(
61
193
 
62
194
  if (settings.stripFillers) {
63
195
  for (const segment of segments) {
64
- segment.text = stripFillerTokens(segment.text);
196
+ segment.text = stripFillerTokens(segment.text, extraFillerTokens);
65
197
  }
66
198
  }
67
199
 
@@ -130,15 +262,30 @@ function canMerge(
130
262
  const label = previous.speakerKey.trim();
131
263
  if (label.length === 0 || GENERIC_SPEAKER_PATTERN.test(label)) return false;
132
264
  }
133
- const previousEnd = previous.endIso ? Date.parse(previous.endIso) : NaN;
134
- const nextStart = next.startIso ? Date.parse(next.startIso) : NaN;
265
+ const previousEnd = previous.endIso
266
+ ? Date.parse(previous.endIso)
267
+ : Number.NaN;
268
+ const nextStart = next.startIso ? Date.parse(next.startIso) : Number.NaN;
135
269
  // Without timestamps, adjacency is the only signal — still merge.
136
270
  if (Number.isNaN(previousEnd) || Number.isNaN(nextStart)) return true;
137
271
  return nextStart - previousEnd <= MERGE_GAP_MS;
138
272
  }
139
273
 
140
- export function stripFillerTokens(text: string): string {
141
- return normalizeWhitespace(text.replace(FILLER_PATTERN, " "));
274
+ /**
275
+ * Remove filler tokens from `text`. The built-in token set is chosen
276
+ * from the scripts present in the text, so a Japanese or Korean segment
277
+ * is cleaned by its own list instead of silently keeping every filler
278
+ * (issue #2196). Operator tokens apply to every script.
279
+ */
280
+ export function stripFillerTokens(
281
+ text: string,
282
+ extraTokens: readonly string[] = [],
283
+ ): string {
284
+ const matchers = buildFillerMatchers(detectScriptHints(text), extraTokens);
285
+ let result = text;
286
+ if (matchers.token) result = result.replace(matchers.token, " ");
287
+ if (matchers.inline) result = result.replace(matchers.inline, "");
288
+ return normalizeWhitespace(result);
142
289
  }
143
290
 
144
291
  /**
@@ -277,3 +277,65 @@ test("fusion knobs reject non-positive integers", () => {
277
277
  test("fusion block must be an object when present", () => {
278
278
  assert.throws(() => parseWearablesConfig({ fusion: "on" }), /must be an object/);
279
279
  });
280
+
281
+ test("off-the-record marker phrases and filler tokens parse", () => {
282
+ const parsed = parseWearablesConfig({
283
+ offTheRecordMarkers: {
284
+ start: [" poza protokołem "],
285
+ end: ["z powrotem do protokołu"],
286
+ },
287
+ fillerTokens: ["bueno", "那个"],
288
+ });
289
+ assert.deepEqual(parsed.offTheRecordMarkers.start, ["poza protokołem"]);
290
+ assert.deepEqual(parsed.offTheRecordMarkers.end, ["z powrotem do protokołu"]);
291
+ assert.equal(parsed.offTheRecordMarkers.useBuiltIns, true);
292
+ assert.deepEqual(parsed.fillerTokens, ["bueno", "那个"]);
293
+ });
294
+
295
+ test("marker and filler lists reject unusable entries loudly", () => {
296
+ assert.throws(
297
+ () => parseWearablesConfig({ offTheRecordMarkers: { start: "poza" } }),
298
+ /offTheRecordMarkers\.start must be an array of strings/,
299
+ );
300
+ assert.throws(
301
+ () => parseWearablesConfig({ offTheRecordMarkers: { end: [" "] } }),
302
+ /offTheRecordMarkers\.end\[0\] must be a non-empty string/,
303
+ );
304
+ assert.throws(
305
+ () => parseWearablesConfig({ fillerTokens: ["a".repeat(65)] }),
306
+ /fillerTokens\[0\] exceeds 64 characters/,
307
+ );
308
+ assert.throws(
309
+ () => parseWearablesConfig({ offTheRecordMarkers: { start: ["a".repeat(129)] } }),
310
+ /offTheRecordMarkers\.start\[0\] exceeds 128 characters/,
311
+ );
312
+ assert.throws(
313
+ () => parseWearablesConfig({ offTheRecordMarkers: [] }),
314
+ /offTheRecordMarkers must be an object/,
315
+ );
316
+ });
317
+
318
+ test("disabling built-in markers without a start phrase is rejected, not silent", () => {
319
+ assert.throws(
320
+ () =>
321
+ parseWearablesConfig({
322
+ offTheRecordEnabled: true,
323
+ offTheRecordMarkers: { useBuiltIns: false },
324
+ }),
325
+ /useBuiltIns is false but no start phrase is configured/,
326
+ );
327
+ // Turning the whole feature off is the documented escape hatch.
328
+ const parsed = parseWearablesConfig({
329
+ offTheRecordEnabled: false,
330
+ offTheRecordMarkers: { useBuiltIns: false },
331
+ });
332
+ assert.equal(parsed.offTheRecordEnabled, false);
333
+ assert.equal(parsed.offTheRecordMarkers.useBuiltIns, false);
334
+ });
335
+
336
+ test("an unknown off-the-record marker key is rejected, not dropped", () => {
337
+ assert.throws(
338
+ () => parseWearablesConfig({ offTheRecordMarkers: { starts: ["private"] } }),
339
+ /offTheRecordMarkers has unknown key\(s\) starts/,
340
+ );
341
+ });
@@ -13,6 +13,7 @@ import type { ImportanceLevel } from "../types.js";
13
13
  import { compileCorrectionRules } from "./corrections.js";
14
14
  import { compileRedactionPatterns } from "./redaction.js";
15
15
  import type {
16
+ OffTheRecordMarkerSettings,
16
17
  WearableCleanupSettings,
17
18
  WearableCorrectionRule,
18
19
  WearableMemoryMode,
@@ -111,6 +112,8 @@ export function defaultWearablesConfig(): WearablesConfig {
111
112
  redactionEnabled: true,
112
113
  redactionPatterns: [],
113
114
  offTheRecordEnabled: true,
115
+ offTheRecordMarkers: { start: [], end: [], useBuiltIns: true },
116
+ fillerTokens: [],
114
117
  digestEnabled: true,
115
118
  autoSyncEnabled: DEFAULT_AUTO_SYNC_ENABLED,
116
119
  autoSyncIntervalMinutes: DEFAULT_AUTO_SYNC_INTERVAL_MINUTES,
@@ -134,6 +137,37 @@ function requireObject(
134
137
  return value as Record<string, unknown>;
135
138
  }
136
139
 
140
+ /** Longest accepted off-the-record marker phrase. */
141
+ const MAX_MARKER_PHRASE_LENGTH = 128;
142
+ /** Longest accepted filler token. */
143
+ const MAX_FILLER_TOKEN_LENGTH = 64;
144
+
145
+ /**
146
+ * Parse an optional array of non-empty phrases. Rejects a non-array, a
147
+ * non-string entry, a blank entry, and an over-long entry — an operator
148
+ * privacy phrase that is silently discarded is the failure this parser
149
+ * exists to prevent (issue #2196).
150
+ */
151
+ function parsePhraseList(
152
+ value: unknown,
153
+ keyPath: string,
154
+ maxLength: number,
155
+ ): string[] {
156
+ if (value === undefined) return [];
157
+ if (!Array.isArray(value)) {
158
+ throw new Error(`${keyPath} must be an array of strings`);
159
+ }
160
+ return value.map((entry, index) => {
161
+ if (typeof entry !== "string" || entry.trim().length === 0) {
162
+ throw new Error(`${keyPath}[${index}] must be a non-empty string`);
163
+ }
164
+ if (entry.length > maxLength) {
165
+ throw new Error(`${keyPath}[${index}] exceeds ${maxLength} characters`);
166
+ }
167
+ return entry.trim();
168
+ });
169
+ }
170
+
137
171
  function parseBool(
138
172
  value: unknown,
139
173
  keyPath: string,
@@ -411,6 +445,68 @@ export function parseWearablesConfig(value: unknown): WearablesConfig {
411
445
  compileRedactionPatterns(redactionPatterns);
412
446
  }
413
447
 
448
+ const offTheRecordEnabled = parseBool(
449
+ raw.offTheRecordEnabled,
450
+ "wearables.offTheRecordEnabled",
451
+ defaults.offTheRecordEnabled,
452
+ );
453
+ const rawMarkers =
454
+ raw.offTheRecordMarkers === undefined
455
+ ? {}
456
+ : requireObject(raw.offTheRecordMarkers, "wearables.offTheRecordMarkers");
457
+ // A typo such as `starts:` would otherwise be dropped without a word,
458
+ // and the operator would believe a private span is protected. Core and
459
+ // standalone callers never see OpenClaw's JSON schema, so the check
460
+ // belongs here. Keys are compared literally so the config-contract
461
+ // extractor can still enumerate this parser's surface.
462
+ const {
463
+ start: rawMarkerStart,
464
+ end: rawMarkerEnd,
465
+ useBuiltIns: rawMarkerUseBuiltIns,
466
+ ...unknownMarkerKeys
467
+ } = rawMarkers;
468
+ const unknownMarkerNames = Object.keys(unknownMarkerKeys);
469
+ if (unknownMarkerNames.length > 0) {
470
+ throw new Error(
471
+ `wearables.offTheRecordMarkers has unknown key(s) ${unknownMarkerNames.sort().join(", ")} — valid keys are start, end, useBuiltIns`,
472
+ );
473
+ }
474
+ const offTheRecordMarkers: OffTheRecordMarkerSettings = {
475
+ start: parsePhraseList(
476
+ rawMarkerStart,
477
+ "wearables.offTheRecordMarkers.start",
478
+ MAX_MARKER_PHRASE_LENGTH,
479
+ ),
480
+ end: parsePhraseList(
481
+ rawMarkerEnd,
482
+ "wearables.offTheRecordMarkers.end",
483
+ MAX_MARKER_PHRASE_LENGTH,
484
+ ),
485
+ useBuiltIns: parseBool(
486
+ rawMarkerUseBuiltIns,
487
+ "wearables.offTheRecordMarkers.useBuiltIns",
488
+ defaults.offTheRecordMarkers.useBuiltIns,
489
+ ),
490
+ };
491
+ // A silent no-op is the exact failure this feature exists to remove
492
+ // (issue #2196): turning the built-ins off without supplying a start
493
+ // phrase would leave the gate on and matching nothing.
494
+ if (
495
+ offTheRecordEnabled &&
496
+ !offTheRecordMarkers.useBuiltIns &&
497
+ offTheRecordMarkers.start.length === 0
498
+ ) {
499
+ throw new Error(
500
+ "wearables.offTheRecordMarkers.useBuiltIns is false but no start phrase is configured — add wearables.offTheRecordMarkers.start, or set wearables.offTheRecordEnabled to false",
501
+ );
502
+ }
503
+
504
+ const fillerTokens = parsePhraseList(
505
+ raw.fillerTokens,
506
+ "wearables.fillerTokens",
507
+ MAX_FILLER_TOKEN_LENGTH,
508
+ );
509
+
414
510
  const parseBoundedInt = (
415
511
  value: unknown,
416
512
  name: string,
@@ -488,11 +584,9 @@ export function parseWearablesConfig(value: unknown): WearablesConfig {
488
584
  defaults.redactionEnabled,
489
585
  ),
490
586
  redactionPatterns,
491
- offTheRecordEnabled: parseBool(
492
- raw.offTheRecordEnabled,
493
- "wearables.offTheRecordEnabled",
494
- defaults.offTheRecordEnabled,
495
- ),
587
+ offTheRecordEnabled,
588
+ offTheRecordMarkers,
589
+ fillerTokens,
496
590
  digestEnabled: parseBool(
497
591
  raw.digestEnabled,
498
592
  "wearables.digestEnabled",
@@ -35,10 +35,20 @@ export {
35
35
  } from "./cleanup.js";
36
36
  export {
37
37
  applyOffTheRecord,
38
+ BUILT_IN_OFF_THE_RECORD_END,
39
+ BUILT_IN_OFF_THE_RECORD_START,
40
+ compileOffTheRecordMarkers,
38
41
  compileRedactionPatterns,
39
42
  redactText,
40
43
  REDACTION_PLACEHOLDER,
44
+ type CompiledOffTheRecordMarkers,
45
+ type OffTheRecordMarkerInput,
41
46
  } from "./redaction.js";
47
+ export {
48
+ buildPhraseMatcher,
49
+ detectScriptHints,
50
+ type ScriptHint,
51
+ } from "./text-language.js";
42
52
  export {
43
53
  applyCorrections,
44
54
  compileCorrectionRule,
@@ -1039,3 +1039,39 @@ test("a transcript write failure prevents the sync watermark from advancing", as
1039
1039
  rmSync(memoryDir, { recursive: true, force: true });
1040
1040
  }
1041
1041
  });
1042
+
1043
+ test("configured non-English off-the-record markers elide a span end to end", async () => {
1044
+ const memoryDir = mkdtempSync(path.join(tmpdir(), "remnic-pipeline-otr-"));
1045
+ try {
1046
+ const byDate = {
1047
+ "2026-06-11": [
1048
+ makeConversation("c1", "2026-06-11", [
1049
+ { speaker: "user", isWearer: true, text: "えーと、ここからはオフレコでお願いします。" },
1050
+ { speaker: "Speaker 2", text: "買収の金額は非公開です。" },
1051
+ { speaker: "user", isWearer: true, text: "オンレコに戻ります。" },
1052
+ { speaker: "Speaker 2", text: "bueno, 天気の話をしましょう。" },
1053
+ ]),
1054
+ ],
1055
+ };
1056
+ const { deps, written } = makeDeps(memoryDir);
1057
+ const summary = await syncWearableSource(
1058
+ fakeConnector(byDate),
1059
+ settings({ memoryMode: "off" }),
1060
+ config({ offTheRecordEnabled: true, fillerTokens: ["bueno"] }),
1061
+ { days: 1 },
1062
+ deps,
1063
+ );
1064
+
1065
+ assert.equal(summary.transcriptsWritten.length, 1);
1066
+ const body = written[0]?.serialized ?? "";
1067
+ assert.ok(!body.includes("買収の金額は非公開です"), "off-record segment must not be stored");
1068
+ assert.ok(body.includes("[off the record"), "elision placeholder is visible");
1069
+ assert.ok(body.includes("[back on the record]"), "span closes on the Japanese end marker");
1070
+ assert.ok(body.includes("天気の話をしましょう"), "on-record speech survives");
1071
+ assert.ok(!body.includes("えーと"), "Japanese filler is stripped");
1072
+ assert.ok(!body.includes("bueno"), "configured filler token is stripped");
1073
+ assert.equal(summary.segmentsDropped, 1);
1074
+ } finally {
1075
+ rmSync(memoryDir, { recursive: true, force: true });
1076
+ }
1077
+ });
@@ -37,7 +37,13 @@ import {
37
37
  type WearableMemoryGenDeps,
38
38
  } from "./memory-gen.js";
39
39
  import { tokenizeDayBody, type CorroborationContext } from "./trust.js";
40
- import { applyOffTheRecord, compileRedactionPatterns, redactText } from "./redaction.js";
40
+ import {
41
+ applyOffTheRecord,
42
+ compileOffTheRecordMarkers,
43
+ compileRedactionPatterns,
44
+ redactText,
45
+ type CompiledOffTheRecordMarkers,
46
+ } from "./redaction.js";
41
47
  import { loadSpeakerRegistry } from "./speakers.js";
42
48
  import {
43
49
  loadSyncState,
@@ -259,6 +265,7 @@ function cleanDay(
259
265
  config: WearablesConfig,
260
266
  userRedaction: RegExp[],
261
267
  correctionRules: CompiledCorrectionRule[],
268
+ offTheRecordMarkers: CompiledOffTheRecordMarkers,
262
269
  ): CleanedDay {
263
270
  const { offTheRecordEnabled, redactionEnabled } = config;
264
271
  const out: CleanedDay = {
@@ -271,11 +278,15 @@ function cleanDay(
271
278
  for (const conversation of raw) {
272
279
  let current = conversation;
273
280
  if (offTheRecordEnabled) {
274
- const otr = applyOffTheRecord(current);
281
+ const otr = applyOffTheRecord(current, offTheRecordMarkers);
275
282
  current = otr.conversation;
276
283
  out.segmentsDropped += otr.droppedSegments;
277
284
  }
278
- const cleaned = cleanConversation(current, settings.cleanup);
285
+ const cleaned = cleanConversation(
286
+ current,
287
+ settings.cleanup,
288
+ config.fillerTokens,
289
+ );
279
290
  current = cleaned.conversation;
280
291
  out.segmentsDropped += cleaned.droppedSegments;
281
292
 
@@ -337,6 +348,9 @@ export async function syncWearableSource(
337
348
  ...compileCorrectionRules(stateRules, "state corrections"),
338
349
  ];
339
350
  const userRedaction = compileRedactionPatterns(config.redactionPatterns);
351
+ const offTheRecordMarkers = compileOffTheRecordMarkers(
352
+ config.offTheRecordMarkers,
353
+ );
340
354
 
341
355
  let syncState = await loadSyncState(deps.memoryDir);
342
356
  const previousState = syncState.sources[connector.id];
@@ -360,6 +374,7 @@ export async function syncWearableSource(
360
374
  config,
361
375
  userRedaction,
362
376
  correctionRules,
377
+ offTheRecordMarkers,
363
378
  );
364
379
  summary.conversations += cleaned.conversations.length;
365
380
  summary.segmentsKept += cleaned.segmentsKept;