@remnic/core 9.53.0 → 9.54.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/access-admin-ops-surface.d.ts +11 -11
- package/dist/access-authorization-probe.d.ts +11 -11
- package/dist/access-boundary.d.ts +11 -11
- package/dist/access-cli.js +7 -7
- package/dist/access-coding-context-resolution.d.ts +1 -1
- package/dist/access-extraction-force-flush.d.ts +11 -11
- package/dist/access-health-types.d.ts +1 -1
- package/dist/access-http-lcm-compaction.d.ts +11 -11
- package/dist/access-http-lifecycle-flush.d.ts +11 -11
- package/dist/access-http-offline-stream.d.ts +11 -11
- package/dist/access-http.d.ts +11 -11
- package/dist/access-identity-continuity-surface.d.ts +9 -9
- package/dist/access-lcm-surface.d.ts +11 -11
- package/dist/access-mcp.d.ts +11 -11
- package/dist/access-memory-search-fanout.d.ts +2 -2
- package/dist/{access-namespace-preflight-Ds6CGlCP.d.ts → access-namespace-preflight-C7n4tgWC.d.ts} +1 -1
- package/dist/access-namespace-preflight.d.ts +3 -3
- package/dist/access-observe-write-surface.d.ts +11 -11
- package/dist/access-offline-manifest.d.ts +9 -9
- package/dist/access-operations.d.ts +11 -11
- package/dist/access-recall-concurrency.d.ts +11 -11
- package/dist/access-recall-response.d.ts +11 -11
- package/dist/access-recall-surface.d.ts +11 -11
- package/dist/{access-service-CgCjvbQu.d.ts → access-service-CXvbxLK-.d.ts} +7 -7
- package/dist/access-service-helpers.d.ts +11 -11
- package/dist/access-service.d.ts +11 -11
- package/dist/access-surface-catalog.d.ts +11 -11
- package/dist/access-wearables-meetings-surface.d.ts +3 -3
- package/dist/action-confidence.d.ts +1 -1
- package/dist/active-memory-bridge.d.ts +1 -1
- package/dist/active-recall.d.ts +1 -1
- package/dist/active-recall.js +2 -2
- package/dist/ambient-provenance.d.ts +1 -1
- package/dist/artifact-search.d.ts +1 -1
- package/dist/{auto-sync-JXEW7444.js → auto-sync-TDHLZPVP.js} +3 -3
- package/dist/behavior-learner.d.ts +1 -1
- package/dist/behavior-signals.d.ts +1 -1
- package/dist/bootstrap.d.ts +9 -9
- package/dist/briefing.d.ts +2 -2
- package/dist/buffer-surprise-report.d.ts +1 -1
- package/dist/buffer-turn-helpers.d.ts +1 -1
- package/dist/buffer.d.ts +2 -2
- package/dist/bulk-import/index.d.ts +3 -3
- package/dist/calibration.d.ts +1 -1
- package/dist/capabilities.d.ts +1 -1
- package/dist/{catalog-D3B4RDy1.d.ts → catalog-BOOxl-I2.d.ts} +1 -1
- package/dist/causal-behavior.d.ts +1 -1
- package/dist/causal-consolidation.d.ts +1 -1
- package/dist/causal-trajectory-graph.d.ts +1 -1
- package/dist/{chunk-QAVH6X2A.js → chunk-47P3RYZP.js} +69 -7
- package/dist/{chunk-QAVH6X2A.js.map → chunk-47P3RYZP.js.map} +1 -1
- package/dist/{chunk-VLL43QEQ.js → chunk-4IIFL7GC.js} +2 -2
- package/dist/{chunk-AY43LBE4.js → chunk-CYF6WVE7.js} +7 -7
- package/dist/{chunk-P7XI2AQ3.js → chunk-HOIRB77A.js} +98 -19
- package/dist/chunk-HOIRB77A.js.map +1 -0
- package/dist/{chunk-NDAH7BJ5.js → chunk-HVP33QVW.js} +103 -6
- package/dist/chunk-HVP33QVW.js.map +1 -0
- package/dist/{chunk-3DWTDC54.js → chunk-JV5YUBIZ.js} +4 -4
- package/dist/{chunk-FHFERRYE.js → chunk-UU7TRFNE.js} +2 -2
- package/dist/{cli-D-HiEYhl.d.ts → cli-DVUFIa5E.d.ts} +5 -5
- package/dist/cli.d.ts +13 -13
- package/dist/cli.js +5 -5
- package/dist/coding/pre-action-gate.d.ts +1 -1
- package/dist/compounding/engine.d.ts +2 -2
- package/dist/compounding/preference-consolidator.d.ts +1 -1
- package/dist/compression-optimizer.d.ts +1 -1
- package/dist/config.d.ts +1 -1
- package/dist/config.js +2 -2
- package/dist/connectors/codex-materialize-runner.d.ts +1 -1
- package/dist/connectors/codex-materialize.d.ts +1 -1
- package/dist/connectors/index.d.ts +1 -1
- package/dist/consolidation-provenance-check.d.ts +2 -2
- package/dist/consolidation-undo.d.ts +2 -2
- package/dist/contradiction/index.d.ts +2 -2
- package/dist/converge-config.d.ts +1 -1
- package/dist/convergence-refresh.d.ts +2 -2
- package/dist/conversation-index/backend.d.ts +1 -1
- package/dist/conversation-index/chunker.d.ts +1 -1
- package/dist/conversation-index/faiss-adapter.d.ts +1 -1
- package/dist/conversation-index/indexer.d.ts +1 -1
- package/dist/conversation-index/search.d.ts +1 -1
- package/dist/corpus-watermark.d.ts +1 -1
- package/dist/day-summary.d.ts +1 -1
- package/dist/delinearize.d.ts +1 -1
- package/dist/dependency-propagation-config.d.ts +1 -1
- package/dist/{dependency-propagation-delivery-D63r24pH.d.ts → dependency-propagation-delivery-XzF76C9n.d.ts} +1 -1
- package/dist/direct-answer-wiring.d.ts +1 -1
- package/dist/direct-answer.d.ts +1 -1
- package/dist/embedding-fallback.d.ts +1 -1
- package/dist/enrichment/index.d.ts +1 -1
- package/dist/entity-retrieval.d.ts +2 -2
- package/dist/entity-schema.d.ts +1 -1
- package/dist/explicit-capture.d.ts +9 -9
- package/dist/external-wiki-access.d.ts +11 -11
- package/dist/external-wiki-collection-registration.d.ts +1 -1
- package/dist/external-wiki-collection.d.ts +1 -1
- package/dist/external-wiki-mcp-tools.d.ts +11 -11
- package/dist/extraction-error-classification.d.ts +1 -1
- package/dist/extraction-faithfulness.d.ts +1 -1
- package/dist/extraction-judge-telemetry.d.ts +1 -1
- package/dist/extraction-judge-training.d.ts +1 -1
- package/dist/extraction-judge.d.ts +1 -1
- package/dist/extraction-liveness.d.ts +1 -1
- package/dist/extraction-normalization.d.ts +1 -1
- package/dist/extraction-prompt.d.ts +1 -1
- package/dist/extraction-source-grounding-rules.d.ts +1 -1
- package/dist/extraction-source-grounding.d.ts +1 -1
- package/dist/extraction.d.ts +1 -1
- package/dist/fallback-llm.d.ts +1 -1
- package/dist/graph-dashboard-diff.d.ts +1 -1
- package/dist/graph-dashboard-key.d.ts +1 -1
- package/dist/graph-dashboard-parser.d.ts +1 -1
- package/dist/graph-edge-reinforcement.d.ts +1 -1
- package/dist/graph-path-reconstruction.d.ts +1 -1
- package/dist/graph-path-scoring.d.ts +1 -1
- package/dist/graph-snapshot.d.ts +1 -1
- package/dist/graph.d.ts +1 -1
- package/dist/importance.d.ts +1 -1
- package/dist/importers/index.d.ts +1 -1
- package/dist/in-flight-reads.d.ts +1 -1
- package/dist/index.d.ts +103 -31
- package/dist/index.js +17 -7
- package/dist/intent.d.ts +1 -1
- package/dist/lcm/engine.d.ts +1 -1
- package/dist/lcm/index.d.ts +1 -1
- package/dist/lcm/tools.d.ts +1 -1
- package/dist/lifecycle.d.ts +1 -1
- package/dist/live-connectors-runner.d.ts +1 -1
- package/dist/local-llm.d.ts +1 -1
- package/dist/local-model-endpoint.d.ts +1 -1
- package/dist/maintenance/memory-governance.d.ts +1 -1
- package/dist/maintenance/rebuild-memory-lifecycle-ledger.d.ts +2 -2
- package/dist/maintenance/rebuild-memory-projection.d.ts +2 -2
- package/dist/{maintenance-CN_Ct-Hr.d.ts → maintenance-BJzNIKBE.d.ts} +3 -3
- package/dist/mcp-memory-inspector-app.d.ts +11 -11
- package/dist/memory-action-policy.d.ts +1 -1
- package/dist/memory-cache.d.ts +1 -1
- package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
- package/dist/memory-projection-store.d.ts +1 -1
- package/dist/memory-provenance.d.ts +1 -1
- package/dist/memory-snapshot.d.ts +1 -1
- package/dist/memory-worth-outcomes.d.ts +2 -2
- package/dist/models-json.d.ts +1 -1
- package/dist/namespaces/migrate.d.ts +3 -3
- package/dist/namespaces/principal.d.ts +1 -1
- package/dist/namespaces/search.d.ts +1 -1
- package/dist/namespaces/storage.d.ts +3 -3
- package/dist/native-knowledge.d.ts +1 -1
- package/dist/offline-sync-impression-drain.d.ts +1 -1
- package/dist/operator-doctor-corpus.d.ts +1 -1
- package/dist/operator-doctor-replica.d.ts +1 -1
- package/dist/operator-toolkit.d.ts +3 -3
- package/dist/operator-toolkit.js +3 -3
- package/dist/orchestration/compression-guideline-coordinator.d.ts +2 -2
- package/dist/orchestration/maintenance.d.ts +4 -4
- package/dist/{orchestrator-B3Xe8qu0.d.ts → orchestrator-BiXgBJBQ.d.ts} +8 -8
- package/dist/orchestrator.d.ts +9 -9
- package/dist/orchestrator.js +7 -7
- package/dist/patterns-cli.d.ts +1 -1
- package/dist/{pipeline-D21Rs6gW.d.ts → pipeline-sz_dcuLg.d.ts} +1 -1
- package/dist/policy-runtime.d.ts +1 -1
- package/dist/proactive-contention.d.ts +1 -1
- package/dist/provenance.d.ts +1 -1
- package/dist/{qmd-Dhle1P4X.d.ts → qmd-eAWObK4M.d.ts} +1 -1
- package/dist/qmd-preflight.d.ts +2 -2
- package/dist/qmd-recall-cache.d.ts +1 -1
- package/dist/qmd.d.ts +2 -2
- package/dist/recall-concurrency-config.d.ts +1 -1
- package/dist/recall-disclosure-escalation.d.ts +1 -1
- package/dist/recall-explain-renderer.d.ts +1 -1
- package/dist/recall-memory-map.d.ts +1 -1
- package/dist/recall-planner-llm.d.ts +1 -1
- package/dist/recall-state.d.ts +1 -1
- package/dist/recall-tag-filter.d.ts +1 -1
- package/dist/recall-timings.d.ts +1 -1
- package/dist/recall-xray-cli.d.ts +1 -1
- package/dist/recall-xray-renderer.d.ts +1 -1
- package/dist/recall-xray.d.ts +1 -1
- package/dist/reconcile/cursor.d.ts +1 -1
- package/dist/reconcile/manifest.d.ts +1 -1
- package/dist/reconcile/plan.d.ts +1 -1
- package/dist/replay/normalizers/chatgpt.d.ts +1 -1
- package/dist/replay/normalizers/claude.d.ts +1 -1
- package/dist/replay/normalizers/openclaw.d.ts +1 -1
- package/dist/replay/normalizers/shared.d.ts +1 -1
- package/dist/replay/runner.d.ts +1 -1
- package/dist/replay/types.d.ts +1 -1
- package/dist/replica-divergence.d.ts +1 -1
- package/dist/replica-peers-config.d.ts +1 -1
- package/dist/resolve-auth-token.d.ts +1 -1
- package/dist/resume-bundles.js +3 -3
- package/dist/retrieval-agents.d.ts +2 -2
- package/dist/retrieval-tiers.d.ts +1 -1
- package/dist/routing/engine.d.ts +1 -1
- package/dist/routing/store.d.ts +1 -1
- package/dist/salvage-envelope.d.ts +1 -1
- package/dist/schemas.d.ts +38 -38
- package/dist/{scope-profiles-NzGPDXLn.d.ts → scope-profiles-BBQsXuyj.d.ts} +1 -1
- package/dist/search/embed-helper.d.ts +1 -1
- package/dist/search/factory.d.ts +1 -1
- package/dist/search/index.d.ts +1 -1
- package/dist/search/lancedb-backend.d.ts +1 -1
- package/dist/search/meilisearch-backend.d.ts +1 -1
- package/dist/search/noop-backend.d.ts +1 -1
- package/dist/search/orama-backend.d.ts +1 -1
- package/dist/search/port.d.ts +1 -1
- package/dist/search/remote-backend.d.ts +1 -1
- package/dist/{semantic-consolidation-DLDKvyy5.d.ts → semantic-consolidation-HD8RkzVA.d.ts} +1 -1
- package/dist/semantic-consolidation.d.ts +2 -2
- package/dist/semantic-rule-verifier.d.ts +1 -1
- package/dist/{service-CQGEqL1g.d.ts → service-DIrtHHEg.d.ts} +2 -2
- package/dist/session-observer-bands.d.ts +1 -1
- package/dist/session-observer-state.d.ts +1 -1
- package/dist/shared-context/manager.d.ts +1 -1
- package/dist/signal.d.ts +1 -1
- package/dist/source-agent-qualifier.d.ts +1 -1
- package/dist/{storage-DmkCjo0l.d.ts → storage-DqCzXi56.d.ts} +1 -1
- package/dist/storage.d.ts +2 -2
- package/dist/summarizer.d.ts +1 -1
- package/dist/summary-snapshot.d.ts +1 -1
- package/dist/temporal-supersession.d.ts +2 -2
- package/dist/temporal-timeline-recall.d.ts +1 -1
- package/dist/temporal-validity.d.ts +1 -1
- package/dist/threading.d.ts +1 -1
- package/dist/tier-migration.d.ts +2 -2
- package/dist/tier-routing.d.ts +1 -1
- package/dist/topics.d.ts +1 -1
- package/dist/transcript.d.ts +1 -1
- package/dist/transfer/types.d.ts +12 -12
- package/dist/trust-score-stage.d.ts +1 -1
- package/dist/trust-score.d.ts +1 -1
- package/dist/{types-CkiDUorT.d.ts → types-d_j4vC6N.d.ts} +26 -1
- package/dist/types.d.ts +1 -1
- package/dist/utility-runtime.d.ts +1 -1
- package/dist/write-envelope.d.ts +1 -1
- package/package.json +2 -2
- package/src/wearables/cleanup.test.ts +71 -0
- package/src/wearables/cleanup.ts +163 -16
- package/src/wearables/config.test.ts +62 -0
- package/src/wearables/config.ts +99 -5
- package/src/wearables/index.ts +10 -0
- package/src/wearables/pipeline.test.ts +36 -0
- package/src/wearables/pipeline.ts +18 -3
- package/src/wearables/redaction.test.ts +156 -0
- package/src/wearables/redaction.ts +102 -10
- package/src/wearables/text-language.ts +146 -0
- package/src/wearables/types.ts +26 -0
- package/dist/chunk-NDAH7BJ5.js.map +0 -1
- package/dist/chunk-P7XI2AQ3.js.map +0 -1
- /package/dist/{auto-sync-JXEW7444.js.map → auto-sync-TDHLZPVP.js.map} +0 -0
- /package/dist/{chunk-VLL43QEQ.js.map → chunk-4IIFL7GC.js.map} +0 -0
- /package/dist/{chunk-AY43LBE4.js.map → chunk-CYF6WVE7.js.map} +0 -0
- /package/dist/{chunk-3DWTDC54.js.map → chunk-JV5YUBIZ.js.map} +0 -0
- /package/dist/{chunk-FHFERRYE.js.map → chunk-UU7TRFNE.js.map} +0 -0
package/src/wearables/cleanup.ts
CHANGED
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
* everything here is conservative and reversible by re-syncing.
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
|
+
import { detectScriptHints, type ScriptHint } from "./text-language.js";
|
|
10
11
|
import type {
|
|
11
12
|
WearableCleanupSettings,
|
|
12
13
|
WearableConversation,
|
|
@@ -33,24 +34,155 @@ const MERGE_GAP_MS = 30_000;
|
|
|
33
34
|
const GENERIC_SPEAKER_PATTERN = /^unknown$/i;
|
|
34
35
|
|
|
35
36
|
/**
|
|
36
|
-
* Standalone filler tokens stripped when `stripFillers` is on
|
|
37
|
-
*
|
|
38
|
-
* inside "umbrella" is never touched.
|
|
39
|
-
*
|
|
37
|
+
* Standalone filler tokens stripped when `stripFillers` is on, for
|
|
38
|
+
* scripts that separate words with spaces. Matched case-insensitively
|
|
39
|
+
* as whole tokens — "um" inside "umbrella" is never touched.
|
|
40
|
+
* Deliberately short and low-risk: meaning-bearing hedges ("like",
|
|
41
|
+
* "well", Spanish "este", Arabic "يعني") are NOT stripped, and neither
|
|
42
|
+
* is "mm", which is also a unit of length.
|
|
40
43
|
*/
|
|
41
|
-
const
|
|
44
|
+
const TOKEN_FILLERS: Readonly<Partial<Record<ScriptHint, readonly string[]>>> = {
|
|
45
|
+
latin: [
|
|
46
|
+
"um",
|
|
47
|
+
"uh",
|
|
48
|
+
"uhm",
|
|
49
|
+
"umm",
|
|
50
|
+
"uhh",
|
|
51
|
+
"erm",
|
|
52
|
+
"hmm",
|
|
53
|
+
"mhm",
|
|
54
|
+
"ähm",
|
|
55
|
+
"äh",
|
|
56
|
+
"ehm",
|
|
57
|
+
"euh",
|
|
58
|
+
"mmm",
|
|
59
|
+
],
|
|
60
|
+
korean: ["음", "으음", "엄", "어어"],
|
|
61
|
+
arabic: ["امم", "اممم", "اه"],
|
|
62
|
+
cyrillic: ["эм", "ээ", "ммм"],
|
|
63
|
+
};
|
|
42
64
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
65
|
+
/**
|
|
66
|
+
* Filler tokens for scripts without word spaces. These are removed
|
|
67
|
+
* inline, so each entry must be unambiguous on its own: the elongated
|
|
68
|
+
* Japanese hesitation forms and the two Chinese hesitation particles
|
|
69
|
+
* qualify, while a bare 「あの」 ("that") does not (issue #2196).
|
|
70
|
+
*/
|
|
71
|
+
const INLINE_FILLERS: Readonly<Partial<Record<ScriptHint, readonly string[]>>> = {
|
|
72
|
+
japanese: ["えーと", "えっと", "ええと", "あのー", "あのう", "うーん"],
|
|
73
|
+
han: ["呃", "嗯"],
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Punctuation that trails a filler and collapses with it, so
|
|
78
|
+
* 「えーと、では」 becomes 「では」 rather than 「、では」. Arabic sentence
|
|
79
|
+
* punctuation is included: an Arabic filler is normally followed by
|
|
80
|
+
* `،`, and without it the token survives the pass.
|
|
81
|
+
*/
|
|
82
|
+
const TRAILING_FILLER_PUNCTUATION = "[,.、。،؛؟!?]?";
|
|
83
|
+
|
|
84
|
+
interface FillerMatchers {
|
|
85
|
+
token: RegExp | null;
|
|
86
|
+
inline: RegExp | null;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Compiled matchers are cached per (script hints + extra tokens) key.
|
|
91
|
+
* A day of transcripts is thousands of segments over a handful of
|
|
92
|
+
* scripts, so this compiles a few regexes instead of one per segment.
|
|
93
|
+
*/
|
|
94
|
+
const FILLER_MATCHER_CACHE = new Map<string, FillerMatchers>();
|
|
95
|
+
|
|
96
|
+
function escapeRegExp(value: string): string {
|
|
97
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Build one alternation from `tokens`, or `null` when none remain.
|
|
102
|
+
*
|
|
103
|
+
* Each token contributes BOTH its composed and its decomposed spelling,
|
|
104
|
+
* so `ähm` also matches an ASR that emits `a\u0308hm`. Matching both
|
|
105
|
+
* forms keeps the surrounding transcript byte-identical, which
|
|
106
|
+
* normalizing the text before stripping would not.
|
|
107
|
+
*
|
|
108
|
+
* JavaScript alternation is first-match, not longest-match, so the
|
|
109
|
+
* tokens are sorted longest-first. Without that a built-in prefix such
|
|
110
|
+
* as 「あのー」 would match ahead of a longer operator token that starts
|
|
111
|
+
* with it and leave the tail in the transcript. Ties break on the token
|
|
112
|
+
* itself so the pattern is stable across runs.
|
|
113
|
+
*/
|
|
114
|
+
function buildAlternation(
|
|
115
|
+
tokens: readonly string[],
|
|
116
|
+
shape: (pattern: string) => { source: string; flags: string },
|
|
117
|
+
): RegExp | null {
|
|
118
|
+
const forms = new Set<string>();
|
|
119
|
+
for (const token of tokens) {
|
|
120
|
+
forms.add(token.normalize("NFC"));
|
|
121
|
+
forms.add(token.normalize("NFD"));
|
|
122
|
+
}
|
|
123
|
+
const unique = [...forms].sort(
|
|
124
|
+
(left, right) => right.length - left.length || (left < right ? -1 : left > right ? 1 : 0),
|
|
125
|
+
);
|
|
126
|
+
if (unique.length === 0) return null;
|
|
127
|
+
const { source, flags } = shape(unique.map((token) => escapeRegExp(token)).join("|"));
|
|
128
|
+
return new RegExp(source, flags);
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function buildFillerMatchers(
|
|
132
|
+
hints: readonly ScriptHint[],
|
|
133
|
+
extraTokens: readonly string[],
|
|
134
|
+
): FillerMatchers {
|
|
135
|
+
// Structural key: comma-joining collapses `["a,b"]` and `["a","b"]`
|
|
136
|
+
// onto one entry, and the first caller's regex would then clean the
|
|
137
|
+
// second caller's transcripts.
|
|
138
|
+
const key = JSON.stringify([hints, extraTokens]);
|
|
139
|
+
const cached = FILLER_MATCHER_CACHE.get(key);
|
|
140
|
+
if (cached) return cached;
|
|
141
|
+
|
|
142
|
+
const tokens: string[] = [];
|
|
143
|
+
const inline: string[] = [];
|
|
144
|
+
for (const hint of hints) {
|
|
145
|
+
tokens.push(...(TOKEN_FILLERS[hint] ?? []));
|
|
146
|
+
inline.push(...(INLINE_FILLERS[hint] ?? []));
|
|
147
|
+
}
|
|
148
|
+
// Operator tokens apply to every script: the operator knows their own
|
|
149
|
+
// language, and routing them by script hint would silently drop them.
|
|
150
|
+
for (const extra of extraTokens) {
|
|
151
|
+
const trimmed = extra.trim();
|
|
152
|
+
if (trimmed.length === 0) continue;
|
|
153
|
+
if (SPACE_DELIMITED_TOKEN.test(trimmed)) tokens.push(trimmed);
|
|
154
|
+
else inline.push(trimmed);
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
const matchers: FillerMatchers = {
|
|
158
|
+
token: buildAlternation(tokens, (pattern) => ({
|
|
159
|
+
source: `(?:^|\\s)(?:${pattern})${TRAILING_FILLER_PUNCTUATION}(?=\\s|$)`,
|
|
160
|
+
flags: "giu",
|
|
161
|
+
})),
|
|
162
|
+
inline: buildAlternation(inline, (pattern) => ({
|
|
163
|
+
source: `(?:${pattern})${TRAILING_FILLER_PUNCTUATION}`,
|
|
164
|
+
flags: "gu",
|
|
165
|
+
})),
|
|
166
|
+
};
|
|
167
|
+
FILLER_MATCHER_CACHE.set(key, matchers);
|
|
168
|
+
return matchers;
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/**
|
|
172
|
+
* A token whose first character belongs to a space-delimited script.
|
|
173
|
+
*
|
|
174
|
+
* Anything outside this set is removed INLINE, so a script that does
|
|
175
|
+
* space its words must appear here. Hebrew `אה` routed to the inline
|
|
176
|
+
* matcher would strip the suffix of `נראה` and corrupt the transcript.
|
|
177
|
+
*/
|
|
178
|
+
const SPACE_DELIMITED_TOKEN =
|
|
179
|
+
/^[\p{Script=Latin}\p{Script=Cyrillic}\p{Script=Greek}\p{Script=Hangul}\p{Script=Arabic}\p{Script=Hebrew}\p{N}]/u;
|
|
49
180
|
|
|
50
181
|
/** Apply configured cleanup passes to one conversation. */
|
|
51
182
|
export function cleanConversation(
|
|
52
183
|
conversation: WearableConversation,
|
|
53
184
|
settings: WearableCleanupSettings,
|
|
185
|
+
extraFillerTokens: readonly string[] = [],
|
|
54
186
|
): CleanupResult {
|
|
55
187
|
let segments = conversation.segments.map((segment) => ({ ...segment }));
|
|
56
188
|
let droppedSegments = 0;
|
|
@@ -61,7 +193,7 @@ export function cleanConversation(
|
|
|
61
193
|
|
|
62
194
|
if (settings.stripFillers) {
|
|
63
195
|
for (const segment of segments) {
|
|
64
|
-
segment.text = stripFillerTokens(segment.text);
|
|
196
|
+
segment.text = stripFillerTokens(segment.text, extraFillerTokens);
|
|
65
197
|
}
|
|
66
198
|
}
|
|
67
199
|
|
|
@@ -130,15 +262,30 @@ function canMerge(
|
|
|
130
262
|
const label = previous.speakerKey.trim();
|
|
131
263
|
if (label.length === 0 || GENERIC_SPEAKER_PATTERN.test(label)) return false;
|
|
132
264
|
}
|
|
133
|
-
const previousEnd = previous.endIso
|
|
134
|
-
|
|
265
|
+
const previousEnd = previous.endIso
|
|
266
|
+
? Date.parse(previous.endIso)
|
|
267
|
+
: Number.NaN;
|
|
268
|
+
const nextStart = next.startIso ? Date.parse(next.startIso) : Number.NaN;
|
|
135
269
|
// Without timestamps, adjacency is the only signal — still merge.
|
|
136
270
|
if (Number.isNaN(previousEnd) || Number.isNaN(nextStart)) return true;
|
|
137
271
|
return nextStart - previousEnd <= MERGE_GAP_MS;
|
|
138
272
|
}
|
|
139
273
|
|
|
140
|
-
|
|
141
|
-
|
|
274
|
+
/**
|
|
275
|
+
* Remove filler tokens from `text`. The built-in token set is chosen
|
|
276
|
+
* from the scripts present in the text, so a Japanese or Korean segment
|
|
277
|
+
* is cleaned by its own list instead of silently keeping every filler
|
|
278
|
+
* (issue #2196). Operator tokens apply to every script.
|
|
279
|
+
*/
|
|
280
|
+
export function stripFillerTokens(
|
|
281
|
+
text: string,
|
|
282
|
+
extraTokens: readonly string[] = [],
|
|
283
|
+
): string {
|
|
284
|
+
const matchers = buildFillerMatchers(detectScriptHints(text), extraTokens);
|
|
285
|
+
let result = text;
|
|
286
|
+
if (matchers.token) result = result.replace(matchers.token, " ");
|
|
287
|
+
if (matchers.inline) result = result.replace(matchers.inline, "");
|
|
288
|
+
return normalizeWhitespace(result);
|
|
142
289
|
}
|
|
143
290
|
|
|
144
291
|
/**
|
|
@@ -277,3 +277,65 @@ test("fusion knobs reject non-positive integers", () => {
|
|
|
277
277
|
test("fusion block must be an object when present", () => {
|
|
278
278
|
assert.throws(() => parseWearablesConfig({ fusion: "on" }), /must be an object/);
|
|
279
279
|
});
|
|
280
|
+
|
|
281
|
+
test("off-the-record marker phrases and filler tokens parse", () => {
|
|
282
|
+
const parsed = parseWearablesConfig({
|
|
283
|
+
offTheRecordMarkers: {
|
|
284
|
+
start: [" poza protokołem "],
|
|
285
|
+
end: ["z powrotem do protokołu"],
|
|
286
|
+
},
|
|
287
|
+
fillerTokens: ["bueno", "那个"],
|
|
288
|
+
});
|
|
289
|
+
assert.deepEqual(parsed.offTheRecordMarkers.start, ["poza protokołem"]);
|
|
290
|
+
assert.deepEqual(parsed.offTheRecordMarkers.end, ["z powrotem do protokołu"]);
|
|
291
|
+
assert.equal(parsed.offTheRecordMarkers.useBuiltIns, true);
|
|
292
|
+
assert.deepEqual(parsed.fillerTokens, ["bueno", "那个"]);
|
|
293
|
+
});
|
|
294
|
+
|
|
295
|
+
test("marker and filler lists reject unusable entries loudly", () => {
|
|
296
|
+
assert.throws(
|
|
297
|
+
() => parseWearablesConfig({ offTheRecordMarkers: { start: "poza" } }),
|
|
298
|
+
/offTheRecordMarkers\.start must be an array of strings/,
|
|
299
|
+
);
|
|
300
|
+
assert.throws(
|
|
301
|
+
() => parseWearablesConfig({ offTheRecordMarkers: { end: [" "] } }),
|
|
302
|
+
/offTheRecordMarkers\.end\[0\] must be a non-empty string/,
|
|
303
|
+
);
|
|
304
|
+
assert.throws(
|
|
305
|
+
() => parseWearablesConfig({ fillerTokens: ["a".repeat(65)] }),
|
|
306
|
+
/fillerTokens\[0\] exceeds 64 characters/,
|
|
307
|
+
);
|
|
308
|
+
assert.throws(
|
|
309
|
+
() => parseWearablesConfig({ offTheRecordMarkers: { start: ["a".repeat(129)] } }),
|
|
310
|
+
/offTheRecordMarkers\.start\[0\] exceeds 128 characters/,
|
|
311
|
+
);
|
|
312
|
+
assert.throws(
|
|
313
|
+
() => parseWearablesConfig({ offTheRecordMarkers: [] }),
|
|
314
|
+
/offTheRecordMarkers must be an object/,
|
|
315
|
+
);
|
|
316
|
+
});
|
|
317
|
+
|
|
318
|
+
test("disabling built-in markers without a start phrase is rejected, not silent", () => {
|
|
319
|
+
assert.throws(
|
|
320
|
+
() =>
|
|
321
|
+
parseWearablesConfig({
|
|
322
|
+
offTheRecordEnabled: true,
|
|
323
|
+
offTheRecordMarkers: { useBuiltIns: false },
|
|
324
|
+
}),
|
|
325
|
+
/useBuiltIns is false but no start phrase is configured/,
|
|
326
|
+
);
|
|
327
|
+
// Turning the whole feature off is the documented escape hatch.
|
|
328
|
+
const parsed = parseWearablesConfig({
|
|
329
|
+
offTheRecordEnabled: false,
|
|
330
|
+
offTheRecordMarkers: { useBuiltIns: false },
|
|
331
|
+
});
|
|
332
|
+
assert.equal(parsed.offTheRecordEnabled, false);
|
|
333
|
+
assert.equal(parsed.offTheRecordMarkers.useBuiltIns, false);
|
|
334
|
+
});
|
|
335
|
+
|
|
336
|
+
test("an unknown off-the-record marker key is rejected, not dropped", () => {
|
|
337
|
+
assert.throws(
|
|
338
|
+
() => parseWearablesConfig({ offTheRecordMarkers: { starts: ["private"] } }),
|
|
339
|
+
/offTheRecordMarkers has unknown key\(s\) starts/,
|
|
340
|
+
);
|
|
341
|
+
});
|
package/src/wearables/config.ts
CHANGED
|
@@ -13,6 +13,7 @@ import type { ImportanceLevel } from "../types.js";
|
|
|
13
13
|
import { compileCorrectionRules } from "./corrections.js";
|
|
14
14
|
import { compileRedactionPatterns } from "./redaction.js";
|
|
15
15
|
import type {
|
|
16
|
+
OffTheRecordMarkerSettings,
|
|
16
17
|
WearableCleanupSettings,
|
|
17
18
|
WearableCorrectionRule,
|
|
18
19
|
WearableMemoryMode,
|
|
@@ -111,6 +112,8 @@ export function defaultWearablesConfig(): WearablesConfig {
|
|
|
111
112
|
redactionEnabled: true,
|
|
112
113
|
redactionPatterns: [],
|
|
113
114
|
offTheRecordEnabled: true,
|
|
115
|
+
offTheRecordMarkers: { start: [], end: [], useBuiltIns: true },
|
|
116
|
+
fillerTokens: [],
|
|
114
117
|
digestEnabled: true,
|
|
115
118
|
autoSyncEnabled: DEFAULT_AUTO_SYNC_ENABLED,
|
|
116
119
|
autoSyncIntervalMinutes: DEFAULT_AUTO_SYNC_INTERVAL_MINUTES,
|
|
@@ -134,6 +137,37 @@ function requireObject(
|
|
|
134
137
|
return value as Record<string, unknown>;
|
|
135
138
|
}
|
|
136
139
|
|
|
140
|
+
/** Longest accepted off-the-record marker phrase. */
|
|
141
|
+
const MAX_MARKER_PHRASE_LENGTH = 128;
|
|
142
|
+
/** Longest accepted filler token. */
|
|
143
|
+
const MAX_FILLER_TOKEN_LENGTH = 64;
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Parse an optional array of non-empty phrases. Rejects a non-array, a
|
|
147
|
+
* non-string entry, a blank entry, and an over-long entry — an operator
|
|
148
|
+
* privacy phrase that is silently discarded is the failure this parser
|
|
149
|
+
* exists to prevent (issue #2196).
|
|
150
|
+
*/
|
|
151
|
+
function parsePhraseList(
|
|
152
|
+
value: unknown,
|
|
153
|
+
keyPath: string,
|
|
154
|
+
maxLength: number,
|
|
155
|
+
): string[] {
|
|
156
|
+
if (value === undefined) return [];
|
|
157
|
+
if (!Array.isArray(value)) {
|
|
158
|
+
throw new Error(`${keyPath} must be an array of strings`);
|
|
159
|
+
}
|
|
160
|
+
return value.map((entry, index) => {
|
|
161
|
+
if (typeof entry !== "string" || entry.trim().length === 0) {
|
|
162
|
+
throw new Error(`${keyPath}[${index}] must be a non-empty string`);
|
|
163
|
+
}
|
|
164
|
+
if (entry.length > maxLength) {
|
|
165
|
+
throw new Error(`${keyPath}[${index}] exceeds ${maxLength} characters`);
|
|
166
|
+
}
|
|
167
|
+
return entry.trim();
|
|
168
|
+
});
|
|
169
|
+
}
|
|
170
|
+
|
|
137
171
|
function parseBool(
|
|
138
172
|
value: unknown,
|
|
139
173
|
keyPath: string,
|
|
@@ -411,6 +445,68 @@ export function parseWearablesConfig(value: unknown): WearablesConfig {
|
|
|
411
445
|
compileRedactionPatterns(redactionPatterns);
|
|
412
446
|
}
|
|
413
447
|
|
|
448
|
+
const offTheRecordEnabled = parseBool(
|
|
449
|
+
raw.offTheRecordEnabled,
|
|
450
|
+
"wearables.offTheRecordEnabled",
|
|
451
|
+
defaults.offTheRecordEnabled,
|
|
452
|
+
);
|
|
453
|
+
const rawMarkers =
|
|
454
|
+
raw.offTheRecordMarkers === undefined
|
|
455
|
+
? {}
|
|
456
|
+
: requireObject(raw.offTheRecordMarkers, "wearables.offTheRecordMarkers");
|
|
457
|
+
// A typo such as `starts:` would otherwise be dropped without a word,
|
|
458
|
+
// and the operator would believe a private span is protected. Core and
|
|
459
|
+
// standalone callers never see OpenClaw's JSON schema, so the check
|
|
460
|
+
// belongs here. Keys are compared literally so the config-contract
|
|
461
|
+
// extractor can still enumerate this parser's surface.
|
|
462
|
+
const {
|
|
463
|
+
start: rawMarkerStart,
|
|
464
|
+
end: rawMarkerEnd,
|
|
465
|
+
useBuiltIns: rawMarkerUseBuiltIns,
|
|
466
|
+
...unknownMarkerKeys
|
|
467
|
+
} = rawMarkers;
|
|
468
|
+
const unknownMarkerNames = Object.keys(unknownMarkerKeys);
|
|
469
|
+
if (unknownMarkerNames.length > 0) {
|
|
470
|
+
throw new Error(
|
|
471
|
+
`wearables.offTheRecordMarkers has unknown key(s) ${unknownMarkerNames.sort().join(", ")} — valid keys are start, end, useBuiltIns`,
|
|
472
|
+
);
|
|
473
|
+
}
|
|
474
|
+
const offTheRecordMarkers: OffTheRecordMarkerSettings = {
|
|
475
|
+
start: parsePhraseList(
|
|
476
|
+
rawMarkerStart,
|
|
477
|
+
"wearables.offTheRecordMarkers.start",
|
|
478
|
+
MAX_MARKER_PHRASE_LENGTH,
|
|
479
|
+
),
|
|
480
|
+
end: parsePhraseList(
|
|
481
|
+
rawMarkerEnd,
|
|
482
|
+
"wearables.offTheRecordMarkers.end",
|
|
483
|
+
MAX_MARKER_PHRASE_LENGTH,
|
|
484
|
+
),
|
|
485
|
+
useBuiltIns: parseBool(
|
|
486
|
+
rawMarkerUseBuiltIns,
|
|
487
|
+
"wearables.offTheRecordMarkers.useBuiltIns",
|
|
488
|
+
defaults.offTheRecordMarkers.useBuiltIns,
|
|
489
|
+
),
|
|
490
|
+
};
|
|
491
|
+
// A silent no-op is the exact failure this feature exists to remove
|
|
492
|
+
// (issue #2196): turning the built-ins off without supplying a start
|
|
493
|
+
// phrase would leave the gate on and matching nothing.
|
|
494
|
+
if (
|
|
495
|
+
offTheRecordEnabled &&
|
|
496
|
+
!offTheRecordMarkers.useBuiltIns &&
|
|
497
|
+
offTheRecordMarkers.start.length === 0
|
|
498
|
+
) {
|
|
499
|
+
throw new Error(
|
|
500
|
+
"wearables.offTheRecordMarkers.useBuiltIns is false but no start phrase is configured — add wearables.offTheRecordMarkers.start, or set wearables.offTheRecordEnabled to false",
|
|
501
|
+
);
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
const fillerTokens = parsePhraseList(
|
|
505
|
+
raw.fillerTokens,
|
|
506
|
+
"wearables.fillerTokens",
|
|
507
|
+
MAX_FILLER_TOKEN_LENGTH,
|
|
508
|
+
);
|
|
509
|
+
|
|
414
510
|
const parseBoundedInt = (
|
|
415
511
|
value: unknown,
|
|
416
512
|
name: string,
|
|
@@ -488,11 +584,9 @@ export function parseWearablesConfig(value: unknown): WearablesConfig {
|
|
|
488
584
|
defaults.redactionEnabled,
|
|
489
585
|
),
|
|
490
586
|
redactionPatterns,
|
|
491
|
-
offTheRecordEnabled
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
defaults.offTheRecordEnabled,
|
|
495
|
-
),
|
|
587
|
+
offTheRecordEnabled,
|
|
588
|
+
offTheRecordMarkers,
|
|
589
|
+
fillerTokens,
|
|
496
590
|
digestEnabled: parseBool(
|
|
497
591
|
raw.digestEnabled,
|
|
498
592
|
"wearables.digestEnabled",
|
package/src/wearables/index.ts
CHANGED
|
@@ -35,10 +35,20 @@ export {
|
|
|
35
35
|
} from "./cleanup.js";
|
|
36
36
|
export {
|
|
37
37
|
applyOffTheRecord,
|
|
38
|
+
BUILT_IN_OFF_THE_RECORD_END,
|
|
39
|
+
BUILT_IN_OFF_THE_RECORD_START,
|
|
40
|
+
compileOffTheRecordMarkers,
|
|
38
41
|
compileRedactionPatterns,
|
|
39
42
|
redactText,
|
|
40
43
|
REDACTION_PLACEHOLDER,
|
|
44
|
+
type CompiledOffTheRecordMarkers,
|
|
45
|
+
type OffTheRecordMarkerInput,
|
|
41
46
|
} from "./redaction.js";
|
|
47
|
+
export {
|
|
48
|
+
buildPhraseMatcher,
|
|
49
|
+
detectScriptHints,
|
|
50
|
+
type ScriptHint,
|
|
51
|
+
} from "./text-language.js";
|
|
42
52
|
export {
|
|
43
53
|
applyCorrections,
|
|
44
54
|
compileCorrectionRule,
|
|
@@ -1039,3 +1039,39 @@ test("a transcript write failure prevents the sync watermark from advancing", as
|
|
|
1039
1039
|
rmSync(memoryDir, { recursive: true, force: true });
|
|
1040
1040
|
}
|
|
1041
1041
|
});
|
|
1042
|
+
|
|
1043
|
+
test("configured non-English off-the-record markers elide a span end to end", async () => {
|
|
1044
|
+
const memoryDir = mkdtempSync(path.join(tmpdir(), "remnic-pipeline-otr-"));
|
|
1045
|
+
try {
|
|
1046
|
+
const byDate = {
|
|
1047
|
+
"2026-06-11": [
|
|
1048
|
+
makeConversation("c1", "2026-06-11", [
|
|
1049
|
+
{ speaker: "user", isWearer: true, text: "えーと、ここからはオフレコでお願いします。" },
|
|
1050
|
+
{ speaker: "Speaker 2", text: "買収の金額は非公開です。" },
|
|
1051
|
+
{ speaker: "user", isWearer: true, text: "オンレコに戻ります。" },
|
|
1052
|
+
{ speaker: "Speaker 2", text: "bueno, 天気の話をしましょう。" },
|
|
1053
|
+
]),
|
|
1054
|
+
],
|
|
1055
|
+
};
|
|
1056
|
+
const { deps, written } = makeDeps(memoryDir);
|
|
1057
|
+
const summary = await syncWearableSource(
|
|
1058
|
+
fakeConnector(byDate),
|
|
1059
|
+
settings({ memoryMode: "off" }),
|
|
1060
|
+
config({ offTheRecordEnabled: true, fillerTokens: ["bueno"] }),
|
|
1061
|
+
{ days: 1 },
|
|
1062
|
+
deps,
|
|
1063
|
+
);
|
|
1064
|
+
|
|
1065
|
+
assert.equal(summary.transcriptsWritten.length, 1);
|
|
1066
|
+
const body = written[0]?.serialized ?? "";
|
|
1067
|
+
assert.ok(!body.includes("買収の金額は非公開です"), "off-record segment must not be stored");
|
|
1068
|
+
assert.ok(body.includes("[off the record"), "elision placeholder is visible");
|
|
1069
|
+
assert.ok(body.includes("[back on the record]"), "span closes on the Japanese end marker");
|
|
1070
|
+
assert.ok(body.includes("天気の話をしましょう"), "on-record speech survives");
|
|
1071
|
+
assert.ok(!body.includes("えーと"), "Japanese filler is stripped");
|
|
1072
|
+
assert.ok(!body.includes("bueno"), "configured filler token is stripped");
|
|
1073
|
+
assert.equal(summary.segmentsDropped, 1);
|
|
1074
|
+
} finally {
|
|
1075
|
+
rmSync(memoryDir, { recursive: true, force: true });
|
|
1076
|
+
}
|
|
1077
|
+
});
|
|
@@ -37,7 +37,13 @@ import {
|
|
|
37
37
|
type WearableMemoryGenDeps,
|
|
38
38
|
} from "./memory-gen.js";
|
|
39
39
|
import { tokenizeDayBody, type CorroborationContext } from "./trust.js";
|
|
40
|
-
import {
|
|
40
|
+
import {
|
|
41
|
+
applyOffTheRecord,
|
|
42
|
+
compileOffTheRecordMarkers,
|
|
43
|
+
compileRedactionPatterns,
|
|
44
|
+
redactText,
|
|
45
|
+
type CompiledOffTheRecordMarkers,
|
|
46
|
+
} from "./redaction.js";
|
|
41
47
|
import { loadSpeakerRegistry } from "./speakers.js";
|
|
42
48
|
import {
|
|
43
49
|
loadSyncState,
|
|
@@ -259,6 +265,7 @@ function cleanDay(
|
|
|
259
265
|
config: WearablesConfig,
|
|
260
266
|
userRedaction: RegExp[],
|
|
261
267
|
correctionRules: CompiledCorrectionRule[],
|
|
268
|
+
offTheRecordMarkers: CompiledOffTheRecordMarkers,
|
|
262
269
|
): CleanedDay {
|
|
263
270
|
const { offTheRecordEnabled, redactionEnabled } = config;
|
|
264
271
|
const out: CleanedDay = {
|
|
@@ -271,11 +278,15 @@ function cleanDay(
|
|
|
271
278
|
for (const conversation of raw) {
|
|
272
279
|
let current = conversation;
|
|
273
280
|
if (offTheRecordEnabled) {
|
|
274
|
-
const otr = applyOffTheRecord(current);
|
|
281
|
+
const otr = applyOffTheRecord(current, offTheRecordMarkers);
|
|
275
282
|
current = otr.conversation;
|
|
276
283
|
out.segmentsDropped += otr.droppedSegments;
|
|
277
284
|
}
|
|
278
|
-
const cleaned = cleanConversation(
|
|
285
|
+
const cleaned = cleanConversation(
|
|
286
|
+
current,
|
|
287
|
+
settings.cleanup,
|
|
288
|
+
config.fillerTokens,
|
|
289
|
+
);
|
|
279
290
|
current = cleaned.conversation;
|
|
280
291
|
out.segmentsDropped += cleaned.droppedSegments;
|
|
281
292
|
|
|
@@ -337,6 +348,9 @@ export async function syncWearableSource(
|
|
|
337
348
|
...compileCorrectionRules(stateRules, "state corrections"),
|
|
338
349
|
];
|
|
339
350
|
const userRedaction = compileRedactionPatterns(config.redactionPatterns);
|
|
351
|
+
const offTheRecordMarkers = compileOffTheRecordMarkers(
|
|
352
|
+
config.offTheRecordMarkers,
|
|
353
|
+
);
|
|
340
354
|
|
|
341
355
|
let syncState = await loadSyncState(deps.memoryDir);
|
|
342
356
|
const previousState = syncState.sources[connector.id];
|
|
@@ -360,6 +374,7 @@ export async function syncWearableSource(
|
|
|
360
374
|
config,
|
|
361
375
|
userRedaction,
|
|
362
376
|
correctionRules,
|
|
377
|
+
offTheRecordMarkers,
|
|
363
378
|
);
|
|
364
379
|
summary.conversations += cleaned.conversations.length;
|
|
365
380
|
summary.segmentsKept += cleaned.segmentsKept;
|