@remnic/core 9.53.0 → 9.54.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/access-admin-ops-surface.d.ts +11 -11
- package/dist/access-authorization-probe.d.ts +11 -11
- package/dist/access-boundary.d.ts +11 -11
- package/dist/access-cli.js +7 -7
- package/dist/access-coding-context-resolution.d.ts +1 -1
- package/dist/access-extraction-force-flush.d.ts +11 -11
- package/dist/access-health-types.d.ts +1 -1
- package/dist/access-http-lcm-compaction.d.ts +11 -11
- package/dist/access-http-lifecycle-flush.d.ts +11 -11
- package/dist/access-http-offline-stream.d.ts +11 -11
- package/dist/access-http.d.ts +11 -11
- package/dist/access-identity-continuity-surface.d.ts +9 -9
- package/dist/access-lcm-surface.d.ts +11 -11
- package/dist/access-mcp.d.ts +11 -11
- package/dist/access-memory-search-fanout.d.ts +2 -2
- package/dist/{access-namespace-preflight-Ds6CGlCP.d.ts → access-namespace-preflight-C7n4tgWC.d.ts} +1 -1
- package/dist/access-namespace-preflight.d.ts +3 -3
- package/dist/access-observe-write-surface.d.ts +11 -11
- package/dist/access-offline-manifest.d.ts +9 -9
- package/dist/access-operations.d.ts +11 -11
- package/dist/access-recall-concurrency.d.ts +11 -11
- package/dist/access-recall-response.d.ts +11 -11
- package/dist/access-recall-surface.d.ts +11 -11
- package/dist/{access-service-CgCjvbQu.d.ts → access-service-CXvbxLK-.d.ts} +7 -7
- package/dist/access-service-helpers.d.ts +11 -11
- package/dist/access-service.d.ts +11 -11
- package/dist/access-surface-catalog.d.ts +11 -11
- package/dist/access-wearables-meetings-surface.d.ts +3 -3
- package/dist/action-confidence.d.ts +1 -1
- package/dist/active-memory-bridge.d.ts +1 -1
- package/dist/active-recall.d.ts +1 -1
- package/dist/active-recall.js +2 -2
- package/dist/ambient-provenance.d.ts +1 -1
- package/dist/artifact-search.d.ts +1 -1
- package/dist/{auto-sync-JXEW7444.js → auto-sync-TDHLZPVP.js} +3 -3
- package/dist/behavior-learner.d.ts +1 -1
- package/dist/behavior-signals.d.ts +1 -1
- package/dist/bootstrap.d.ts +9 -9
- package/dist/briefing.d.ts +2 -2
- package/dist/buffer-surprise-report.d.ts +1 -1
- package/dist/buffer-turn-helpers.d.ts +1 -1
- package/dist/buffer.d.ts +2 -2
- package/dist/bulk-import/index.d.ts +3 -3
- package/dist/calibration.d.ts +1 -1
- package/dist/capabilities.d.ts +1 -1
- package/dist/{catalog-D3B4RDy1.d.ts → catalog-BOOxl-I2.d.ts} +1 -1
- package/dist/causal-behavior.d.ts +1 -1
- package/dist/causal-consolidation.d.ts +1 -1
- package/dist/causal-trajectory-graph.d.ts +1 -1
- package/dist/{chunk-QAVH6X2A.js → chunk-47P3RYZP.js} +69 -7
- package/dist/{chunk-QAVH6X2A.js.map → chunk-47P3RYZP.js.map} +1 -1
- package/dist/{chunk-VLL43QEQ.js → chunk-4IIFL7GC.js} +2 -2
- package/dist/{chunk-AY43LBE4.js → chunk-CYF6WVE7.js} +7 -7
- package/dist/{chunk-P7XI2AQ3.js → chunk-HOIRB77A.js} +98 -19
- package/dist/chunk-HOIRB77A.js.map +1 -0
- package/dist/{chunk-NDAH7BJ5.js → chunk-HVP33QVW.js} +103 -6
- package/dist/chunk-HVP33QVW.js.map +1 -0
- package/dist/{chunk-3DWTDC54.js → chunk-JV5YUBIZ.js} +4 -4
- package/dist/{chunk-FHFERRYE.js → chunk-UU7TRFNE.js} +2 -2
- package/dist/{cli-D-HiEYhl.d.ts → cli-DVUFIa5E.d.ts} +5 -5
- package/dist/cli.d.ts +13 -13
- package/dist/cli.js +5 -5
- package/dist/coding/pre-action-gate.d.ts +1 -1
- package/dist/compounding/engine.d.ts +2 -2
- package/dist/compounding/preference-consolidator.d.ts +1 -1
- package/dist/compression-optimizer.d.ts +1 -1
- package/dist/config.d.ts +1 -1
- package/dist/config.js +2 -2
- package/dist/connectors/codex-materialize-runner.d.ts +1 -1
- package/dist/connectors/codex-materialize.d.ts +1 -1
- package/dist/connectors/index.d.ts +1 -1
- package/dist/consolidation-provenance-check.d.ts +2 -2
- package/dist/consolidation-undo.d.ts +2 -2
- package/dist/contradiction/index.d.ts +2 -2
- package/dist/converge-config.d.ts +1 -1
- package/dist/convergence-refresh.d.ts +2 -2
- package/dist/conversation-index/backend.d.ts +1 -1
- package/dist/conversation-index/chunker.d.ts +1 -1
- package/dist/conversation-index/faiss-adapter.d.ts +1 -1
- package/dist/conversation-index/indexer.d.ts +1 -1
- package/dist/conversation-index/search.d.ts +1 -1
- package/dist/corpus-watermark.d.ts +1 -1
- package/dist/day-summary.d.ts +1 -1
- package/dist/delinearize.d.ts +1 -1
- package/dist/dependency-propagation-config.d.ts +1 -1
- package/dist/{dependency-propagation-delivery-D63r24pH.d.ts → dependency-propagation-delivery-XzF76C9n.d.ts} +1 -1
- package/dist/direct-answer-wiring.d.ts +1 -1
- package/dist/direct-answer.d.ts +1 -1
- package/dist/embedding-fallback.d.ts +1 -1
- package/dist/enrichment/index.d.ts +1 -1
- package/dist/entity-retrieval.d.ts +2 -2
- package/dist/entity-schema.d.ts +1 -1
- package/dist/explicit-capture.d.ts +9 -9
- package/dist/external-wiki-access.d.ts +11 -11
- package/dist/external-wiki-collection-registration.d.ts +1 -1
- package/dist/external-wiki-collection.d.ts +1 -1
- package/dist/external-wiki-mcp-tools.d.ts +11 -11
- package/dist/extraction-error-classification.d.ts +1 -1
- package/dist/extraction-faithfulness.d.ts +1 -1
- package/dist/extraction-judge-telemetry.d.ts +1 -1
- package/dist/extraction-judge-training.d.ts +1 -1
- package/dist/extraction-judge.d.ts +1 -1
- package/dist/extraction-liveness.d.ts +1 -1
- package/dist/extraction-normalization.d.ts +1 -1
- package/dist/extraction-prompt.d.ts +1 -1
- package/dist/extraction-source-grounding-rules.d.ts +1 -1
- package/dist/extraction-source-grounding.d.ts +1 -1
- package/dist/extraction.d.ts +1 -1
- package/dist/fallback-llm.d.ts +1 -1
- package/dist/graph-dashboard-diff.d.ts +1 -1
- package/dist/graph-dashboard-key.d.ts +1 -1
- package/dist/graph-dashboard-parser.d.ts +1 -1
- package/dist/graph-edge-reinforcement.d.ts +1 -1
- package/dist/graph-path-reconstruction.d.ts +1 -1
- package/dist/graph-path-scoring.d.ts +1 -1
- package/dist/graph-snapshot.d.ts +1 -1
- package/dist/graph.d.ts +1 -1
- package/dist/importance.d.ts +1 -1
- package/dist/importers/index.d.ts +1 -1
- package/dist/in-flight-reads.d.ts +1 -1
- package/dist/index.d.ts +103 -31
- package/dist/index.js +17 -7
- package/dist/intent.d.ts +1 -1
- package/dist/lcm/engine.d.ts +1 -1
- package/dist/lcm/index.d.ts +1 -1
- package/dist/lcm/tools.d.ts +1 -1
- package/dist/lifecycle.d.ts +1 -1
- package/dist/live-connectors-runner.d.ts +1 -1
- package/dist/local-llm.d.ts +1 -1
- package/dist/local-model-endpoint.d.ts +1 -1
- package/dist/maintenance/memory-governance.d.ts +1 -1
- package/dist/maintenance/rebuild-memory-lifecycle-ledger.d.ts +2 -2
- package/dist/maintenance/rebuild-memory-projection.d.ts +2 -2
- package/dist/{maintenance-CN_Ct-Hr.d.ts → maintenance-BJzNIKBE.d.ts} +3 -3
- package/dist/mcp-memory-inspector-app.d.ts +11 -11
- package/dist/memory-action-policy.d.ts +1 -1
- package/dist/memory-cache.d.ts +1 -1
- package/dist/memory-lifecycle-ledger-utils.d.ts +1 -1
- package/dist/memory-projection-store.d.ts +1 -1
- package/dist/memory-provenance.d.ts +1 -1
- package/dist/memory-snapshot.d.ts +1 -1
- package/dist/memory-worth-outcomes.d.ts +2 -2
- package/dist/models-json.d.ts +1 -1
- package/dist/namespaces/migrate.d.ts +3 -3
- package/dist/namespaces/principal.d.ts +1 -1
- package/dist/namespaces/search.d.ts +1 -1
- package/dist/namespaces/storage.d.ts +3 -3
- package/dist/native-knowledge.d.ts +1 -1
- package/dist/offline-sync-impression-drain.d.ts +1 -1
- package/dist/operator-doctor-corpus.d.ts +1 -1
- package/dist/operator-doctor-replica.d.ts +1 -1
- package/dist/operator-toolkit.d.ts +3 -3
- package/dist/operator-toolkit.js +3 -3
- package/dist/orchestration/compression-guideline-coordinator.d.ts +2 -2
- package/dist/orchestration/maintenance.d.ts +4 -4
- package/dist/{orchestrator-B3Xe8qu0.d.ts → orchestrator-BiXgBJBQ.d.ts} +8 -8
- package/dist/orchestrator.d.ts +9 -9
- package/dist/orchestrator.js +7 -7
- package/dist/patterns-cli.d.ts +1 -1
- package/dist/{pipeline-D21Rs6gW.d.ts → pipeline-sz_dcuLg.d.ts} +1 -1
- package/dist/policy-runtime.d.ts +1 -1
- package/dist/proactive-contention.d.ts +1 -1
- package/dist/provenance.d.ts +1 -1
- package/dist/{qmd-Dhle1P4X.d.ts → qmd-eAWObK4M.d.ts} +1 -1
- package/dist/qmd-preflight.d.ts +2 -2
- package/dist/qmd-recall-cache.d.ts +1 -1
- package/dist/qmd.d.ts +2 -2
- package/dist/recall-concurrency-config.d.ts +1 -1
- package/dist/recall-disclosure-escalation.d.ts +1 -1
- package/dist/recall-explain-renderer.d.ts +1 -1
- package/dist/recall-memory-map.d.ts +1 -1
- package/dist/recall-planner-llm.d.ts +1 -1
- package/dist/recall-state.d.ts +1 -1
- package/dist/recall-tag-filter.d.ts +1 -1
- package/dist/recall-timings.d.ts +1 -1
- package/dist/recall-xray-cli.d.ts +1 -1
- package/dist/recall-xray-renderer.d.ts +1 -1
- package/dist/recall-xray.d.ts +1 -1
- package/dist/reconcile/cursor.d.ts +1 -1
- package/dist/reconcile/manifest.d.ts +1 -1
- package/dist/reconcile/plan.d.ts +1 -1
- package/dist/replay/normalizers/chatgpt.d.ts +1 -1
- package/dist/replay/normalizers/claude.d.ts +1 -1
- package/dist/replay/normalizers/openclaw.d.ts +1 -1
- package/dist/replay/normalizers/shared.d.ts +1 -1
- package/dist/replay/runner.d.ts +1 -1
- package/dist/replay/types.d.ts +1 -1
- package/dist/replica-divergence.d.ts +1 -1
- package/dist/replica-peers-config.d.ts +1 -1
- package/dist/resolve-auth-token.d.ts +1 -1
- package/dist/resume-bundles.js +3 -3
- package/dist/retrieval-agents.d.ts +2 -2
- package/dist/retrieval-tiers.d.ts +1 -1
- package/dist/routing/engine.d.ts +1 -1
- package/dist/routing/store.d.ts +1 -1
- package/dist/salvage-envelope.d.ts +1 -1
- package/dist/schemas.d.ts +38 -38
- package/dist/{scope-profiles-NzGPDXLn.d.ts → scope-profiles-BBQsXuyj.d.ts} +1 -1
- package/dist/search/embed-helper.d.ts +1 -1
- package/dist/search/factory.d.ts +1 -1
- package/dist/search/index.d.ts +1 -1
- package/dist/search/lancedb-backend.d.ts +1 -1
- package/dist/search/meilisearch-backend.d.ts +1 -1
- package/dist/search/noop-backend.d.ts +1 -1
- package/dist/search/orama-backend.d.ts +1 -1
- package/dist/search/port.d.ts +1 -1
- package/dist/search/remote-backend.d.ts +1 -1
- package/dist/{semantic-consolidation-DLDKvyy5.d.ts → semantic-consolidation-HD8RkzVA.d.ts} +1 -1
- package/dist/semantic-consolidation.d.ts +2 -2
- package/dist/semantic-rule-verifier.d.ts +1 -1
- package/dist/{service-CQGEqL1g.d.ts → service-DIrtHHEg.d.ts} +2 -2
- package/dist/session-observer-bands.d.ts +1 -1
- package/dist/session-observer-state.d.ts +1 -1
- package/dist/shared-context/manager.d.ts +1 -1
- package/dist/signal.d.ts +1 -1
- package/dist/source-agent-qualifier.d.ts +1 -1
- package/dist/{storage-DmkCjo0l.d.ts → storage-DqCzXi56.d.ts} +1 -1
- package/dist/storage.d.ts +2 -2
- package/dist/summarizer.d.ts +1 -1
- package/dist/summary-snapshot.d.ts +1 -1
- package/dist/temporal-supersession.d.ts +2 -2
- package/dist/temporal-timeline-recall.d.ts +1 -1
- package/dist/temporal-validity.d.ts +1 -1
- package/dist/threading.d.ts +1 -1
- package/dist/tier-migration.d.ts +2 -2
- package/dist/tier-routing.d.ts +1 -1
- package/dist/topics.d.ts +1 -1
- package/dist/transcript.d.ts +1 -1
- package/dist/transfer/types.d.ts +12 -12
- package/dist/trust-score-stage.d.ts +1 -1
- package/dist/trust-score.d.ts +1 -1
- package/dist/{types-CkiDUorT.d.ts → types-d_j4vC6N.d.ts} +26 -1
- package/dist/types.d.ts +1 -1
- package/dist/utility-runtime.d.ts +1 -1
- package/dist/write-envelope.d.ts +1 -1
- package/package.json +2 -2
- package/src/wearables/cleanup.test.ts +71 -0
- package/src/wearables/cleanup.ts +163 -16
- package/src/wearables/config.test.ts +62 -0
- package/src/wearables/config.ts +99 -5
- package/src/wearables/index.ts +10 -0
- package/src/wearables/pipeline.test.ts +36 -0
- package/src/wearables/pipeline.ts +18 -3
- package/src/wearables/redaction.test.ts +156 -0
- package/src/wearables/redaction.ts +102 -10
- package/src/wearables/text-language.ts +146 -0
- package/src/wearables/types.ts +26 -0
- package/dist/chunk-NDAH7BJ5.js.map +0 -1
- package/dist/chunk-P7XI2AQ3.js.map +0 -1
- /package/dist/{auto-sync-JXEW7444.js.map → auto-sync-TDHLZPVP.js.map} +0 -0
- /package/dist/{chunk-VLL43QEQ.js.map → chunk-4IIFL7GC.js.map} +0 -0
- /package/dist/{chunk-AY43LBE4.js.map → chunk-CYF6WVE7.js.map} +0 -0
- /package/dist/{chunk-3DWTDC54.js.map → chunk-JV5YUBIZ.js.map} +0 -0
- /package/dist/{chunk-FHFERRYE.js.map → chunk-UU7TRFNE.js.map} +0 -0
|
@@ -3,6 +3,7 @@ import { test } from "node:test";
|
|
|
3
3
|
|
|
4
4
|
import {
|
|
5
5
|
applyOffTheRecord,
|
|
6
|
+
compileOffTheRecordMarkers,
|
|
6
7
|
compileRedactionPatterns,
|
|
7
8
|
redactText,
|
|
8
9
|
REDACTION_PLACEHOLDER,
|
|
@@ -92,3 +93,158 @@ test("conversations without the marker pass through untouched", () => {
|
|
|
92
93
|
);
|
|
93
94
|
assert.equal(result.droppedSegments, 0);
|
|
94
95
|
});
|
|
96
|
+
|
|
97
|
+
test("built-in markers elide a Japanese off-the-record span", () => {
|
|
98
|
+
const result = applyOffTheRecord(
|
|
99
|
+
conversation([
|
|
100
|
+
"ここからはオフレコでお願いします。",
|
|
101
|
+
"来週の買収は金曜に完了します。",
|
|
102
|
+
"オンレコに戻ります。",
|
|
103
|
+
"昼食はおいしかったです。",
|
|
104
|
+
]),
|
|
105
|
+
);
|
|
106
|
+
assert.deepEqual(
|
|
107
|
+
result.conversation.segments.map((segment) => segment.text),
|
|
108
|
+
[
|
|
109
|
+
"[off the record — segment elided]",
|
|
110
|
+
"[back on the record]",
|
|
111
|
+
"昼食はおいしかったです。",
|
|
112
|
+
],
|
|
113
|
+
);
|
|
114
|
+
assert.equal(result.droppedSegments, 1);
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
test("built-in markers elide a Korean span through conversation end", () => {
|
|
118
|
+
const result = applyOffTheRecord(
|
|
119
|
+
conversation(["지금부터 오프더레코드입니다", "비밀 계약 조건입니다"]),
|
|
120
|
+
);
|
|
121
|
+
assert.equal(result.conversation.segments.length, 1);
|
|
122
|
+
assert.equal(result.droppedSegments, 1);
|
|
123
|
+
});
|
|
124
|
+
|
|
125
|
+
test("configured markers extend the built-in phrases", () => {
|
|
126
|
+
const markers = compileOffTheRecordMarkers({
|
|
127
|
+
start: ["poza protokołem"],
|
|
128
|
+
end: ["z powrotem do protokołu"],
|
|
129
|
+
});
|
|
130
|
+
const result = applyOffTheRecord(
|
|
131
|
+
conversation([
|
|
132
|
+
"To jest poza protokołem.",
|
|
133
|
+
"Tajna informacja.",
|
|
134
|
+
"Wracamy z powrotem do protokołu.",
|
|
135
|
+
"Normalna rozmowa.",
|
|
136
|
+
]),
|
|
137
|
+
markers,
|
|
138
|
+
);
|
|
139
|
+
assert.deepEqual(
|
|
140
|
+
result.conversation.segments.map((segment) => segment.text),
|
|
141
|
+
[
|
|
142
|
+
"[off the record — segment elided]",
|
|
143
|
+
"[back on the record]",
|
|
144
|
+
"Normalna rozmowa.",
|
|
145
|
+
],
|
|
146
|
+
);
|
|
147
|
+
assert.equal(result.droppedSegments, 1);
|
|
148
|
+
// The built-in English phrase still applies alongside the custom set.
|
|
149
|
+
assert.equal(
|
|
150
|
+
applyOffTheRecord(conversation(["off the record", "secret"]), markers)
|
|
151
|
+
.droppedSegments,
|
|
152
|
+
1,
|
|
153
|
+
);
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
test("useBuiltIns false honors only the configured phrases", () => {
|
|
157
|
+
const markers = compileOffTheRecordMarkers({
|
|
158
|
+
start: ["poza protokołem"],
|
|
159
|
+
useBuiltIns: false,
|
|
160
|
+
});
|
|
161
|
+
const result = applyOffTheRecord(
|
|
162
|
+
conversation(["Let me say this off the record.", "The merger closes Friday."]),
|
|
163
|
+
markers,
|
|
164
|
+
);
|
|
165
|
+
assert.equal(result.droppedSegments, 0);
|
|
166
|
+
assert.equal(result.conversation.segments.length, 2);
|
|
167
|
+
// Non-vacuous: the configured phrase must still start an elision, so
|
|
168
|
+
// this cannot pass by discarding `start` along with the built-ins.
|
|
169
|
+
assert.equal(
|
|
170
|
+
applyOffTheRecord(conversation(["poza protokołem", "secret"]), markers)
|
|
171
|
+
.droppedSegments,
|
|
172
|
+
1,
|
|
173
|
+
);
|
|
174
|
+
});
|
|
175
|
+
|
|
176
|
+
test("a Latin marker phrase never matches inside a longer word", () => {
|
|
177
|
+
const result = applyOffTheRecord(
|
|
178
|
+
conversation(["We discussed hors microphone placement.", "Normal talk."]),
|
|
179
|
+
);
|
|
180
|
+
assert.equal(result.droppedSegments, 0);
|
|
181
|
+
assert.equal(result.conversation.segments.length, 2);
|
|
182
|
+
});
|
|
183
|
+
|
|
184
|
+
test("an Arabic marker does not match inside a longer Arabic word", () => {
|
|
185
|
+
assert.equal(
|
|
186
|
+
applyOffTheRecord(conversation(["لدينا بدون تسجيلات كثيرة", "كلام عادي"]))
|
|
187
|
+
.droppedSegments,
|
|
188
|
+
0,
|
|
189
|
+
);
|
|
190
|
+
assert.equal(
|
|
191
|
+
applyOffTheRecord(conversation(["هذا بدون تسجيل من فضلك", "سر"])).droppedSegments,
|
|
192
|
+
1,
|
|
193
|
+
);
|
|
194
|
+
});
|
|
195
|
+
|
|
196
|
+
test("an Arabic proclitic still reaches the built-in marker", () => {
|
|
197
|
+
// Arabic writes و/ف/ب attached to the next word, so a leading guard
|
|
198
|
+
// would silently disable the marker for ordinary phrasing.
|
|
199
|
+
assert.equal(
|
|
200
|
+
applyOffTheRecord(conversation(["وبدون تسجيل من فضلك", "سر"])).droppedSegments,
|
|
201
|
+
1,
|
|
202
|
+
);
|
|
203
|
+
});
|
|
204
|
+
|
|
205
|
+
test("a decomposed accent is a word character, not a boundary", () => {
|
|
206
|
+
const markers = compileOffTheRecordMarkers({
|
|
207
|
+
start: ["cafe"],
|
|
208
|
+
useBuiltIns: false,
|
|
209
|
+
});
|
|
210
|
+
const decomposed = `cafe\u0301teria is open`;
|
|
211
|
+
assert.equal(applyOffTheRecord(conversation([decomposed, "x"]), markers).droppedSegments, 0);
|
|
212
|
+
assert.equal(applyOffTheRecord(conversation(["cafe is open", "x"]), markers).droppedSegments, 1);
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
test("a decomposed transcript still reaches a composed marker", () => {
|
|
216
|
+
// Some ASR output is canonically decomposed. Without normalizing the
|
|
217
|
+
// probe, `nicht fürs protokoll` would never match and the span would be
|
|
218
|
+
// persisted.
|
|
219
|
+
const decomposed = "nicht fu\u0308rs protokoll, bitte".normalize("NFD");
|
|
220
|
+
const result = applyOffTheRecord(conversation([decomposed, "geheim", "wieder fürs protokoll"]));
|
|
221
|
+
assert.equal(result.droppedSegments, 1);
|
|
222
|
+
assert.equal(result.conversation.segments[0]?.text, "[off the record — segment elided]");
|
|
223
|
+
assert.equal(result.conversation.segments[1]?.text, "[back on the record]");
|
|
224
|
+
});
|
|
225
|
+
|
|
226
|
+
test("a marker never fires at the tail of a longer word", () => {
|
|
227
|
+
// Korean particles attach at the END, so the leading guard is safe there
|
|
228
|
+
// while the trailing edge stays open.
|
|
229
|
+
const korean = compileOffTheRecordMarkers({ start: ["기록"], useBuiltIns: false });
|
|
230
|
+
assert.equal(applyOffTheRecord(conversation(["신기록 입니다", "x"]), korean).droppedSegments, 0);
|
|
231
|
+
assert.equal(applyOffTheRecord(conversation(["기록을 멈춰주세요", "비밀"]), korean).droppedSegments, 1);
|
|
232
|
+
|
|
233
|
+
// Arabic admits ONE word-initial proclitic and nothing longer.
|
|
234
|
+
const arabic = compileOffTheRecordMarkers({ start: ["خاص"], useBuiltIns: false });
|
|
235
|
+
assert.equal(applyOffTheRecord(conversation(["أشخاص كثيرون", "x"]), arabic).droppedSegments, 0);
|
|
236
|
+
assert.equal(applyOffTheRecord(conversation(["وخاص جدا", "سر"]), arabic).droppedSegments, 1);
|
|
237
|
+
});
|
|
238
|
+
|
|
239
|
+
test("a Turkish marker matches an uppercase transcript", () => {
|
|
240
|
+
// JS `/iu` folding is locale-independent: it never equates `ı` with `I`.
|
|
241
|
+
const markers = compileOffTheRecordMarkers({ start: ["kayıt dışı"], useBuiltIns: false });
|
|
242
|
+
assert.equal(
|
|
243
|
+
applyOffTheRecord(conversation(["BU KAYIT DIŞI", "gizli bilgi"]), markers).droppedSegments,
|
|
244
|
+
1,
|
|
245
|
+
);
|
|
246
|
+
assert.equal(
|
|
247
|
+
applyOffTheRecord(conversation(["bu kayıt dışı", "gizli bilgi"]), markers).droppedSegments,
|
|
248
|
+
1,
|
|
249
|
+
);
|
|
250
|
+
});
|
|
@@ -12,7 +12,11 @@
|
|
|
12
12
|
* to stay safely outside polynomial-ReDoS territory.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
-
import
|
|
15
|
+
import { buildPhraseMatcher, foldForMatching } from "./text-language.js";
|
|
16
|
+
import type {
|
|
17
|
+
OffTheRecordMarkerSettings,
|
|
18
|
+
WearableConversation,
|
|
19
|
+
} from "./types.js";
|
|
16
20
|
|
|
17
21
|
export const REDACTION_PLACEHOLDER = "[redacted]";
|
|
18
22
|
|
|
@@ -105,8 +109,88 @@ export function compileRedactionPatterns(patterns: string[]): RegExp[] {
|
|
|
105
109
|
});
|
|
106
110
|
}
|
|
107
111
|
|
|
108
|
-
|
|
109
|
-
|
|
112
|
+
/**
|
|
113
|
+
* Built-in phrases that BEGIN an off-the-record span.
|
|
114
|
+
*
|
|
115
|
+
* Conservative by design: a false positive elides real transcript
|
|
116
|
+
* content. Every phrase is a fixed multi-word expression, or a loanword
|
|
117
|
+
* that only means "off the record" (issue #2196). Operators extend this
|
|
118
|
+
* list through `wearables.offTheRecordMarkers.start`.
|
|
119
|
+
*/
|
|
120
|
+
export const BUILT_IN_OFF_THE_RECORD_START: readonly string[] = [
|
|
121
|
+
"off the record",
|
|
122
|
+
"fuera de registro",
|
|
123
|
+
"extraoficialmente",
|
|
124
|
+
"fora de registro",
|
|
125
|
+
"fora do registro",
|
|
126
|
+
"hors micro",
|
|
127
|
+
"nicht fürs protokoll",
|
|
128
|
+
"nicht für das protokoll",
|
|
129
|
+
"fuori registro",
|
|
130
|
+
"オフレコ",
|
|
131
|
+
"오프더레코드",
|
|
132
|
+
"不要记录",
|
|
133
|
+
"не для протокола",
|
|
134
|
+
"بدون تسجيل",
|
|
135
|
+
];
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Built-in phrases that END an off-the-record span.
|
|
139
|
+
*
|
|
140
|
+
* Shorter than the start list on purpose. A start phrase with no
|
|
141
|
+
* matching end phrase elides through the end of the conversation, which
|
|
142
|
+
* is the fail-closed direction; an over-eager end phrase would leak. So
|
|
143
|
+
* a language only appears here when the phrase is unambiguous.
|
|
144
|
+
*/
|
|
145
|
+
export const BUILT_IN_OFF_THE_RECORD_END: readonly string[] = [
|
|
146
|
+
"back on the record",
|
|
147
|
+
"on the record",
|
|
148
|
+
"de nuevo en registro",
|
|
149
|
+
"de volta ao registro",
|
|
150
|
+
"wieder fürs protokoll",
|
|
151
|
+
"wieder für das protokoll",
|
|
152
|
+
"オンレコ",
|
|
153
|
+
"온더레코드",
|
|
154
|
+
];
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Loose form of the parsed `OffTheRecordMarkerSettings`: callers that
|
|
158
|
+
* only override one field pass a partial, so every property is
|
|
159
|
+
* optional and read-only here.
|
|
160
|
+
*/
|
|
161
|
+
export type OffTheRecordMarkerInput = {
|
|
162
|
+
readonly [K in keyof OffTheRecordMarkerSettings]?: K extends "useBuiltIns"
|
|
163
|
+
? boolean
|
|
164
|
+
: readonly string[];
|
|
165
|
+
};
|
|
166
|
+
|
|
167
|
+
export interface CompiledOffTheRecordMarkers {
|
|
168
|
+
start: RegExp | null;
|
|
169
|
+
end: RegExp | null;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Compile marker settings into matchers. Omitting `settings` yields the
|
|
174
|
+
* built-in lists, so callers that never configured markers keep the
|
|
175
|
+
* previous behavior plus the new languages.
|
|
176
|
+
*/
|
|
177
|
+
export function compileOffTheRecordMarkers(
|
|
178
|
+
settings?: OffTheRecordMarkerInput,
|
|
179
|
+
): CompiledOffTheRecordMarkers {
|
|
180
|
+
const useBuiltIns = settings?.useBuiltIns !== false;
|
|
181
|
+
const start = [
|
|
182
|
+
...(useBuiltIns ? BUILT_IN_OFF_THE_RECORD_START : []),
|
|
183
|
+
...(settings?.start ?? []),
|
|
184
|
+
];
|
|
185
|
+
const end = [
|
|
186
|
+
...(useBuiltIns ? BUILT_IN_OFF_THE_RECORD_END : []),
|
|
187
|
+
...(settings?.end ?? []),
|
|
188
|
+
];
|
|
189
|
+
return {
|
|
190
|
+
start: buildPhraseMatcher(start),
|
|
191
|
+
end: buildPhraseMatcher(end),
|
|
192
|
+
};
|
|
193
|
+
}
|
|
110
194
|
|
|
111
195
|
export interface OffTheRecordResult {
|
|
112
196
|
conversation: WearableConversation;
|
|
@@ -114,20 +198,28 @@ export interface OffTheRecordResult {
|
|
|
114
198
|
}
|
|
115
199
|
|
|
116
200
|
/**
|
|
117
|
-
* Drop segments between a spoken
|
|
118
|
-
*
|
|
119
|
-
*
|
|
120
|
-
*
|
|
121
|
-
*
|
|
201
|
+
* Drop segments between a spoken off-the-record marker and the next
|
|
202
|
+
* back-on-the-record marker (or conversation end). The marker segments
|
|
203
|
+
* themselves are kept, with the off-record span replaced by a visible
|
|
204
|
+
* placeholder so the transcript shows that content was elided by
|
|
205
|
+
* request rather than lost.
|
|
122
206
|
*/
|
|
123
207
|
export function applyOffTheRecord(
|
|
124
208
|
conversation: WearableConversation,
|
|
209
|
+
markers: CompiledOffTheRecordMarkers = compileOffTheRecordMarkers(),
|
|
125
210
|
): OffTheRecordResult {
|
|
211
|
+
if (!markers.start) {
|
|
212
|
+
return { conversation, droppedSegments: 0 };
|
|
213
|
+
}
|
|
126
214
|
let offRecord = false;
|
|
127
215
|
let droppedSegments = 0;
|
|
128
216
|
const segments = [];
|
|
129
217
|
for (const segment of conversation.segments) {
|
|
130
|
-
|
|
218
|
+
// Matched folded, stored as written: the transcript keeps the ASR's own
|
|
219
|
+
// bytes while a decomposed or Turkic-uppercase spelling still reaches
|
|
220
|
+
// the marker.
|
|
221
|
+
const probe = foldForMatching(segment.text);
|
|
222
|
+
if (!offRecord && markers.start.test(probe)) {
|
|
131
223
|
offRecord = true;
|
|
132
224
|
segments.push({
|
|
133
225
|
...segment,
|
|
@@ -136,7 +228,7 @@ export function applyOffTheRecord(
|
|
|
136
228
|
continue;
|
|
137
229
|
}
|
|
138
230
|
if (offRecord) {
|
|
139
|
-
if (
|
|
231
|
+
if (markers.end?.test(probe)) {
|
|
140
232
|
offRecord = false;
|
|
141
233
|
segments.push({
|
|
142
234
|
...segment,
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Script-aware text helpers shared by wearable cleanup and redaction.
|
|
3
|
+
*
|
|
4
|
+
* Both features were written for space-delimited English. A spoken
|
|
5
|
+
* marker phrase or a filler token in Japanese, Korean, Chinese, Arabic,
|
|
6
|
+
* or Russian never matched, so a privacy feature silently did nothing
|
|
7
|
+
* (issue #2196). These helpers make phrase matching correct for scripts
|
|
8
|
+
* that do not separate words with spaces.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** Coarse script class used to select built-in token sets. */
|
|
12
|
+
export type ScriptHint =
|
|
13
|
+
| "latin"
|
|
14
|
+
| "japanese"
|
|
15
|
+
| "han"
|
|
16
|
+
| "korean"
|
|
17
|
+
| "arabic"
|
|
18
|
+
| "cyrillic";
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Whether a phrase edge needs a word boundary is decided per EDGE and per
|
|
22
|
+
* script, because scripts attach material at different ends.
|
|
23
|
+
*
|
|
24
|
+
* Leading edge: guarded for every script that spaces its words, so a
|
|
25
|
+
* marker cannot match at the tail of a longer word — Korean `기록` must
|
|
26
|
+
* not fire inside `신기록`, Arabic `خاص` must not fire inside `أشخاص`.
|
|
27
|
+
* Arabic and Hebrew additionally write single-letter proclitics (`و`,
|
|
28
|
+
* `ف`, `ב`, `ל`) with no space, so their guard admits ONE such letter
|
|
29
|
+
* when that letter itself starts a word: `وبدون تسجيل` still reaches the
|
|
30
|
+
* built-in `بدون تسجيل`.
|
|
31
|
+
*
|
|
32
|
+
* Trailing edge: guarded for the space-delimited scripts, so `بدون تسجيل`
|
|
33
|
+
* does not match inside `بدون تسجيلات`. Hangul is excluded — Korean
|
|
34
|
+
* particles attach to the END of a word, and guarding there would stop
|
|
35
|
+
* `기록을` from matching `기록`.
|
|
36
|
+
*
|
|
37
|
+
* Han and Kana running text has no boundary at either end: requiring one
|
|
38
|
+
* is the bug that made every non-Latin marker unreachable.
|
|
39
|
+
*/
|
|
40
|
+
const PREFIXABLE_EDGE_CHAR =
|
|
41
|
+
/[\p{Script=Latin}\p{Script=Cyrillic}\p{Script=Greek}\p{Script=Arabic}\p{Script=Hebrew}\p{Script=Hangul}\p{N}]/u;
|
|
42
|
+
const SUFFIXABLE_EDGE_CHAR =
|
|
43
|
+
/[\p{Script=Latin}\p{Script=Cyrillic}\p{Script=Greek}\p{Script=Arabic}\p{Script=Hebrew}\p{N}\p{M}]/u;
|
|
44
|
+
const PROCLITIC_EDGE_CHAR = /[\p{Script=Arabic}\p{Script=Hebrew}]/u;
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Single-letter proclitics that attach to the following word in Arabic
|
|
48
|
+
* and Hebrew. Kept to the unambiguous conjunctions and prepositions;
|
|
49
|
+
* a longer prefix is a different word, not a clitic.
|
|
50
|
+
*/
|
|
51
|
+
const PROCLITIC_LETTERS = "وفبكلسהוכלמשב";
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Combining marks count as word characters. Without `\p{M}` a decomposed
|
|
55
|
+
* `café` (base `e` plus U+0301) would let the phrase `cafe` match its
|
|
56
|
+
* prefix and elide a span nobody marked.
|
|
57
|
+
*/
|
|
58
|
+
const BOUNDARY_LOOKBEHIND = "(?<![\\p{L}\\p{M}\\p{N}])";
|
|
59
|
+
const BOUNDARY_LOOKAHEAD = "(?![\\p{L}\\p{M}\\p{N}])";
|
|
60
|
+
/** Word start, or exactly one word-initial proclitic before the phrase. */
|
|
61
|
+
const PROCLITIC_LOOKBEHIND = `(?:${BOUNDARY_LOOKBEHIND}|(?<=${BOUNDARY_LOOKBEHIND}[${PROCLITIC_LETTERS}]))`;
|
|
62
|
+
|
|
63
|
+
const HAS_KANA = /[\p{Script=Hiragana}\p{Script=Katakana}]/u;
|
|
64
|
+
const HAS_HAN = /\p{Script=Han}/u;
|
|
65
|
+
const HAS_HANGUL = /\p{Script=Hangul}/u;
|
|
66
|
+
const HAS_ARABIC = /\p{Script=Arabic}/u;
|
|
67
|
+
const HAS_CYRILLIC = /\p{Script=Cyrillic}/u;
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Normalize text for phrase matching: NFC, then fold the Turkic dotted and
|
|
71
|
+
* dotless I onto plain `i`.
|
|
72
|
+
*
|
|
73
|
+
* JavaScript's `/iu` folding is locale-independent, so it equates `i` with
|
|
74
|
+
* `I` but never `ı` with `I` or `i` with `İ`. A Turkish marker such as
|
|
75
|
+
* `kayıt dışı` would therefore miss an uppercase `KAYIT DIŞI` transcript and
|
|
76
|
+
* leave the span on the record. Both sides of the comparison run through
|
|
77
|
+
* this, so the fold cannot make one side drift from the other.
|
|
78
|
+
*/
|
|
79
|
+
export function foldForMatching(text: string): string {
|
|
80
|
+
return text.normalize("NFC").replace(/[\u0131\u0130I]/g, "i");
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/**
|
|
84
|
+
* Report which script-specific token sets apply to `text`.
|
|
85
|
+
*
|
|
86
|
+
* `latin` is always present: transcripts mix scripts freely, and the
|
|
87
|
+
* Latin filler tokens are whole-token matches that cannot fire inside
|
|
88
|
+
* non-Latin text. Japanese wins over Han when kana are present, because
|
|
89
|
+
* Japanese text also contains Han characters.
|
|
90
|
+
*/
|
|
91
|
+
export function detectScriptHints(text: string): ScriptHint[] {
|
|
92
|
+
const hints: ScriptHint[] = ["latin"];
|
|
93
|
+
if (HAS_KANA.test(text)) hints.push("japanese");
|
|
94
|
+
else if (HAS_HAN.test(text)) hints.push("han");
|
|
95
|
+
if (HAS_HANGUL.test(text)) hints.push("korean");
|
|
96
|
+
if (HAS_ARABIC.test(text)) hints.push("arabic");
|
|
97
|
+
if (HAS_CYRILLIC.test(text)) hints.push("cyrillic");
|
|
98
|
+
return hints;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
function escapeRegExp(value: string): string {
|
|
102
|
+
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Compile `phrases` into one case-insensitive matcher, or `null` when
|
|
107
|
+
* no usable phrase remains.
|
|
108
|
+
*
|
|
109
|
+
* Internal whitespace matches any whitespace run, so a transcript that
|
|
110
|
+
* breaks a phrase across a line still matches. Letter boundaries are
|
|
111
|
+
* added per edge, and only when that edge is a space-delimited script.
|
|
112
|
+
*/
|
|
113
|
+
export function buildPhraseMatcher(
|
|
114
|
+
phrases: readonly string[],
|
|
115
|
+
): RegExp | null {
|
|
116
|
+
const parts: string[] = [];
|
|
117
|
+
const seen = new Set<string>();
|
|
118
|
+
for (const raw of phrases) {
|
|
119
|
+
if (typeof raw !== "string") continue;
|
|
120
|
+
// Folded on both sides (the tester folds its input too): an ASR that
|
|
121
|
+
// emits decomposed or Turkic-uppercase text would otherwise never match
|
|
122
|
+
// its marker, and the span would be persisted (issue #2196).
|
|
123
|
+
const phrase = foldForMatching(raw.trim());
|
|
124
|
+
if (phrase.length === 0) continue;
|
|
125
|
+
const key = phrase.toLowerCase();
|
|
126
|
+
if (seen.has(key)) continue;
|
|
127
|
+
seen.add(key);
|
|
128
|
+
const characters = Array.from(phrase);
|
|
129
|
+
const body = phrase
|
|
130
|
+
.split(/\s+/)
|
|
131
|
+
.map((word) => escapeRegExp(word))
|
|
132
|
+
.join("\\s+");
|
|
133
|
+
const first = characters[0];
|
|
134
|
+
const lead = PROCLITIC_EDGE_CHAR.test(first)
|
|
135
|
+
? PROCLITIC_LOOKBEHIND
|
|
136
|
+
: PREFIXABLE_EDGE_CHAR.test(first)
|
|
137
|
+
? BOUNDARY_LOOKBEHIND
|
|
138
|
+
: "";
|
|
139
|
+
const tail = SUFFIXABLE_EDGE_CHAR.test(characters[characters.length - 1])
|
|
140
|
+
? BOUNDARY_LOOKAHEAD
|
|
141
|
+
: "";
|
|
142
|
+
parts.push(`${lead}${body}${tail}`);
|
|
143
|
+
}
|
|
144
|
+
if (parts.length === 0) return null;
|
|
145
|
+
return new RegExp(parts.join("|"), "iu");
|
|
146
|
+
}
|
package/src/wearables/types.ts
CHANGED
|
@@ -241,6 +241,20 @@ export interface WearableCorrectionRule {
|
|
|
241
241
|
sources?: string[];
|
|
242
242
|
}
|
|
243
243
|
|
|
244
|
+
/**
|
|
245
|
+
* Off-the-record marker configuration (issue #2196). Matching is
|
|
246
|
+
* case-insensitive and phrase-level. Built-in phrases stay active
|
|
247
|
+
* unless `useBuiltIns` is false.
|
|
248
|
+
*/
|
|
249
|
+
export interface OffTheRecordMarkerSettings {
|
|
250
|
+
/** Extra phrases that begin an off-the-record span. */
|
|
251
|
+
start: string[];
|
|
252
|
+
/** Extra phrases that end an off-the-record span. */
|
|
253
|
+
end: string[];
|
|
254
|
+
/** Include the built-in phrase lists. Default true. */
|
|
255
|
+
useBuiltIns: boolean;
|
|
256
|
+
}
|
|
257
|
+
|
|
244
258
|
/** Top-level wearables configuration (parsed). */
|
|
245
259
|
export interface WearablesConfig {
|
|
246
260
|
/** Master gate for the whole subsystem. Default false. */
|
|
@@ -259,6 +273,18 @@ export interface WearablesConfig {
|
|
|
259
273
|
* "back on the record" (or conversation end). Default false.
|
|
260
274
|
*/
|
|
261
275
|
offTheRecordEnabled: boolean;
|
|
276
|
+
/**
|
|
277
|
+
* Off-the-record marker phrases (issue #2196). Built-in phrases cover
|
|
278
|
+
* a documented set of languages; `start`/`end` add more, and
|
|
279
|
+
* `useBuiltIns: false` uses only the configured phrases.
|
|
280
|
+
*/
|
|
281
|
+
offTheRecordMarkers: OffTheRecordMarkerSettings;
|
|
282
|
+
/**
|
|
283
|
+
* Extra filler tokens removed when a source enables
|
|
284
|
+
* `cleanup.stripFillers`. Built-in tokens are selected per script; a
|
|
285
|
+
* language outside that set is configured here.
|
|
286
|
+
*/
|
|
287
|
+
fillerTokens: string[];
|
|
262
288
|
/**
|
|
263
289
|
* Write one compact daily-digest memory per synced source/day.
|
|
264
290
|
* Default false.
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/wearables/corrections.ts","../src/wearables/redaction.ts"],"sourcesContent":["/**\n * Wearable transcript corrections — user-specific replacement rules.\n *\n * ASR engines consistently mishear the same proper nouns for the same\n * person (\"remnick\" for \"Remnic\", a colleague's name, product jargon).\n * Rules come from two places, merged at sync time:\n *\n * 1. `wearables.corrections` in plugin config (declarative, versioned\n * with the operator's config).\n * 2. A CLI-managed rules file at `state/wearables/corrections.json`\n * (added interactively via `remnic wearables corrections add`).\n *\n * Literal rules are regex-escaped before compilation and replacements\n * are applied via a function (never a replacement string) so `$` in\n * either side can't corrupt output.\n */\n\nimport { promises as fsPromises } from \"node:fs\";\nimport * as path from \"node:path\";\n\nimport type { WearableCorrectionRule } from \"./types.js\";\n\nexport interface CompiledCorrectionRule {\n rule: WearableCorrectionRule;\n pattern: RegExp;\n}\n\nexport interface CorrectionApplication {\n text: string;\n applied: number;\n}\n\n/** Hard cap on rule pattern length (bounds hostile/pathological regexes). */\nconst MAX_PATTERN_LENGTH = 256;\n\nfunction escapeRegExp(value: string): string {\n return value.replace(/[.*+?^${}()|[\\]\\\\]/g, \"\\\\$&\");\n}\n\n/**\n * Validate and compile a correction rule. Throws a descriptive error on\n * invalid input (empty match, regex that doesn't compile, regex that\n * matches the empty string) — callers surface this at config parse or\n * CLI time rather than skipping the rule silently.\n */\nexport function compileCorrectionRule(\n rule: WearableCorrectionRule,\n label: string,\n): CompiledCorrectionRule {\n if (typeof rule.match !== \"string\" || rule.match.length === 0) {\n throw new Error(`${label}: match must be a non-empty string`);\n }\n if (rule.match.length > MAX_PATTERN_LENGTH) {\n throw new Error(\n `${label}: match exceeds ${MAX_PATTERN_LENGTH} characters — correction patterns must stay short`,\n );\n }\n if (typeof rule.replace !== \"string\") {\n throw new Error(`${label}: replace must be a string`);\n }\n const flags = rule.caseInsensitive === false ? \"g\" : \"gi\";\n let pattern: RegExp;\n if (rule.regex === true) {\n try {\n // Operator-supplied regexes are the documented feature here\n // (rules live in the operator's own config / state file, never in\n // request input); the length cap above bounds pathological\n // patterns. CodeQL js/regex-injection is dismissed by design for\n // this site.\n pattern = new RegExp(rule.match, flags);\n } catch (err) {\n throw new Error(\n `${label}: match is not a valid regular expression: ${\n err instanceof Error ? err.message : String(err)\n }`,\n );\n }\n } else {\n // Literal rules match on word boundaries when both edges of the\n // match are word characters, so \"remnick\" doesn't fire inside\n // \"remnickson\" unless the user opts into regex mode.\n const escaped = escapeRegExp(rule.match);\n const leading = /^[\\p{L}\\p{N}_]/u.test(rule.match) ? \"\\\\b\" : \"\";\n const trailing = /[\\p{L}\\p{N}_]$/u.test(rule.match) ? \"\\\\b\" : \"\";\n pattern = new RegExp(`${leading}${escaped}${trailing}`, flags);\n }\n if (pattern.test(\"\")) {\n throw new Error(\n `${label}: pattern matches the empty string and would corrupt every transcript`,\n );\n }\n pattern.lastIndex = 0;\n return { rule, pattern };\n}\n\n/** Compile a rule list, labeling errors with their index. */\nexport function compileCorrectionRules(\n rules: WearableCorrectionRule[],\n labelPrefix: string,\n): CompiledCorrectionRule[] {\n return rules.map((rule, index) =>\n compileCorrectionRule(rule, `${labelPrefix}[${index}]`),\n );\n}\n\n/** Apply every applicable rule to a piece of transcript text. */\nexport function applyCorrections(\n text: string,\n rules: CompiledCorrectionRule[],\n sourceId: string,\n): CorrectionApplication {\n let applied = 0;\n let result = text;\n for (const { rule, pattern } of rules) {\n if (\n Array.isArray(rule.sources) &&\n rule.sources.length > 0 &&\n !rule.sources.includes(sourceId)\n ) {\n continue;\n }\n pattern.lastIndex = 0;\n result = result.replace(pattern, () => {\n applied += 1;\n // Replacement via function: `$` in rule.replace stays literal.\n return rule.replace;\n });\n }\n return { text: result, applied };\n}\n\n// ---------------------------------------------------------------------------\n// CLI-managed rules file\n// ---------------------------------------------------------------------------\n\ninterface CorrectionsFileShape {\n version: 1;\n rules: WearableCorrectionRule[];\n}\n\nexport function correctionsFilePath(memoryDir: string): string {\n return path.join(memoryDir, \"state\", \"wearables\", \"corrections.json\");\n}\n\n/**\n * Load CLI-managed correction rules. A missing file means no rules; a\n * malformed file throws (operators should know their corrections are\n * not being applied rather than silently losing them).\n */\nexport async function loadCorrectionsFile(\n memoryDir: string,\n): Promise<WearableCorrectionRule[]> {\n const filePath = correctionsFilePath(memoryDir);\n let raw: string;\n try {\n raw = await fsPromises.readFile(filePath, \"utf-8\");\n } catch (err) {\n if ((err as NodeJS.ErrnoException).code === \"ENOENT\") return [];\n throw err;\n }\n let parsed: unknown;\n try {\n parsed = JSON.parse(raw);\n } catch (err) {\n throw new Error(\n `wearables corrections file is not valid JSON (state/wearables/corrections.json): ${\n err instanceof Error ? err.message : String(err)\n }`,\n );\n }\n if (\n typeof parsed !== \"object\" ||\n parsed === null ||\n Array.isArray(parsed) ||\n !Array.isArray((parsed as CorrectionsFileShape).rules)\n ) {\n throw new Error(\n 'wearables corrections file has an unexpected shape (state/wearables/corrections.json); expected {\"version\":1,\"rules\":[...]}',\n );\n }\n const rules = (parsed as CorrectionsFileShape).rules;\n // Validate every persisted rule up front so a hand-edited bad rule\n // fails at load with its index, not mid-sync.\n compileCorrectionRules(rules, \"state corrections\");\n return rules;\n}\n\n/** Persist CLI-managed rules atomically (temp file + rename). */\nexport async function saveCorrectionsFile(\n memoryDir: string,\n rules: WearableCorrectionRule[],\n): Promise<void> {\n compileCorrectionRules(rules, \"state corrections\");\n const filePath = correctionsFilePath(memoryDir);\n await fsPromises.mkdir(path.dirname(filePath), { recursive: true });\n const payload: CorrectionsFileShape = { version: 1, rules };\n const tmpPath = `${filePath}.tmp-${process.pid}-${Date.now().toString(36)}`;\n await fsPromises.writeFile(\n tmpPath,\n `${JSON.stringify(payload, null, 2)}\\n`,\n \"utf-8\",\n );\n try {\n await fsPromises.rename(tmpPath, filePath);\n } catch (err) {\n // Clean up the temp file on rename failure; the original (if any)\n // is untouched.\n await fsPromises.unlink(tmpPath).catch(() => undefined);\n throw err;\n }\n}\n","/**\n * Wearable transcript redaction — privacy guard applied before any\n * transcript text is persisted or fed to extraction.\n *\n * Always-on recorders capture things nobody intended to store: card\n * numbers read aloud, SSNs dictated to a pharmacy line. Built-in\n * patterns cover the unambiguous, high-sensitivity cases; users can add\n * their own regexes via `wearables.redactionPatterns` (validated at\n * config parse — invalid patterns are rejected loudly, never ignored).\n *\n * All built-in patterns are simple linear scans (no nested quantifiers)\n * to stay safely outside polynomial-ReDoS territory.\n */\n\nimport type { WearableConversation } from \"./types.js\";\n\nexport const REDACTION_PLACEHOLDER = \"[redacted]\";\n\n/**\n * Built-in patterns. Conservative by design — false positives erase\n * real transcript content, so each pattern targets formats that are\n * near-certain PII:\n * - US SSN with separators (123-45-6789). Bare 9-digit runs are NOT\n * matched (too many false positives: ids, tracking numbers).\n * - Payment-card-like runs: 13–19 digits in groups separated by\n * spaces/dashes (4111 1111 1111 1111) or contiguous 15–16 digits.\n */\nconst BUILT_IN_PATTERNS: RegExp[] = [\n /\\b\\d{3}-\\d{2}-\\d{4}\\b/g,\n // Starts and ENDS on a digit so a trailing separator is never\n // consumed (replacing it would glue the placeholder to the next word).\n /\\b\\d(?:[ -]?\\d){12,18}\\b/g,\n];\n\n/** Minimum digit count before a digit-run is treated as a card number. */\nconst CARD_MIN_DIGITS = 13;\n\nexport interface RedactionResult {\n text: string;\n redactions: number;\n}\n\nexport function redactText(\n text: string,\n userPatterns: RegExp[],\n): RedactionResult {\n let redactions = 0;\n let result = text;\n\n // SSN pattern first (more specific than the digit-run pattern).\n result = result.replace(BUILT_IN_PATTERNS[0], () => {\n redactions += 1;\n return REDACTION_PLACEHOLDER;\n });\n\n // Digit-run pattern with a post-match digit-count check so short\n // grouped numbers (\"call 555 0125 today\") survive.\n result = result.replace(BUILT_IN_PATTERNS[1], (match) => {\n const digits = match.replace(/\\D/g, \"\");\n if (digits.length < CARD_MIN_DIGITS || digits.length > 19) {\n return match;\n }\n redactions += 1;\n return REDACTION_PLACEHOLDER;\n });\n\n for (const pattern of userPatterns) {\n result = result.replace(pattern, () => {\n redactions += 1;\n return REDACTION_PLACEHOLDER;\n });\n }\n\n return { text: result, redactions };\n}\n\n/**\n * Compile user-supplied redaction patterns. Throws with a descriptive\n * message on the first invalid pattern — config parsing surfaces this\n * to the operator instead of silently skipping the rule.\n */\nexport function compileRedactionPatterns(patterns: string[]): RegExp[] {\n return patterns.map((pattern, index) => {\n if (typeof pattern !== \"string\" || pattern.trim().length === 0) {\n throw new Error(\n `wearables.redactionPatterns[${index}] must be a non-empty string`,\n );\n }\n if (pattern.length > 256) {\n throw new Error(\n `wearables.redactionPatterns[${index}] exceeds 256 characters — redaction patterns must stay short`,\n );\n }\n try {\n // Operator-supplied regexes from the operator's own config —\n // length-capped above; never request input.\n return new RegExp(pattern, \"gi\");\n } catch (err) {\n throw new Error(\n `wearables.redactionPatterns[${index}] is not a valid regular expression: ${\n err instanceof Error ? err.message : String(err)\n }`,\n );\n }\n });\n}\n\nconst OFF_THE_RECORD = /\\boff\\s+the\\s+record\\b/i;\nconst BACK_ON_THE_RECORD = /\\b(?:back\\s+)?on\\s+the\\s+record\\b/i;\n\nexport interface OffTheRecordResult {\n conversation: WearableConversation;\n droppedSegments: number;\n}\n\n/**\n * Drop segments between a spoken \"off the record\" marker and the next\n * \"(back) on the record\" marker (or conversation end). The marker\n * segments themselves are kept, with the off-record span replaced by a\n * visible placeholder so the transcript shows that content was elided\n * by request rather than lost.\n */\nexport function applyOffTheRecord(\n conversation: WearableConversation,\n): OffTheRecordResult {\n let offRecord = false;\n let droppedSegments = 0;\n const segments = [];\n for (const segment of conversation.segments) {\n if (!offRecord && OFF_THE_RECORD.test(segment.text)) {\n offRecord = true;\n segments.push({\n ...segment,\n text: \"[off the record — segment elided]\",\n });\n continue;\n }\n if (offRecord) {\n if (BACK_ON_THE_RECORD.test(segment.text)) {\n offRecord = false;\n segments.push({\n ...segment,\n text: \"[back on the record]\",\n });\n } else {\n droppedSegments += 1;\n }\n continue;\n }\n segments.push(segment);\n }\n return {\n conversation: { ...conversation, segments },\n droppedSegments,\n };\n}\n"],"mappings":";AAiBA,SAAS,YAAY,kBAAkB;AACvC,YAAY,UAAU;AAetB,IAAM,qBAAqB;AAE3B,SAAS,aAAa,OAAuB;AAC3C,SAAO,MAAM,QAAQ,uBAAuB,MAAM;AACpD;AAQO,SAAS,sBACd,MACA,OACwB;AACxB,MAAI,OAAO,KAAK,UAAU,YAAY,KAAK,MAAM,WAAW,GAAG;AAC7D,UAAM,IAAI,MAAM,GAAG,KAAK,oCAAoC;AAAA,EAC9D;AACA,MAAI,KAAK,MAAM,SAAS,oBAAoB;AAC1C,UAAM,IAAI;AAAA,MACR,GAAG,KAAK,mBAAmB,kBAAkB;AAAA,IAC/C;AAAA,EACF;AACA,MAAI,OAAO,KAAK,YAAY,UAAU;AACpC,UAAM,IAAI,MAAM,GAAG,KAAK,4BAA4B;AAAA,EACtD;AACA,QAAM,QAAQ,KAAK,oBAAoB,QAAQ,MAAM;AACrD,MAAI;AACJ,MAAI,KAAK,UAAU,MAAM;AACvB,QAAI;AAMF,gBAAU,IAAI,OAAO,KAAK,OAAO,KAAK;AAAA,IACxC,SAAS,KAAK;AACZ,YAAM,IAAI;AAAA,QACR,GAAG,KAAK,8CACN,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,CACjD;AAAA,MACF;AAAA,IACF;AAAA,EACF,OAAO;AAIL,UAAM,UAAU,aAAa,KAAK,KAAK;AACvC,UAAM,UAAU,kBAAkB,KAAK,KAAK,KAAK,IAAI,QAAQ;AAC7D,UAAM,WAAW,kBAAkB,KAAK,KAAK,KAAK,IAAI,QAAQ;AAC9D,cAAU,IAAI,OAAO,GAAG,OAAO,GAAG,OAAO,GAAG,QAAQ,IAAI,KAAK;AAAA,EAC/D;AACA,MAAI,QAAQ,KAAK,EAAE,GAAG;AACpB,UAAM,IAAI;AAAA,MACR,GAAG,KAAK;AAAA,IACV;AAAA,EACF;AACA,UAAQ,YAAY;AACpB,SAAO,EAAE,MAAM,QAAQ;AACzB;AAGO,SAAS,uBACd,OACA,aAC0B;AAC1B,SAAO,MAAM;AAAA,IAAI,CAAC,MAAM,UACtB,sBAAsB,MAAM,GAAG,WAAW,IAAI,KAAK,GAAG;AAAA,EACxD;AACF;AAGO,SAAS,iBACd,MACA,OACA,UACuB;AACvB,MAAI,UAAU;AACd,MAAI,SAAS;AACb,aAAW,EAAE,MAAM,QAAQ,KAAK,OAAO;AACrC,QACE,MAAM,QAAQ,KAAK,OAAO,KAC1B,KAAK,QAAQ,SAAS,KACtB,CAAC,KAAK,QAAQ,SAAS,QAAQ,GAC/B;AACA;AAAA,IACF;AACA,YAAQ,YAAY;AACpB,aAAS,OAAO,QAAQ,SAAS,MAAM;AACrC,iBAAW;AAEX,aAAO,KAAK;AAAA,IACd,CAAC;AAAA,EACH;AACA,SAAO,EAAE,MAAM,QAAQ,QAAQ;AACjC;AAWO,SAAS,oBAAoB,WAA2B;AAC7D,SAAY,UAAK,WAAW,SAAS,aAAa,kBAAkB;AACtE;AAOA,eAAsB,oBACpB,WACmC;AACnC,QAAM,WAAW,oBAAoB,SAAS;AAC9C,MAAI;AACJ,MAAI;AACF,UAAM,MAAM,WAAW,SAAS,UAAU,OAAO;AAAA,EACnD,SAAS,KAAK;AACZ,QAAK,IAA8B,SAAS,SAAU,QAAO,CAAC;AAC9D,UAAM;AAAA,EACR;AACA,MAAI;AACJ,MAAI;AACF,aAAS,KAAK,MAAM,GAAG;AAAA,EACzB,SAAS,KAAK;AACZ,UAAM,IAAI;AAAA,MACR,oFACE,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,CACjD;AAAA,IACF;AAAA,EACF;AACA,MACE,OAAO,WAAW,YAClB,WAAW,QACX,MAAM,QAAQ,MAAM,KACpB,CAAC,MAAM,QAAS,OAAgC,KAAK,GACrD;AACA,UAAM,IAAI;AAAA,MACR;AAAA,IACF;AAAA,EACF;AACA,QAAM,QAAS,OAAgC;AAG/C,yBAAuB,OAAO,mBAAmB;AACjD,SAAO;AACT;AAGA,eAAsB,oBACpB,WACA,OACe;AACf,yBAAuB,OAAO,mBAAmB;AACjD,QAAM,WAAW,oBAAoB,SAAS;AAC9C,QAAM,WAAW,MAAW,aAAQ,QAAQ,GAAG,EAAE,WAAW,KAAK,CAAC;AAClE,QAAM,UAAgC,EAAE,SAAS,GAAG,MAAM;AAC1D,QAAM,UAAU,GAAG,QAAQ,QAAQ,QAAQ,GAAG,IAAI,KAAK,IAAI,EAAE,SAAS,EAAE,CAAC;AACzE,QAAM,WAAW;AAAA,IACf;AAAA,IACA,GAAG,KAAK,UAAU,SAAS,MAAM,CAAC,CAAC;AAAA;AAAA,IACnC;AAAA,EACF;AACA,MAAI;AACF,UAAM,WAAW,OAAO,SAAS,QAAQ;AAAA,EAC3C,SAAS,KAAK;AAGZ,UAAM,WAAW,OAAO,OAAO,EAAE,MAAM,MAAM,MAAS;AACtD,UAAM;AAAA,EACR;AACF;;;AClMO,IAAM,wBAAwB;AAWrC,IAAM,oBAA8B;AAAA,EAClC;AAAA;AAAA;AAAA,EAGA;AACF;AAGA,IAAM,kBAAkB;AAOjB,SAAS,WACd,MACA,cACiB;AACjB,MAAI,aAAa;AACjB,MAAI,SAAS;AAGb,WAAS,OAAO,QAAQ,kBAAkB,CAAC,GAAG,MAAM;AAClD,kBAAc;AACd,WAAO;AAAA,EACT,CAAC;AAID,WAAS,OAAO,QAAQ,kBAAkB,CAAC,GAAG,CAAC,UAAU;AACvD,UAAM,SAAS,MAAM,QAAQ,OAAO,EAAE;AACtC,QAAI,OAAO,SAAS,mBAAmB,OAAO,SAAS,IAAI;AACzD,aAAO;AAAA,IACT;AACA,kBAAc;AACd,WAAO;AAAA,EACT,CAAC;AAED,aAAW,WAAW,cAAc;AAClC,aAAS,OAAO,QAAQ,SAAS,MAAM;AACrC,oBAAc;AACd,aAAO;AAAA,IACT,CAAC;AAAA,EACH;AAEA,SAAO,EAAE,MAAM,QAAQ,WAAW;AACpC;AAOO,SAAS,yBAAyB,UAA8B;AACrE,SAAO,SAAS,IAAI,CAAC,SAAS,UAAU;AACtC,QAAI,OAAO,YAAY,YAAY,QAAQ,KAAK,EAAE,WAAW,GAAG;AAC9D,YAAM,IAAI;AAAA,QACR,+BAA+B,KAAK;AAAA,MACtC;AAAA,IACF;AACA,QAAI,QAAQ,SAAS,KAAK;AACxB,YAAM,IAAI;AAAA,QACR,+BAA+B,KAAK;AAAA,MACtC;AAAA,IACF;AACA,QAAI;AAGF,aAAO,IAAI,OAAO,SAAS,IAAI;AAAA,IACjC,SAAS,KAAK;AACZ,YAAM,IAAI;AAAA,QACR,+BAA+B,KAAK,wCAClC,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG,CACjD;AAAA,MACF;AAAA,IACF;AAAA,EACF,CAAC;AACH;AAEA,IAAM,iBAAiB;AACvB,IAAM,qBAAqB;AAcpB,SAAS,kBACd,cACoB;AACpB,MAAI,YAAY;AAChB,MAAI,kBAAkB;AACtB,QAAM,WAAW,CAAC;AAClB,aAAW,WAAW,aAAa,UAAU;AAC3C,QAAI,CAAC,aAAa,eAAe,KAAK,QAAQ,IAAI,GAAG;AACnD,kBAAY;AACZ,eAAS,KAAK;AAAA,QACZ,GAAG;AAAA,QACH,MAAM;AAAA,MACR,CAAC;AACD;AAAA,IACF;AACA,QAAI,WAAW;AACb,UAAI,mBAAmB,KAAK,QAAQ,IAAI,GAAG;AACzC,oBAAY;AACZ,iBAAS,KAAK;AAAA,UACZ,GAAG;AAAA,UACH,MAAM;AAAA,QACR,CAAC;AAAA,MACH,OAAO;AACL,2BAAmB;AAAA,MACrB;AACA;AAAA,IACF;AACA,aAAS,KAAK,OAAO;AAAA,EACvB;AACA,SAAO;AAAA,IACL,cAAc,EAAE,GAAG,cAAc,SAAS;AAAA,IAC1C;AAAA,EACF;AACF;","names":[]}
|