@cap-js/agents 0.9.5 → 0.9.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -131
- package/_i18n/messages_ar.properties +48 -0
- package/_i18n/messages_bg.properties +48 -0
- package/_i18n/messages_cs.properties +48 -0
- package/_i18n/messages_da.properties +48 -0
- package/_i18n/messages_de.properties +48 -0
- package/_i18n/messages_el.properties +48 -0
- package/_i18n/messages_en.properties +48 -0
- package/_i18n/messages_es.properties +48 -0
- package/_i18n/messages_es_MX.properties +48 -0
- package/_i18n/messages_fi.properties +48 -0
- package/_i18n/messages_fr.properties +48 -0
- package/_i18n/messages_he.properties +48 -0
- package/_i18n/messages_hr.properties +48 -0
- package/_i18n/messages_hu.properties +48 -0
- package/_i18n/messages_it.properties +48 -0
- package/_i18n/messages_ja.properties +48 -0
- package/_i18n/messages_kk.properties +48 -0
- package/_i18n/messages_ko.properties +48 -0
- package/_i18n/messages_ms.properties +48 -0
- package/_i18n/messages_nl.properties +48 -0
- package/_i18n/messages_no.properties +48 -0
- package/_i18n/messages_pl.properties +48 -0
- package/_i18n/messages_pt.properties +48 -0
- package/_i18n/messages_ro.properties +48 -0
- package/_i18n/messages_ru.properties +48 -0
- package/_i18n/messages_sh.properties +48 -0
- package/_i18n/messages_sk.properties +48 -0
- package/_i18n/messages_sl.properties +48 -0
- package/_i18n/messages_sv.properties +48 -0
- package/_i18n/messages_th.properties +48 -0
- package/_i18n/messages_tr.properties +48 -0
- package/_i18n/messages_uk.properties +48 -0
- package/_i18n/messages_vi.properties +48 -0
- package/_i18n/messages_zh_CN.properties +48 -0
- package/_i18n/messages_zh_TW.properties +48 -0
- package/cds-plugin.js +2 -0
- package/lib/agents/middleware/index.js +2 -0
- package/lib/agents/middleware/masking.js +124 -0
- package/lib/agents/middleware/remote-mcp.js +5 -3
- package/lib/compile.js +2 -6
- package/lib/masking/index.js +90 -0
- package/lib/masking/store.js +80 -0
- package/lib/masking/structured/findElements.js +459 -0
- package/lib/masking/structured/index.js +170 -0
- package/lib/masking/unstructured/dpi.js +84 -0
- package/lib/masking/unstructured/hana.js +64 -0
- package/lib/masking/unstructured/index.js +52 -0
- package/lib/preview/chat.html +66 -30
- package/lib/protocol/agent-card.js +2 -6
- package/lib/protocol/persistence/cleanup.js +13 -4
- package/lib/telemetry/chat-tracing.js +7 -2
- package/lib/telemetry/mlflow/tracing.js +3 -3
- package/lib/telemetry/span-masking.js +165 -0
- package/lib/telemetry/tool-tracing.js +4 -4
- package/lib/utils/markdown.js +2 -2
- package/lib/utils/utils.js +2 -34
- package/package.json +4 -2
- package/srv/handlers/graph-executor.js +52 -17
- package/srv/handlers/system-prompt.js +7 -0
- package/srv/handlers/tools.js +4 -3
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# Agent status messages (shown during task execution)
|
|
2
|
+
#XMSG Status info message send to the user client when a tool is called. {0} = tool name
|
|
3
|
+
agent_status_calling_tools=正在呼叫 {0}
|
|
4
|
+
#XMSG Status info message send to the user client when an entity is being queried. {0} = entity name
|
|
5
|
+
agent_status_querying=正在查詢 {0}
|
|
6
|
+
#XMSG Status info message send to the user client after a tool result returned and the agent is processing it
|
|
7
|
+
agent_status_processing_response=正在處理工具回應
|
|
8
|
+
#XMSG Status info message send to the user client after multiple tool results returned and the agent is processing it
|
|
9
|
+
agent_status_processing_responses=正在處理工具回應
|
|
10
|
+
agent_status_summarizing_progress=已中斷代理程式;正在彙總進度。
|
|
11
|
+
AGENT_SUMMARY_QUOTA_FALLBACK=由於工作細項已到達資源耗用限制,因此已停止。可能已完成部份工作 - 請查看結果或使用較簡單的請求重試。
|
|
12
|
+
AGENT_SUMMARY_TIMEOUT_FALLBACK=由於工作細項已到達時間限制,因此已停止。進度可能不完整,要繼續執行或停止?
|
|
13
|
+
|
|
14
|
+
# Authentication
|
|
15
|
+
#XMSG Error message for HTTP 401 status
|
|
16
|
+
UNAUTHORIZED=未授權
|
|
17
|
+
#XMSG Error message for HTTP 403 status
|
|
18
|
+
FORBIDDEN=已禁止
|
|
19
|
+
|
|
20
|
+
# Input validation
|
|
21
|
+
#XMSG Error message send to the users client when their message exceeds the limit. {0} = number of allowed characters
|
|
22
|
+
MESSAGE_TOO_LONG=訊息不可超過 {0} 個字元。
|
|
23
|
+
|
|
24
|
+
# Quota enforcement
|
|
25
|
+
QUOTA_CONCURRENT_TASKS=已達到 {0} 個同時工作細項上限,請稍後再試一次。
|
|
26
|
+
QUOTA_CONCURRENT_TASKS_PER_USER=已達到每位使用者 {0} 個同時工作細項上限,請稍後再試一次。
|
|
27
|
+
QUOTA_TASKS_PER_HOUR=此租用戶已達到每小時 {0} 個工作細項上限,請稍後再試一次。
|
|
28
|
+
QUOTA_TASKS_PER_HOUR_PER_USER=已達到每位使用者每小時 {0} 個工作細項上限,請稍後再試一次。
|
|
29
|
+
QUOTA_TOOL_CALLS_PER_HOUR=已使用每小時的工具呼叫數量上限 ({0}),請稍後再試一次。
|
|
30
|
+
QUOTA_LLM_TOKENS_PER_DAY=已使用今天的 LLM 權杖數量上限 ({0}),請明天再試一次。
|
|
31
|
+
|
|
32
|
+
# Content filter
|
|
33
|
+
CONTENT_FILTER_BLOCKED=內容安全篩選已凍結您的輸入,請重新撰寫。
|
|
34
|
+
CONTENT_FILTER_UNAVAILABLE=內容安全檢查暫時無法使用,請再試一次。
|
|
35
|
+
CONTENT_FILTER_TOOL_BLOCKED=上次呼叫的工具產生惡意結果,請嘗試不同方向以避免相同惡意資料。
|
|
36
|
+
|
|
37
|
+
# Push notifications
|
|
38
|
+
PUSH_URL_NOT_ALLOWED=推送通知回呼 URL ''{0}'' 不在允許的網域中:{1}
|
|
39
|
+
|
|
40
|
+
# HITL
|
|
41
|
+
HITL_APPROVE=核准
|
|
42
|
+
HITL_REJECT=拒絕
|
|
43
|
+
HITL_CONTINUE=繼續
|
|
44
|
+
HITL_STOP=停止
|
|
45
|
+
RESUME_REQUIRES_TEXT=繼續訊息必須包含文字 (例如:''核准'' 或 ''拒絕'') 或資料部份。
|
|
46
|
+
|
|
47
|
+
# LLM
|
|
48
|
+
LLM_UNAVAILABLE=代理程式暫時無法使用,請稍後再試一次。
|
package/cds-plugin.js
CHANGED
|
@@ -103,7 +103,9 @@ if (cds.env.requires?.telemetry)
|
|
|
103
103
|
// Schedule active_users metric computation + MLflow exporter
|
|
104
104
|
cds.on("served", async () => {
|
|
105
105
|
const { setupActiveUsersMetric } = await import("./lib/telemetry/active-users.js")
|
|
106
|
+
const { setupTraceScrubbing } = await import("./lib/telemetry/span-masking.js")
|
|
106
107
|
setupActiveUsersMetric()
|
|
108
|
+
await setupTraceScrubbing()
|
|
107
109
|
// Defer LangChain patching so the CDS model is fully loaded before patches land.
|
|
108
110
|
// opt-out via cds.env.agents.trace_langchain = false
|
|
109
111
|
if (cds.env.agents?.trace_langchain !== false) {
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
export default async function buildMiddleware(srv, options = {}) {
|
|
2
2
|
const { tools, model } = options
|
|
3
|
+
const { masking } = await import("./masking.js")
|
|
3
4
|
const { quotaEnforcerMiddleware } = await import("./quota-enforcer.js")
|
|
4
5
|
const { contentFilterMiddleware } = await import("./content-filter.js")
|
|
5
6
|
const { agentActionsMiddleware } = await import("./agent-actions.js")
|
|
@@ -18,6 +19,7 @@ export default async function buildMiddleware(srv, options = {}) {
|
|
|
18
19
|
|
|
19
20
|
return [
|
|
20
21
|
...dynamicMcpMiddlewares,
|
|
22
|
+
masking(srv), // first — all downstream middleware see pseudonymized data
|
|
21
23
|
...(await quotaEnforcerMiddleware()),
|
|
22
24
|
await contentFilterMiddleware(model),
|
|
23
25
|
await agentActionsMiddleware(),
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
import cds from "@sap/cds"
|
|
2
|
+
import { createMiddleware, ToolMessage } from "langchain"
|
|
3
|
+
import { randomBytes } from "node:crypto"
|
|
4
|
+
import { decode, encode } from "@toon-format/toon"
|
|
5
|
+
import { ensureSession, pseudonymize, resolveRemoteMcpTool } from "../../masking/index.js"
|
|
6
|
+
import { resolveArgs, needsMasking } from "../../masking/structured/index.js"
|
|
7
|
+
import z from "zod"
|
|
8
|
+
import { Command, StateSchema, ReducedValue, isCommand } from "@langchain/langgraph"
|
|
9
|
+
|
|
10
|
+
// REVISIT: Whether it is possible to make it easier without the ReducedValue class
|
|
11
|
+
const stateSchema = new StateSchema({
|
|
12
|
+
seed: new ReducedValue(z.string().default(randomBytes(16).toString("hex")), {
|
|
13
|
+
reducer: (_current, next) => next,
|
|
14
|
+
}),
|
|
15
|
+
hashToOriginal: new ReducedValue(z.map(z.string(), z.string()).default(new Map()), {
|
|
16
|
+
reducer: (_current, next) => next,
|
|
17
|
+
}),
|
|
18
|
+
})
|
|
19
|
+
|
|
20
|
+
export function masking(srv) {
|
|
21
|
+
if (!cds.env.agents?.masking) return null
|
|
22
|
+
|
|
23
|
+
return createMiddleware({
|
|
24
|
+
name: "masking",
|
|
25
|
+
stateSchema,
|
|
26
|
+
|
|
27
|
+
beforeAgent: async () => {
|
|
28
|
+
if (!needsMasking(srv.name)) return
|
|
29
|
+
await ensureSession()
|
|
30
|
+
cds.context["agent.masking.tools"] = true
|
|
31
|
+
},
|
|
32
|
+
|
|
33
|
+
wrapToolCall: async (request, handler) => {
|
|
34
|
+
const session = cds.context?.["agent.pseudonyms"]
|
|
35
|
+
if (!session || !cds.context["agent.masking.tools"]) return handler(request)
|
|
36
|
+
|
|
37
|
+
const toolName = request.toolCall?.name
|
|
38
|
+
const toolArgs = request.toolCall?.args ?? {}
|
|
39
|
+
const resolvedArgs = resolveArgs(toolArgs, session)
|
|
40
|
+
const resolvedRequest =
|
|
41
|
+
resolvedArgs !== toolArgs
|
|
42
|
+
? { ...request, toolCall: { ...request.toolCall, args: resolvedArgs } }
|
|
43
|
+
: request
|
|
44
|
+
|
|
45
|
+
const result = await handler(resolvedRequest)
|
|
46
|
+
const isCmd = isCommand(result)
|
|
47
|
+
const toolMessage = isCmd ? result.update?.messages?.[0] : result
|
|
48
|
+
|
|
49
|
+
let outMessage = toolMessage
|
|
50
|
+
if (toolMessage?.content) {
|
|
51
|
+
// Resolve remote MCP tool prefix before srv.send (cds.context differs inside event)
|
|
52
|
+
const remote = resolveRemoteMcpTool(toolName)
|
|
53
|
+
const effectiveToolName = remote?.bareName ?? toolName
|
|
54
|
+
|
|
55
|
+
// Decode TOON or JSON — pseudonymize operates on decoded objects
|
|
56
|
+
const { decoded, format } = decodeContent(toolMessage.content)
|
|
57
|
+
if (decoded) {
|
|
58
|
+
const result = await pseudonymize(
|
|
59
|
+
{
|
|
60
|
+
data: decoded,
|
|
61
|
+
type: toolArgs.cql ? "cql" : "action",
|
|
62
|
+
seed: session._seed,
|
|
63
|
+
metadata: {
|
|
64
|
+
cql: toolArgs.cql,
|
|
65
|
+
toolName: effectiveToolName,
|
|
66
|
+
...(remote?.remoteSrv && { serviceName: remote.remoteSrv.name }),
|
|
67
|
+
},
|
|
68
|
+
},
|
|
69
|
+
srv,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
session.addMappings(result?.mappings)
|
|
73
|
+
|
|
74
|
+
const encoded =
|
|
75
|
+
format === "toon"
|
|
76
|
+
? encode(result.data)
|
|
77
|
+
: format === "json"
|
|
78
|
+
? JSON.stringify(result.data)
|
|
79
|
+
: result.data
|
|
80
|
+
if (encoded !== toolMessage.content) {
|
|
81
|
+
outMessage = new ToolMessage({ ...toolMessage, content: encoded })
|
|
82
|
+
}
|
|
83
|
+
} else {
|
|
84
|
+
const session = cds.context["agent.pseudonyms"]
|
|
85
|
+
outMessage = session.scrubText(toolMessage.content)
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
const cmd = new Command({
|
|
90
|
+
update: {
|
|
91
|
+
messages: [outMessage],
|
|
92
|
+
seed: session._seed,
|
|
93
|
+
hashToOriginal: session._hashToOriginal,
|
|
94
|
+
},
|
|
95
|
+
})
|
|
96
|
+
if (isCmd) return merge(result, cmd)
|
|
97
|
+
return cmd
|
|
98
|
+
},
|
|
99
|
+
})
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function merge(c1, c2) {
|
|
103
|
+
return new Command({
|
|
104
|
+
...c1,
|
|
105
|
+
...c2,
|
|
106
|
+
update: {
|
|
107
|
+
...c1.update,
|
|
108
|
+
...c2.update,
|
|
109
|
+
},
|
|
110
|
+
})
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function decodeContent(content) {
|
|
114
|
+
if (typeof content !== "string" || !content) return {}
|
|
115
|
+
try {
|
|
116
|
+
return { decoded: decode(content), format: "toon" }
|
|
117
|
+
} catch {
|
|
118
|
+
try {
|
|
119
|
+
return { decoded: JSON.parse(content), format: "json" }
|
|
120
|
+
} catch {
|
|
121
|
+
return {}
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
}
|
|
@@ -28,19 +28,21 @@ export function remoteMcpMiddleware() {
|
|
|
28
28
|
let raw = await client.getTools()
|
|
29
29
|
if (toolFilter?.length) raw = raw.filter((t) => toolFilter.includes(t.name))
|
|
30
30
|
for (const t of raw) t.name = toolName(`${serviceName}_${t.name}`)
|
|
31
|
-
cache[mcpUrl] = raw
|
|
31
|
+
cache[mcpUrl] = { serviceName, tools: raw }
|
|
32
32
|
LOG.debug(
|
|
33
33
|
`Got ${raw.length} MCP tools from ${mcpUrl}: ${raw.map((t) => t.name).join(", ")}`,
|
|
34
34
|
)
|
|
35
35
|
}),
|
|
36
36
|
)
|
|
37
|
-
const resolved = placeholders.flatMap(({ mcpUrl }) => cache[mcpUrl] ?? [])
|
|
37
|
+
const resolved = placeholders.flatMap(({ mcpUrl }) => cache[mcpUrl]?.tools ?? [])
|
|
38
38
|
const staticTools = (request.tools ?? []).filter((t) => !t._mcpDynamic)
|
|
39
39
|
return handler({ ...request, tools: [...staticTools, ...resolved] })
|
|
40
40
|
},
|
|
41
41
|
|
|
42
42
|
wrapToolCall: async (request, handler) => {
|
|
43
|
-
const allCached = Object.values(cds.context.__mcpDynamicTools ?? {}).
|
|
43
|
+
const allCached = Object.values(cds.context.__mcpDynamicTools ?? {}).flatMap(
|
|
44
|
+
(e) => e.tools ?? [],
|
|
45
|
+
)
|
|
44
46
|
const tool = allCached.find((t) => t.name === request.toolCall.name)
|
|
45
47
|
if (!tool) return handler(request)
|
|
46
48
|
try {
|
package/lib/compile.js
CHANGED
|
@@ -1,10 +1,6 @@
|
|
|
1
1
|
import cds from "@sap/cds"
|
|
2
|
-
import {
|
|
3
|
-
|
|
4
|
-
getFilteredEntities,
|
|
5
|
-
getFilteredActions,
|
|
6
|
-
slugified,
|
|
7
|
-
} from "./utils/utils.js"
|
|
2
|
+
import { getFilteredEntities, getFilteredActions } from "@cap-js/mcp/lib/utils/tools-shared.js"
|
|
3
|
+
import { getDescription, slugified } from "./utils/utils.js"
|
|
8
4
|
import { buildAgentCard } from "./protocol/agent-card.js"
|
|
9
5
|
|
|
10
6
|
const A2A_BASE_PATH = "/a2a"
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import cds from "@sap/cds"
|
|
2
|
+
import { randomBytes } from "node:crypto"
|
|
3
|
+
import { toolName as normaliseName } from "../utils/utils.js"
|
|
4
|
+
import { PseudonymStore } from "./store.js"
|
|
5
|
+
import { pseudonymizeToolResult } from "./structured/index.js"
|
|
6
|
+
import anonymizeUnstructured from "./unstructured/index.js"
|
|
7
|
+
|
|
8
|
+
export function resolvePseudonyms(text) {
|
|
9
|
+
if (typeof text !== "string") return text
|
|
10
|
+
return cds.context?.["agent.pseudonyms"]?.resolveText(text) ?? text
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Input: { data, type, seed?, metadata? }
|
|
15
|
+
* Output: { data, mappings?, metadata: { textAnalysisResults?, seed, type } }
|
|
16
|
+
*/
|
|
17
|
+
export async function pseudonymize({ data, type, seed, metadata }, srv) {
|
|
18
|
+
const effectiveSeed = seed ?? randomBytes(16).toString("hex")
|
|
19
|
+
const result = { metadata: { seed: effectiveSeed, type } }
|
|
20
|
+
|
|
21
|
+
if (type === "unstructured") {
|
|
22
|
+
const r = await anonymizeUnstructured(data, effectiveSeed)
|
|
23
|
+
result.data = r.text
|
|
24
|
+
if (r.mappings?.length) result.mappings = r.mappings
|
|
25
|
+
if (r.textAnalysisResults) result.metadata.textAnalysisResults = r.textAnalysisResults
|
|
26
|
+
} else {
|
|
27
|
+
// data is a decoded object (caller handles TOON/JSON decode+encode)
|
|
28
|
+
const toolName = metadata?.toolName
|
|
29
|
+
const srvName = metadata?.serviceName ?? srv?.name
|
|
30
|
+
const targetSrv = cds.services?.[srvName] ?? srv
|
|
31
|
+
const model = targetSrv?.model
|
|
32
|
+
|
|
33
|
+
const store = new PseudonymStore(effectiveSeed)
|
|
34
|
+
|
|
35
|
+
pseudonymizeToolResult({
|
|
36
|
+
decoded: data,
|
|
37
|
+
toolName,
|
|
38
|
+
cql: metadata?.cql,
|
|
39
|
+
strictMasking: metadata?.strictMasking ?? false,
|
|
40
|
+
session: store,
|
|
41
|
+
srv: targetSrv,
|
|
42
|
+
model,
|
|
43
|
+
})
|
|
44
|
+
result.data = data
|
|
45
|
+
|
|
46
|
+
if (store._hashToOriginal.size) result.mappings = [...store._hashToOriginal]
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
return result
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
export async function ensureSession(checkpointer, threadId) {
|
|
53
|
+
if (cds.context?.["agent.pseudonyms"]) return
|
|
54
|
+
const { seed, hashToOriginal } = await readMaskingState(checkpointer, threadId)
|
|
55
|
+
const session = new PseudonymStore(seed, hashToOriginal)
|
|
56
|
+
cds.context["agent.pseudonyms"] = session
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
// REVISIT: Should not read checkpointer, should move into a beforeAgent hook
|
|
60
|
+
async function readMaskingState(checkpointer, thread_id) {
|
|
61
|
+
if (!checkpointer || !thread_id) return {}
|
|
62
|
+
try {
|
|
63
|
+
const tuple = await checkpointer.getTuple({ configurable: { thread_id } })
|
|
64
|
+
const values = tuple?.checkpoint?.channel_values
|
|
65
|
+
if (!values) return {}
|
|
66
|
+
return {
|
|
67
|
+
seed: values.seed,
|
|
68
|
+
hashToOriginal: values.hashToOriginal,
|
|
69
|
+
}
|
|
70
|
+
} catch {
|
|
71
|
+
return {}
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// For a remote MCP tool name like "catalogservice_query", strip the service prefix
|
|
76
|
+
// and return { bareName, remoteSrv }. Returns null for local tools.
|
|
77
|
+
export function resolveRemoteMcpTool(prefixedName) {
|
|
78
|
+
const cache = cds.context?.__mcpDynamicTools
|
|
79
|
+
if (!cache) return null
|
|
80
|
+
for (const { serviceName, tools } of Object.values(cache)) {
|
|
81
|
+
const prefix = normaliseName(`${serviceName}_`)
|
|
82
|
+
if (!tools?.some((t) => t.name === prefixedName)) continue
|
|
83
|
+
// If multiple CAP MCPs are included its possible that one of the CAP MCPs owns query if the agent itself does not expose any entities
|
|
84
|
+
return {
|
|
85
|
+
bareName: prefixedName.startsWith(prefix) ? prefixedName.slice(prefix.length) : prefixedName,
|
|
86
|
+
remoteSrv: cds.services?.[serviceName],
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
return null
|
|
90
|
+
}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
import { createHash, randomBytes } from "node:crypto"
|
|
2
|
+
|
|
3
|
+
export function generatePseudonymTag(seed, value, category) {
|
|
4
|
+
const hash = createHash("sha256")
|
|
5
|
+
.update(seed + String(value))
|
|
6
|
+
.digest("hex")
|
|
7
|
+
.slice(0, 16)
|
|
8
|
+
return `${category}-${hash}`
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
export class PseudonymStore {
|
|
12
|
+
constructor(seed, existing = new Map()) {
|
|
13
|
+
this._seed = seed ?? randomBytes(16).toString("hex")
|
|
14
|
+
this._hashToOriginal = new Map(existing)
|
|
15
|
+
this._originalToHash = new Map()
|
|
16
|
+
// Sort by descending key length so tags ("name-abc12345") are processed
|
|
17
|
+
// before bare hashes ("abc12345") and win in _originalToHash.
|
|
18
|
+
for (const [hash, original] of [...existing].sort((a, b) => b[0].length - a[0].length)) {
|
|
19
|
+
const str = String(original)
|
|
20
|
+
if (!this._originalToHash.has(str)) this._originalToHash.set(str, hash)
|
|
21
|
+
}
|
|
22
|
+
this._sortedOriginalPairs = null
|
|
23
|
+
this._sortedHashPairs = null
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
pseudonymize(value, propertyName) {
|
|
27
|
+
const str = String(value)
|
|
28
|
+
if (this._originalToHash.has(str)) return this._originalToHash.get(str)
|
|
29
|
+
const tag = generatePseudonymTag(this._seed, str, propertyName)
|
|
30
|
+
const hash = tag.slice(propertyName.length + 1)
|
|
31
|
+
this._hashToOriginal.set(tag, str)
|
|
32
|
+
this._hashToOriginal.set(hash, str)
|
|
33
|
+
this._originalToHash.set(str, tag)
|
|
34
|
+
// Invalidate sorted cache; rebuilt lazily on next scrub.
|
|
35
|
+
this._sortedOriginalPairs = null
|
|
36
|
+
this._sortedHashPairs = null
|
|
37
|
+
return tag
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
addMappings(mappings) {
|
|
41
|
+
if (!mappings?.length) return
|
|
42
|
+
// Sort by descending key length so tags ("name-abc12345") win over bare hashes ("abc12345").
|
|
43
|
+
const sorted = [...mappings].sort((a, b) => b[0].length - a[0].length)
|
|
44
|
+
for (const [hash, original] of sorted) {
|
|
45
|
+
const str = String(original)
|
|
46
|
+
const key = String(hash)
|
|
47
|
+
if (!str || !key || str === key) continue
|
|
48
|
+
if (!this._hashToOriginal.has(key)) this._hashToOriginal.set(key, str)
|
|
49
|
+
if (!this._originalToHash.has(str)) this._originalToHash.set(str, key)
|
|
50
|
+
}
|
|
51
|
+
this._sortedOriginalPairs = null
|
|
52
|
+
this._sortedHashPairs = null
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
resolve(hash) {
|
|
56
|
+
return this._hashToOriginal.get(hash) ?? hash
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
resolveText(text) {
|
|
60
|
+
if (!text || !this._hashToOriginal.size) return text
|
|
61
|
+
// tag and ID are registered. First go for tags then ID, else artefacts are left in text
|
|
62
|
+
this._sortedHashPairs ??= [...this._hashToOriginal].sort((a, b) => b[0].length - a[0].length)
|
|
63
|
+
let result = String(text)
|
|
64
|
+
for (const [hash, original] of this._sortedHashPairs) result = result.replaceAll(hash, original)
|
|
65
|
+
return result
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
scrubText(text) {
|
|
69
|
+
if (!text || !this._originalToHash.size) return text
|
|
70
|
+
// Replace longest originals first so a shorter original that is a substring
|
|
71
|
+
// of a longer one (e.g. "Emily" vs "Emily Brontë") does not corrupt it.
|
|
72
|
+
this._sortedOriginalPairs ??= [...this._originalToHash].sort(
|
|
73
|
+
(a, b) => b[0].length - a[0].length,
|
|
74
|
+
)
|
|
75
|
+
let result = String(text)
|
|
76
|
+
for (const [original, hash] of this._sortedOriginalPairs)
|
|
77
|
+
result = result.replaceAll(original, hash)
|
|
78
|
+
return result
|
|
79
|
+
}
|
|
80
|
+
}
|