@onco-foundry/mask-port 0.1.6 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -0
- package/dist/create_masker.d.ts +3 -0
- package/dist/create_masker.js +3 -0
- package/dist/index.d.ts +3 -1
- package/dist/index.js +3 -1
- package/dist/llm_sensitive_word_finder.d.ts +4 -1
- package/dist/llm_sensitive_word_finder.js +7 -4
- package/dist/paddle_masker.d.ts +14 -0
- package/dist/paddle_masker.js +252 -0
- package/dist/sensitive_word_finder.d.ts +10 -0
- package/dist/sensitive_word_finder.js +11 -0
- package/dist/textin_masker.d.ts +2 -10
- package/dist/textin_masker.js +1 -11
- package/package.json +3 -3
package/README.md
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# mask-port
|
|
2
|
+
|
|
3
|
+
`createLlmSensitiveWordFinder` 使用 Anthropic 兼容 Messages API 提取敏感词,供图片脱敏器定位。
|
|
4
|
+
|
|
5
|
+
## 模型请求预算
|
|
6
|
+
|
|
7
|
+
- `maxTokens`:正整数,默认 **65536**(此前固定为 16384)。推理型模型的思考可能占用同一预算;实际消耗由模型响应决定,设置上限不等于每次用满。
|
|
8
|
+
- `timeoutMs`:正整数,默认 **300000 ms**(此前为 120000 ms)。调用方包裹的 OCR 请求也需提供足够的总超时;内层增加超时不会延长外层截止时间。
|
|
9
|
+
|
|
10
|
+
按供应商和所选模型支持的输出上限设置 `maxTokens`,支持较小预算的模型需显式覆盖默认值。无效配置在创建判定器时抛错。配置调整不改变输出格式校验,也不会将缺失输出视为脱敏成功。
|
package/dist/create_masker.d.ts
CHANGED
|
@@ -2,6 +2,7 @@ import type { Masker } from './masker.ts';
|
|
|
2
2
|
import { type QwenAgentMaskerOptions } from './qwen_agent_masker.ts';
|
|
3
3
|
import { type TencentMaskerOptions } from './tencent_masker.ts';
|
|
4
4
|
import { type TextInMaskerOptions } from './textin_masker.ts';
|
|
5
|
+
import { type PaddleMaskerOptions } from './paddle_masker.ts';
|
|
5
6
|
/**
|
|
6
7
|
* 脱敏器的装配选项:kind 判别用哪个实现。
|
|
7
8
|
* 加新供应商时在这里加一个 union 成员,调用方只改 kind。
|
|
@@ -13,6 +14,8 @@ export type MaskerOptions = {
|
|
|
13
14
|
} & TencentMaskerOptions) | ({
|
|
14
15
|
readonly kind: 'textin';
|
|
15
16
|
} & TextInMaskerOptions) | ({
|
|
17
|
+
readonly kind: 'paddle';
|
|
18
|
+
} & PaddleMaskerOptions) | ({
|
|
16
19
|
readonly kind: 'qwen-agent-name';
|
|
17
20
|
} & QwenAgentMaskerOptions);
|
|
18
21
|
/** 脱敏端口的唯一工厂:换供应商只改 kind,调用方不 import 具体实现。 */
|
package/dist/create_masker.js
CHANGED
|
@@ -2,6 +2,7 @@ import { createFakeMasker } from './fake_masker.js';
|
|
|
2
2
|
import { createQwenAgentMasker, } from './qwen_agent_masker.js';
|
|
3
3
|
import { createTencentMasker } from './tencent_masker.js';
|
|
4
4
|
import { createTextInMasker } from './textin_masker.js';
|
|
5
|
+
import { createPaddleMasker } from './paddle_masker.js';
|
|
5
6
|
/** 脱敏端口的唯一工厂:换供应商只改 kind,调用方不 import 具体实现。 */
|
|
6
7
|
export const createMasker = async (options) => {
|
|
7
8
|
switch (options.kind) {
|
|
@@ -11,6 +12,8 @@ export const createMasker = async (options) => {
|
|
|
11
12
|
return createTencentMasker(options);
|
|
12
13
|
case 'textin':
|
|
13
14
|
return createTextInMasker(options);
|
|
15
|
+
case 'paddle':
|
|
16
|
+
return createPaddleMasker(options);
|
|
14
17
|
case 'qwen-agent-name':
|
|
15
18
|
return await createQwenAgentMasker(options);
|
|
16
19
|
}
|
package/dist/index.d.ts
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
export { MASK_TARGETS, type Masker, type MaskMappingEntry, type MaskRequest, type MaskResult, type MaskTarget, } from './masker.ts';
|
|
2
2
|
export { createMasker, type MaskerOptions } from './create_masker.ts';
|
|
3
|
+
export { createPaddleMasker, type PaddleMaskerOptions } from './paddle_masker.ts';
|
|
3
4
|
export { createTencentMasker, type TencentMaskerOptions } from './tencent_masker.ts';
|
|
4
|
-
export { createTextInMasker,
|
|
5
|
+
export { createTextInMasker, type TextInMaskerOptions, } from './textin_masker.ts';
|
|
6
|
+
export { findSensitiveWordsByRules, type SensitiveWord, type SensitiveWordFinder, } from './sensitive_word_finder.ts';
|
|
5
7
|
export { redactTextByMapping } from './redact_text.ts';
|
|
6
8
|
export { createLlmSensitiveWordFinder, type LlmSensitiveWordFinderOptions, } from './llm_sensitive_word_finder.ts';
|
|
7
9
|
export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, type QwenAgentMasker, type QwenAgentMaskerOptions, type QwenAgentNameMaskInput, } from './qwen_agent_masker.ts';
|
package/dist/index.js
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
export { MASK_TARGETS, } from './masker.js';
|
|
2
2
|
export { createMasker } from './create_masker.js';
|
|
3
|
+
export { createPaddleMasker } from './paddle_masker.js';
|
|
3
4
|
export { createTencentMasker } from './tencent_masker.js';
|
|
4
|
-
export { createTextInMasker,
|
|
5
|
+
export { createTextInMasker, } from './textin_masker.js';
|
|
6
|
+
export { findSensitiveWordsByRules, } from './sensitive_word_finder.js';
|
|
5
7
|
export { redactTextByMapping } from './redact_text.js';
|
|
6
8
|
export { createLlmSensitiveWordFinder, } from './llm_sensitive_word_finder.js';
|
|
7
9
|
export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, } from './qwen_agent_masker.js';
|
|
@@ -6,11 +6,14 @@
|
|
|
6
6
|
* 注意:OCR 全文含隐私,启用它意味着文本出域给该模型服务;
|
|
7
7
|
* 只有该服务被批准作为隐私数据处理方时才可装配。
|
|
8
8
|
*/
|
|
9
|
-
import type { SensitiveWordFinder } from './
|
|
9
|
+
import type { SensitiveWordFinder } from './sensitive_word_finder.ts';
|
|
10
10
|
export type LlmSensitiveWordFinderOptions = {
|
|
11
11
|
readonly baseURL: string;
|
|
12
12
|
readonly apiKey: string;
|
|
13
13
|
readonly model: string;
|
|
14
|
+
/** 输出预算,包含推理 token;默认 65536,调用方须匹配模型支持的上限。 */
|
|
15
|
+
readonly maxTokens?: number;
|
|
16
|
+
/** 单次模型请求超时,默认 300000 ms;外层调用链须给予相应时间。 */
|
|
14
17
|
readonly timeoutMs?: number;
|
|
15
18
|
/** 测试注入缝:替换掉真实的 HTTP 层,不碰网络。 */
|
|
16
19
|
readonly fetch?: typeof fetch;
|
|
@@ -10,7 +10,8 @@ import { z } from 'zod';
|
|
|
10
10
|
import { AppError } from '@onco-foundry/errors';
|
|
11
11
|
import { MASK_TARGETS } from './masker.js';
|
|
12
12
|
import { MaskingError } from './masking_error.js';
|
|
13
|
-
const DEFAULT_TIMEOUT_MS =
|
|
13
|
+
const DEFAULT_TIMEOUT_MS = 300_000;
|
|
14
|
+
const DEFAULT_MAX_TOKENS = 65_536;
|
|
14
15
|
// prompt 是模块内置资产,不开放注入:本模块定位就是医疗文书脱敏,提取口径由模块自己保证。
|
|
15
16
|
const SYSTEM_PROMPT = `你从医疗文书扫描件(病历、检查报告、住院单等)的 OCR 文本中提取隐私信息。只提取以下类别:
|
|
16
17
|
- patient_name:患者姓名(包括「姓名:」标签后的、正文叙述中出现的)
|
|
@@ -67,12 +68,14 @@ export const createLlmSensitiveWordFinder = (options) => {
|
|
|
67
68
|
baseURL: z.url(),
|
|
68
69
|
apiKey: z.string().trim().min(1),
|
|
69
70
|
model: z.string().trim().min(1),
|
|
70
|
-
|
|
71
|
+
maxTokens: z.number().int().positive().max(Number.MAX_SAFE_INTEGER).default(DEFAULT_MAX_TOKENS),
|
|
72
|
+
timeoutMs: z.number().int().positive().max(2_147_483_647).default(DEFAULT_TIMEOUT_MS),
|
|
73
|
+
}).safeParse({ baseURL: options.baseURL, apiKey: options.apiKey, model: options.model, maxTokens: options.maxTokens, timeoutMs: options.timeoutMs });
|
|
71
74
|
if (!credentials.success) {
|
|
72
75
|
throw new AppError('LLM 敏感词判定器配置非法', 500);
|
|
73
76
|
}
|
|
74
77
|
const doFetch = options.fetch ?? fetch;
|
|
75
|
-
const timeoutMs =
|
|
78
|
+
const { maxTokens, timeoutMs } = credentials.data;
|
|
76
79
|
return async (text) => {
|
|
77
80
|
if (text.trim() === '')
|
|
78
81
|
return [];
|
|
@@ -91,7 +94,7 @@ export const createLlmSensitiveWordFinder = (options) => {
|
|
|
91
94
|
messages: [{ role: 'user', content: text }],
|
|
92
95
|
temperature: 0,
|
|
93
96
|
// 推理型模型的思考也吃 max_tokens:给小了会在正文输出前被截断,响应里一个 text 块都没有。
|
|
94
|
-
max_tokens:
|
|
97
|
+
max_tokens: maxTokens,
|
|
95
98
|
stream: false,
|
|
96
99
|
}),
|
|
97
100
|
signal: AbortSignal.timeout(timeoutMs),
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
import type { Masker } from './masker.ts';
|
|
2
|
+
import { type SensitiveWordFinder } from './sensitive_word_finder.ts';
|
|
3
|
+
export type PaddleMaskerOptions = {
|
|
4
|
+
readonly baseURL: string;
|
|
5
|
+
readonly appId: string;
|
|
6
|
+
readonly secretCode: string;
|
|
7
|
+
readonly timeoutMs?: number;
|
|
8
|
+
readonly maxInputBytes?: number;
|
|
9
|
+
readonly requireAllMatches?: boolean;
|
|
10
|
+
readonly fetch?: typeof fetch;
|
|
11
|
+
readonly findSensitiveWords?: SensitiveWordFinder;
|
|
12
|
+
};
|
|
13
|
+
/** 创建使用自建 Paddle OCR token 坐标的医疗图片脱敏器。 */
|
|
14
|
+
export declare const createPaddleMasker: (options: PaddleMaskerOptions) => Masker;
|
|
@@ -0,0 +1,252 @@
|
|
|
1
|
+
import { Buffer } from 'node:buffer';
|
|
2
|
+
import sharp from 'sharp';
|
|
3
|
+
import { z } from 'zod';
|
|
4
|
+
import { AppError } from '@onco-foundry/errors';
|
|
5
|
+
import { MaskingError } from './masking_error.js';
|
|
6
|
+
import { clampQuad, polygonSvg, scaleQuad } from './quad.js';
|
|
7
|
+
import { findSensitiveWordsByRules, } from './sensitive_word_finder.js';
|
|
8
|
+
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
9
|
+
const DEFAULT_MAX_INPUT_BYTES = 30 * 1024 * 1024;
|
|
10
|
+
const MAX_INPUT_PIXELS = 40_000_000;
|
|
11
|
+
const TOKEN_SAFETY_SCALE = 1.3;
|
|
12
|
+
const RULE_TARGETS = ['id_number', 'phone'];
|
|
13
|
+
const TARGET_LABELS = {
|
|
14
|
+
patient_name: '姓名',
|
|
15
|
+
id_number: '证件号',
|
|
16
|
+
phone: '手机号',
|
|
17
|
+
address: '地址',
|
|
18
|
+
doctor_signature: '签字',
|
|
19
|
+
medical_record_number: '病历号',
|
|
20
|
+
};
|
|
21
|
+
const quad8Schema = z.array(z.number().finite()).length(8);
|
|
22
|
+
const tokenSchema = z.object({
|
|
23
|
+
text: z.string(),
|
|
24
|
+
score: z.number().finite().optional(),
|
|
25
|
+
quad: quad8Schema,
|
|
26
|
+
});
|
|
27
|
+
const lineSchema = z.object({
|
|
28
|
+
text: z.string(),
|
|
29
|
+
score: z.number().finite().optional(),
|
|
30
|
+
quad: quad8Schema,
|
|
31
|
+
tokens: z.array(tokenSchema).default([]),
|
|
32
|
+
});
|
|
33
|
+
const responseSchema = z.object({
|
|
34
|
+
code: z.number(),
|
|
35
|
+
message: z.string().default(''),
|
|
36
|
+
engine: z.object({
|
|
37
|
+
name: z.string().min(1),
|
|
38
|
+
runtime: z.string().min(1),
|
|
39
|
+
}).optional(),
|
|
40
|
+
pages: z.array(z.object({
|
|
41
|
+
width: z.number().positive(),
|
|
42
|
+
height: z.number().positive(),
|
|
43
|
+
lines: z.array(lineSchema).default([]),
|
|
44
|
+
})).optional(),
|
|
45
|
+
});
|
|
46
|
+
const toQuad = (position) => [
|
|
47
|
+
[position[0], position[1]],
|
|
48
|
+
[position[2], position[3]],
|
|
49
|
+
[position[4], position[5]],
|
|
50
|
+
[position[6], position[7]],
|
|
51
|
+
];
|
|
52
|
+
const locateWordSpans = (lines, recognizedText, word) => {
|
|
53
|
+
const ranges = [];
|
|
54
|
+
let offset = 0;
|
|
55
|
+
for (const line of lines) {
|
|
56
|
+
ranges.push({ line, start: offset, end: offset + line.text.length });
|
|
57
|
+
offset += line.text.length + 1;
|
|
58
|
+
}
|
|
59
|
+
const occurrences = [];
|
|
60
|
+
let fromIndex = 0;
|
|
61
|
+
while (word.length > 0) {
|
|
62
|
+
const hit = recognizedText.indexOf(word, fromIndex);
|
|
63
|
+
if (hit < 0)
|
|
64
|
+
break;
|
|
65
|
+
fromIndex = hit + word.length;
|
|
66
|
+
const end = hit + word.length;
|
|
67
|
+
occurrences.push(ranges.flatMap((range) => {
|
|
68
|
+
const start = Math.max(hit, range.start);
|
|
69
|
+
const finish = Math.min(end, range.end);
|
|
70
|
+
return start < finish
|
|
71
|
+
? [{ line: range.line, beginIndex: start - range.start, length: finish - start }]
|
|
72
|
+
: [];
|
|
73
|
+
}));
|
|
74
|
+
}
|
|
75
|
+
return occurrences;
|
|
76
|
+
};
|
|
77
|
+
/**
|
|
78
|
+
* 选择与字符区间相交的原生 token。结构不自洽时返回整行框,宁可多盖。
|
|
79
|
+
*/
|
|
80
|
+
const locateSpanQuads = (span) => {
|
|
81
|
+
if (span.line.tokens.map((token) => token.text).join('') !== span.line.text) {
|
|
82
|
+
return [toQuad(span.line.quad)];
|
|
83
|
+
}
|
|
84
|
+
const end = span.beginIndex + span.length;
|
|
85
|
+
let offset = 0;
|
|
86
|
+
const tokens = [];
|
|
87
|
+
for (const token of span.line.tokens) {
|
|
88
|
+
const tokenEnd = offset + token.text.length;
|
|
89
|
+
if (offset < end && tokenEnd > span.beginIndex)
|
|
90
|
+
tokens.push(token);
|
|
91
|
+
offset = tokenEnd;
|
|
92
|
+
}
|
|
93
|
+
return tokens.length > 0
|
|
94
|
+
? tokens.map((token) => toQuad(token.quad))
|
|
95
|
+
: [toQuad(span.line.quad)];
|
|
96
|
+
};
|
|
97
|
+
const scaledToImage = (quad, scaleX, scaleY) => quad.map(([x, y]) => [x * scaleX, y * scaleY]);
|
|
98
|
+
const validateQuad = (quad, width, height) => {
|
|
99
|
+
const xs = quad.map(([x]) => x);
|
|
100
|
+
const ys = quad.map(([, y]) => y);
|
|
101
|
+
const twiceArea = Math.abs(quad.reduce((sum, [x, y], index) => {
|
|
102
|
+
const next = quad[(index + 1) % quad.length];
|
|
103
|
+
return sum + x * next[1] - next[0] * y;
|
|
104
|
+
}, 0));
|
|
105
|
+
if (twiceArea < 1
|
|
106
|
+
|| quad.flat().some((value) => !Number.isFinite(value))
|
|
107
|
+
|| Math.min(...xs) < 0
|
|
108
|
+
|| Math.min(...ys) < 0
|
|
109
|
+
|| Math.max(...xs) > width
|
|
110
|
+
|| Math.max(...ys) > height) {
|
|
111
|
+
throw new MaskingError('mask_coordinates_invalid', 'Paddle 脱敏坐标无效');
|
|
112
|
+
}
|
|
113
|
+
};
|
|
114
|
+
/** 创建使用自建 Paddle OCR token 坐标的医疗图片脱敏器。 */
|
|
115
|
+
export const createPaddleMasker = (options) => {
|
|
116
|
+
const parsed = z.object({
|
|
117
|
+
baseURL: z.url(),
|
|
118
|
+
appId: z.string().trim().min(1),
|
|
119
|
+
secretCode: z.string().trim().min(1),
|
|
120
|
+
}).safeParse(options);
|
|
121
|
+
if (!parsed.success)
|
|
122
|
+
throw new AppError('Paddle 脱敏服务配置非法', 500);
|
|
123
|
+
const baseUrl = options.baseURL.replace(/\/+$/u, '');
|
|
124
|
+
const doFetch = options.fetch ?? fetch;
|
|
125
|
+
const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
|
|
126
|
+
const finder = options.findSensitiveWords ?? findSensitiveWordsByRules;
|
|
127
|
+
return {
|
|
128
|
+
async mask(request) {
|
|
129
|
+
if (request.targets.length === 0) {
|
|
130
|
+
return {
|
|
131
|
+
maskedImageBytes: request.imageBytes,
|
|
132
|
+
mapping: [],
|
|
133
|
+
processorVersion: { engine: 'paddle-token-masker-v1' },
|
|
134
|
+
};
|
|
135
|
+
}
|
|
136
|
+
if (options.findSensitiveWords === undefined) {
|
|
137
|
+
const unsupported = request.targets.filter((target) => !RULE_TARGETS.includes(target));
|
|
138
|
+
if (unsupported.length > 0) {
|
|
139
|
+
throw new AppError(`Paddle 脱敏内置规则只支持 ${RULE_TARGETS.join('、')},`
|
|
140
|
+
+ `不能处理:${unsupported.join('、')}(可注入 findSensitiveWords 扩展)`, 400);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
if (request.imageBytes.byteLength > (options.maxInputBytes ?? DEFAULT_MAX_INPUT_BYTES)) {
|
|
144
|
+
throw new AppError('Paddle 脱敏输入图片超过大小上限', 413);
|
|
145
|
+
}
|
|
146
|
+
const metadata = await sharp(request.imageBytes, { limitInputPixels: MAX_INPUT_PIXELS })
|
|
147
|
+
.metadata()
|
|
148
|
+
.catch(() => {
|
|
149
|
+
throw new MaskingError('image_decode_failed', 'Paddle 脱敏无法解码输入图片', 400);
|
|
150
|
+
});
|
|
151
|
+
if (!metadata.width || !metadata.height) {
|
|
152
|
+
throw new MaskingError('image_decode_failed', 'Paddle 脱敏无法读取图片尺寸', 400);
|
|
153
|
+
}
|
|
154
|
+
const rotation = request.rotationClockwiseDegrees ?? 0;
|
|
155
|
+
let response;
|
|
156
|
+
try {
|
|
157
|
+
response = await doFetch(`${baseUrl}/v1/ocr?rotation_hint=${rotation}`, {
|
|
158
|
+
method: 'POST',
|
|
159
|
+
headers: {
|
|
160
|
+
'x-ti-app-id': options.appId,
|
|
161
|
+
'x-ti-secret-code': options.secretCode,
|
|
162
|
+
'Content-Type': 'application/octet-stream',
|
|
163
|
+
},
|
|
164
|
+
body: Buffer.from(request.imageBytes),
|
|
165
|
+
signal: AbortSignal.timeout(timeoutMs),
|
|
166
|
+
});
|
|
167
|
+
}
|
|
168
|
+
catch (error) {
|
|
169
|
+
throw new MaskingError('text_recognition_request_failed', `Paddle 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
|
|
170
|
+
}
|
|
171
|
+
if (!response.ok) {
|
|
172
|
+
throw new MaskingError('text_recognition_http_failed', `Paddle 脱敏 HTTP ${response.status}`);
|
|
173
|
+
}
|
|
174
|
+
const payload = responseSchema.safeParse(await response.json().catch(() => undefined));
|
|
175
|
+
if (!payload.success) {
|
|
176
|
+
throw new MaskingError('text_recognition_invalid_response', 'Paddle 脱敏响应结构校验失败');
|
|
177
|
+
}
|
|
178
|
+
if (payload.data.code !== 200 || !payload.data.pages) {
|
|
179
|
+
throw new MaskingError('text_recognition_provider_failed', `Paddle 脱敏接口返回业务错误(code=${payload.data.code})`);
|
|
180
|
+
}
|
|
181
|
+
if (payload.data.pages.length !== 1) {
|
|
182
|
+
throw new MaskingError('text_recognition_multiple_pages', 'Paddle 脱敏需要单页图片识别结果');
|
|
183
|
+
}
|
|
184
|
+
const page = payload.data.pages[0];
|
|
185
|
+
const lines = page.lines;
|
|
186
|
+
const recognizedText = lines.map((line) => line.text).join('\n');
|
|
187
|
+
if (options.requireAllMatches && !recognizedText.trim()) {
|
|
188
|
+
throw new MaskingError('text_recognition_empty', 'Paddle 脱敏未识别出文字');
|
|
189
|
+
}
|
|
190
|
+
let detected;
|
|
191
|
+
try {
|
|
192
|
+
detected = await finder(recognizedText);
|
|
193
|
+
}
|
|
194
|
+
catch (error) {
|
|
195
|
+
if (error instanceof MaskingError)
|
|
196
|
+
throw error;
|
|
197
|
+
throw new MaskingError('sensitive_word_failed', '敏感词判定器执行失败');
|
|
198
|
+
}
|
|
199
|
+
if (options.requireAllMatches && detected.some((word) => !word.text || !(word.target in TARGET_LABELS) || !recognizedText.includes(word.text))) {
|
|
200
|
+
throw new MaskingError('sensitive_word_not_locatable', 'Paddle 脱敏判定结果无法定位');
|
|
201
|
+
}
|
|
202
|
+
const scaleX = metadata.width / page.width;
|
|
203
|
+
const scaleY = metadata.height / page.height;
|
|
204
|
+
const quads = [];
|
|
205
|
+
const mapping = [];
|
|
206
|
+
const counters = new Map();
|
|
207
|
+
for (const word of detected.filter((item) => item.text !== '' && request.targets.includes(item.target))) {
|
|
208
|
+
let located = 0;
|
|
209
|
+
for (const spans of locateWordSpans(lines, recognizedText, word.text)) {
|
|
210
|
+
const occurrence = spans.flatMap(locateSpanQuads).map((quad) => scaledToImage(quad, scaleX, scaleY));
|
|
211
|
+
if (occurrence.length === 0)
|
|
212
|
+
continue;
|
|
213
|
+
for (const quad of occurrence) {
|
|
214
|
+
if (options.requireAllMatches)
|
|
215
|
+
validateQuad(quad, metadata.width, metadata.height);
|
|
216
|
+
quads.push(clampQuad(scaleQuad(quad, TOKEN_SAFETY_SCALE), metadata.width, metadata.height));
|
|
217
|
+
}
|
|
218
|
+
located++;
|
|
219
|
+
const count = (counters.get(word.target) ?? 0) + 1;
|
|
220
|
+
counters.set(word.target, count);
|
|
221
|
+
mapping.push({
|
|
222
|
+
placeholder: `[${TARGET_LABELS[word.target]}${count}]`,
|
|
223
|
+
target: word.target,
|
|
224
|
+
originalText: word.text,
|
|
225
|
+
});
|
|
226
|
+
}
|
|
227
|
+
if (options.requireAllMatches && located === 0) {
|
|
228
|
+
throw new MaskingError('sensitive_word_not_locatable', 'Paddle 脱敏敏感词无法定位');
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
let maskedImageBytes;
|
|
232
|
+
try {
|
|
233
|
+
maskedImageBytes = await sharp(request.imageBytes, { limitInputPixels: MAX_INPUT_PIXELS })
|
|
234
|
+
.composite([{ input: polygonSvg(metadata.width, metadata.height, quads), blend: 'over' }])
|
|
235
|
+
.jpeg({ quality: 94 })
|
|
236
|
+
.toBuffer();
|
|
237
|
+
}
|
|
238
|
+
catch {
|
|
239
|
+
throw new MaskingError('mask_render_failed', 'Paddle 脱敏图片遮盖失败');
|
|
240
|
+
}
|
|
241
|
+
const engine = payload.data.engine;
|
|
242
|
+
return {
|
|
243
|
+
maskedImageBytes: new Uint8Array(maskedImageBytes),
|
|
244
|
+
mapping,
|
|
245
|
+
processorVersion: {
|
|
246
|
+
engine: 'paddle-token-masker-v1',
|
|
247
|
+
...(engine ? { model: `${engine.name}@${engine.runtime}` } : {}),
|
|
248
|
+
},
|
|
249
|
+
};
|
|
250
|
+
},
|
|
251
|
+
};
|
|
252
|
+
};
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { MaskTarget } from './masker.ts';
|
|
2
|
+
/** 一个待遮盖的敏感词及其类别。判定哪些词敏感是业务判断,由调用方注入。 */
|
|
3
|
+
export type SensitiveWord = {
|
|
4
|
+
readonly text: string;
|
|
5
|
+
readonly target: MaskTarget;
|
|
6
|
+
};
|
|
7
|
+
/** 吃 OCR 原文全文,吐出必须逐字引用原文的敏感词。 */
|
|
8
|
+
export type SensitiveWordFinder = (text: string) => readonly SensitiveWord[] | Promise<readonly SensitiveWord[]>;
|
|
9
|
+
/** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
|
|
10
|
+
export declare const findSensitiveWordsByRules: (text: string) => readonly SensitiveWord[];
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
|
|
2
|
+
export const findSensitiveWordsByRules = (text) => {
|
|
3
|
+
const words = [];
|
|
4
|
+
for (const match of text.matchAll(/(?<!\d)\d{17}[\dXx](?!\d)/gu)) {
|
|
5
|
+
words.push({ text: match[0], target: 'id_number' });
|
|
6
|
+
}
|
|
7
|
+
for (const match of text.matchAll(/(?<!\d)1[3-9]\d{9}(?!\d)/gu)) {
|
|
8
|
+
words.push({ text: match[0], target: 'phone' });
|
|
9
|
+
}
|
|
10
|
+
return words;
|
|
11
|
+
};
|
package/dist/textin_masker.d.ts
CHANGED
|
@@ -9,14 +9,8 @@
|
|
|
9
9
|
* 下游按 mapping 替换出脱敏文本后可省掉第二次 OCR。注意本引擎接触的是原文图片,
|
|
10
10
|
* 装配它意味着「原文出域给 TextIn」这一策略决定已经做出。
|
|
11
11
|
*/
|
|
12
|
-
import type { Masker
|
|
13
|
-
|
|
14
|
-
export type SensitiveWord = {
|
|
15
|
-
readonly text: string;
|
|
16
|
-
readonly target: MaskTarget;
|
|
17
|
-
};
|
|
18
|
-
/** 敏感词判定器:吃 OCR 原文全文,吐出要遮盖的词。可同步可异步。 */
|
|
19
|
-
export type SensitiveWordFinder = (text: string) => readonly SensitiveWord[] | Promise<readonly SensitiveWord[]>;
|
|
12
|
+
import type { Masker } from './masker.ts';
|
|
13
|
+
import { type SensitiveWordFinder } from './sensitive_word_finder.ts';
|
|
20
14
|
export type TextInMaskerOptions = {
|
|
21
15
|
readonly appId: string;
|
|
22
16
|
readonly secretCode: string;
|
|
@@ -33,8 +27,6 @@ export type TextInMaskerOptions = {
|
|
|
33
27
|
*/
|
|
34
28
|
readonly findSensitiveWords?: SensitiveWordFinder;
|
|
35
29
|
};
|
|
36
|
-
/** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
|
|
37
|
-
export declare const findSensitiveWordsByRules: (text: string) => readonly SensitiveWord[];
|
|
38
30
|
/**
|
|
39
31
|
* 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
|
|
40
32
|
* 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
|
package/dist/textin_masker.js
CHANGED
|
@@ -15,6 +15,7 @@ import { z } from 'zod';
|
|
|
15
15
|
import { AppError } from '@onco-foundry/errors';
|
|
16
16
|
import { MaskingError } from './masking_error.js';
|
|
17
17
|
import { clampQuad, polygonSvg, scaleQuad } from './quad.js';
|
|
18
|
+
import { findSensitiveWordsByRules, } from './sensitive_word_finder.js';
|
|
18
19
|
const DEFAULT_TEXTIN_BASE_URL = 'https://api.textin.com';
|
|
19
20
|
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
20
21
|
const DEFAULT_MAX_INPUT_BYTES = 30 * 1024 * 1024;
|
|
@@ -32,17 +33,6 @@ const TARGET_LABELS = {
|
|
|
32
33
|
doctor_signature: '签字',
|
|
33
34
|
medical_record_number: '病历号',
|
|
34
35
|
};
|
|
35
|
-
/** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
|
|
36
|
-
export const findSensitiveWordsByRules = (text) => {
|
|
37
|
-
const words = [];
|
|
38
|
-
for (const match of text.matchAll(/(?<!\d)\d{17}[\dXx](?!\d)/gu)) {
|
|
39
|
-
words.push({ text: match[0], target: 'id_number' });
|
|
40
|
-
}
|
|
41
|
-
for (const match of text.matchAll(/(?<!\d)1[3-9]\d{9}(?!\d)/gu)) {
|
|
42
|
-
words.push({ text: match[0], target: 'phone' });
|
|
43
|
-
}
|
|
44
|
-
return words;
|
|
45
|
-
};
|
|
46
36
|
/** TextIn 行坐标:四边形 4 顶点 8 个数,顺序左上、右上、右下、左下。 */
|
|
47
37
|
const quad8Schema = z.array(z.number()).length(8);
|
|
48
38
|
const lineSchema = z.object({
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@onco-foundry/mask-port",
|
|
3
|
-
"version": "0.1
|
|
3
|
+
"version": "0.2.1",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"files": [
|
|
6
6
|
"dist",
|
|
@@ -17,9 +17,9 @@
|
|
|
17
17
|
"sharp": "^0.35.3",
|
|
18
18
|
"zod": "^4.4.3",
|
|
19
19
|
"@onco-foundry/capability-registry": "0.1.1",
|
|
20
|
-
"@onco-foundry/errors": "0.1.1",
|
|
21
20
|
"@onco-foundry/resource-versioning": "0.3.0",
|
|
22
|
-
"@onco-foundry/
|
|
21
|
+
"@onco-foundry/errors": "0.1.1",
|
|
22
|
+
"@onco-foundry/trace-port": "0.3.3"
|
|
23
23
|
},
|
|
24
24
|
"publishConfig": {
|
|
25
25
|
"access": "public"
|