@onco-foundry/mask-port 0.1.5 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md ADDED
@@ -0,0 +1,10 @@
1
+ # mask-port
2
+
3
+ `createLlmSensitiveWordFinder` 使用 Anthropic 兼容 Messages API 提取敏感词,供图片脱敏器定位。
4
+
5
+ ## 模型请求预算
6
+
7
+ - `maxTokens`:正整数,默认 **65536**(此前固定为 16384)。推理型模型的思考可能占用同一预算;实际消耗由模型响应决定,设置上限不等于每次用满。
8
+ - `timeoutMs`:正整数,默认 **300000 ms**(此前为 120000 ms)。调用方包裹的 OCR 请求也需提供足够的总超时;内层增加超时不会延长外层截止时间。
9
+
10
+ 按供应商和所选模型支持的输出上限设置 `maxTokens`,支持较小预算的模型需显式覆盖默认值。无效配置在创建判定器时抛错。配置调整不改变输出格式校验,也不会将缺失输出视为脱敏成功。
@@ -2,6 +2,7 @@ import type { Masker } from './masker.ts';
2
2
  import { type QwenAgentMaskerOptions } from './qwen_agent_masker.ts';
3
3
  import { type TencentMaskerOptions } from './tencent_masker.ts';
4
4
  import { type TextInMaskerOptions } from './textin_masker.ts';
5
+ import { type PaddleMaskerOptions } from './paddle_masker.ts';
5
6
  /**
6
7
  * 脱敏器的装配选项:kind 判别用哪个实现。
7
8
  * 加新供应商时在这里加一个 union 成员,调用方只改 kind。
@@ -13,6 +14,8 @@ export type MaskerOptions = {
13
14
  } & TencentMaskerOptions) | ({
14
15
  readonly kind: 'textin';
15
16
  } & TextInMaskerOptions) | ({
17
+ readonly kind: 'paddle';
18
+ } & PaddleMaskerOptions) | ({
16
19
  readonly kind: 'qwen-agent-name';
17
20
  } & QwenAgentMaskerOptions);
18
21
  /** 脱敏端口的唯一工厂:换供应商只改 kind,调用方不 import 具体实现。 */
@@ -2,6 +2,7 @@ import { createFakeMasker } from './fake_masker.js';
2
2
  import { createQwenAgentMasker, } from './qwen_agent_masker.js';
3
3
  import { createTencentMasker } from './tencent_masker.js';
4
4
  import { createTextInMasker } from './textin_masker.js';
5
+ import { createPaddleMasker } from './paddle_masker.js';
5
6
  /** 脱敏端口的唯一工厂:换供应商只改 kind,调用方不 import 具体实现。 */
6
7
  export const createMasker = async (options) => {
7
8
  switch (options.kind) {
@@ -11,6 +12,8 @@ export const createMasker = async (options) => {
11
12
  return createTencentMasker(options);
12
13
  case 'textin':
13
14
  return createTextInMasker(options);
15
+ case 'paddle':
16
+ return createPaddleMasker(options);
14
17
  case 'qwen-agent-name':
15
18
  return await createQwenAgentMasker(options);
16
19
  }
package/dist/index.d.ts CHANGED
@@ -1,8 +1,11 @@
1
1
  export { MASK_TARGETS, type Masker, type MaskMappingEntry, type MaskRequest, type MaskResult, type MaskTarget, } from './masker.ts';
2
2
  export { createMasker, type MaskerOptions } from './create_masker.ts';
3
+ export { createPaddleMasker, type PaddleMaskerOptions } from './paddle_masker.ts';
3
4
  export { createTencentMasker, type TencentMaskerOptions } from './tencent_masker.ts';
4
- export { createTextInMasker, findSensitiveWordsByRules, type SensitiveWord, type SensitiveWordFinder, type TextInMaskerOptions, } from './textin_masker.ts';
5
+ export { createTextInMasker, type TextInMaskerOptions, } from './textin_masker.ts';
6
+ export { findSensitiveWordsByRules, type SensitiveWord, type SensitiveWordFinder, } from './sensitive_word_finder.ts';
5
7
  export { redactTextByMapping } from './redact_text.ts';
6
8
  export { createLlmSensitiveWordFinder, type LlmSensitiveWordFinderOptions, } from './llm_sensitive_word_finder.ts';
7
9
  export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, type QwenAgentMasker, type QwenAgentMaskerOptions, type QwenAgentNameMaskInput, } from './qwen_agent_masker.ts';
8
10
  export { createFakeMasker } from './fake_masker.ts';
11
+ export { MaskingError, type MaskingFailureReason } from './masking_error.ts';
package/dist/index.js CHANGED
@@ -1,8 +1,11 @@
1
1
  export { MASK_TARGETS, } from './masker.js';
2
2
  export { createMasker } from './create_masker.js';
3
+ export { createPaddleMasker } from './paddle_masker.js';
3
4
  export { createTencentMasker } from './tencent_masker.js';
4
- export { createTextInMasker, findSensitiveWordsByRules, } from './textin_masker.js';
5
+ export { createTextInMasker, } from './textin_masker.js';
6
+ export { findSensitiveWordsByRules, } from './sensitive_word_finder.js';
5
7
  export { redactTextByMapping } from './redact_text.js';
6
8
  export { createLlmSensitiveWordFinder, } from './llm_sensitive_word_finder.js';
7
9
  export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, } from './qwen_agent_masker.js';
8
10
  export { createFakeMasker } from './fake_masker.js';
11
+ export { MaskingError } from './masking_error.js';
@@ -6,11 +6,14 @@
6
6
  * 注意:OCR 全文含隐私,启用它意味着文本出域给该模型服务;
7
7
  * 只有该服务被批准作为隐私数据处理方时才可装配。
8
8
  */
9
- import type { SensitiveWordFinder } from './textin_masker.ts';
9
+ import type { SensitiveWordFinder } from './sensitive_word_finder.ts';
10
10
  export type LlmSensitiveWordFinderOptions = {
11
11
  readonly baseURL: string;
12
12
  readonly apiKey: string;
13
13
  readonly model: string;
14
+ /** 输出预算,包含推理 token;默认 65536,调用方须匹配模型支持的上限。 */
15
+ readonly maxTokens?: number;
16
+ /** 单次模型请求超时,默认 300000 ms;外层调用链须给予相应时间。 */
14
17
  readonly timeoutMs?: number;
15
18
  /** 测试注入缝:替换掉真实的 HTTP 层,不碰网络。 */
16
19
  readonly fetch?: typeof fetch;
@@ -9,7 +9,9 @@
9
9
  import { z } from 'zod';
10
10
  import { AppError } from '@onco-foundry/errors';
11
11
  import { MASK_TARGETS } from './masker.js';
12
- const DEFAULT_TIMEOUT_MS = 120_000;
12
+ import { MaskingError } from './masking_error.js';
13
+ const DEFAULT_TIMEOUT_MS = 300_000;
14
+ const DEFAULT_MAX_TOKENS = 65_536;
13
15
  // prompt 是模块内置资产,不开放注入:本模块定位就是医疗文书脱敏,提取口径由模块自己保证。
14
16
  const SYSTEM_PROMPT = `你从医疗文书扫描件(病历、检查报告、住院单等)的 OCR 文本中提取隐私信息。只提取以下类别:
15
17
  - patient_name:患者姓名(包括「姓名:」标签后的、正文叙述中出现的)
@@ -46,13 +48,13 @@ const extractJsonArray = (content) => {
46
48
  const start = cleaned.indexOf('[');
47
49
  const end = cleaned.lastIndexOf(']');
48
50
  if (start < 0 || end <= start) {
49
- throw new AppError('LLM 敏感词提取返回的结构不是 JSON 数组', 502);
51
+ throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构不是 JSON 数组');
50
52
  }
51
53
  try {
52
54
  return JSON.parse(cleaned.slice(start, end + 1));
53
55
  }
54
56
  catch {
55
- throw new AppError('LLM 敏感词提取返回的 JSON 无法解析', 502);
57
+ throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的 JSON 无法解析');
56
58
  }
57
59
  };
58
60
  /**
@@ -66,12 +68,14 @@ export const createLlmSensitiveWordFinder = (options) => {
66
68
  baseURL: z.url(),
67
69
  apiKey: z.string().trim().min(1),
68
70
  model: z.string().trim().min(1),
69
- }).safeParse({ baseURL: options.baseURL, apiKey: options.apiKey, model: options.model });
71
+ maxTokens: z.number().int().positive().max(Number.MAX_SAFE_INTEGER).default(DEFAULT_MAX_TOKENS),
72
+ timeoutMs: z.number().int().positive().max(2_147_483_647).default(DEFAULT_TIMEOUT_MS),
73
+ }).safeParse({ baseURL: options.baseURL, apiKey: options.apiKey, model: options.model, maxTokens: options.maxTokens, timeoutMs: options.timeoutMs });
70
74
  if (!credentials.success) {
71
75
  throw new AppError('LLM 敏感词判定器配置非法', 500);
72
76
  }
73
77
  const doFetch = options.fetch ?? fetch;
74
- const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
78
+ const { maxTokens, timeoutMs } = credentials.data;
75
79
  return async (text) => {
76
80
  if (text.trim() === '')
77
81
  return [];
@@ -90,32 +94,32 @@ export const createLlmSensitiveWordFinder = (options) => {
90
94
  messages: [{ role: 'user', content: text }],
91
95
  temperature: 0,
92
96
  // 推理型模型的思考也吃 max_tokens:给小了会在正文输出前被截断,响应里一个 text 块都没有。
93
- max_tokens: 16384,
97
+ max_tokens: maxTokens,
94
98
  stream: false,
95
99
  }),
96
100
  signal: AbortSignal.timeout(timeoutMs),
97
101
  });
98
102
  }
99
103
  catch (error) {
100
- throw new AppError(`LLM 敏感词提取请求未完成:${error instanceof Error ? error.name : '未知错误'}`, 502);
104
+ throw new MaskingError('sensitive_word_request_failed', `LLM 敏感词提取请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
101
105
  }
102
106
  if (!response.ok) {
103
- throw new AppError(`LLM 敏感词提取 HTTP ${response.status}`, 502);
107
+ throw new MaskingError('sensitive_word_http_failed', `LLM 敏感词提取 HTTP ${response.status}`);
104
108
  }
105
109
  const payload = modelResponseSchema.safeParse(await response.json().catch(() => undefined));
106
110
  if (!payload.success) {
107
- throw new AppError('LLM 敏感词提取响应缺少模型输出', 502);
111
+ throw new MaskingError('sensitive_word_invalid_response', 'LLM 敏感词提取响应缺少模型输出');
108
112
  }
109
113
  // 带推理的模型会先回 thinking 块,取第一个 text 块。
110
114
  const textBlock = payload.data.content.find((block) => block.type === 'text' && block.text);
111
115
  if (textBlock?.text === undefined) {
112
116
  const blockTypes = payload.data.content.map((block) => block.type).join('、');
113
- throw new AppError(`LLM 敏感词提取响应缺少文本输出(stop_reason=${payload.data.stop_reason ?? '未知'},`
114
- + `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token 上限)`, 502);
117
+ throw new MaskingError('sensitive_word_output_missing', `LLM 敏感词提取响应缺少文本输出(stop_reason=${payload.data.stop_reason ?? '未知'},`
118
+ + `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token 上限)`);
115
119
  }
116
120
  const parsed = extractedSchema.safeParse(extractJsonArray(textBlock.text));
117
121
  if (!parsed.success) {
118
- throw new AppError('LLM 敏感词提取返回的结构校验失败', 502);
122
+ throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构校验失败');
119
123
  }
120
124
  const seen = new Set();
121
125
  const words = [];
@@ -0,0 +1,7 @@
1
+ import { AppError } from '@onco-foundry/errors';
2
+ /** Safe machine-readable categories. Values never contain source text or provider response bodies. */
3
+ export type MaskingFailureReason = 'image_decode_failed' | 'text_recognition_request_failed' | 'text_recognition_http_failed' | 'text_recognition_invalid_response' | 'text_recognition_provider_failed' | 'text_recognition_multiple_pages' | 'text_recognition_empty' | 'sensitive_word_request_failed' | 'sensitive_word_http_failed' | 'sensitive_word_invalid_response' | 'sensitive_word_invalid_output' | 'sensitive_word_output_missing' | 'sensitive_word_failed' | 'sensitive_word_not_locatable' | 'mask_coordinates_missing' | 'mask_coordinates_invalid' | 'mask_render_failed';
4
+ export declare class MaskingError extends AppError {
5
+ readonly reason: MaskingFailureReason;
6
+ constructor(reason: MaskingFailureReason, message: string, statusCode?: number);
7
+ }
@@ -0,0 +1,9 @@
1
+ import { AppError } from '@onco-foundry/errors';
2
+ export class MaskingError extends AppError {
3
+ reason;
4
+ constructor(reason, message, statusCode = 502) {
5
+ super(message, statusCode);
6
+ this.name = 'MaskingError';
7
+ this.reason = reason;
8
+ }
9
+ }
@@ -0,0 +1,14 @@
1
+ import type { Masker } from './masker.ts';
2
+ import { type SensitiveWordFinder } from './sensitive_word_finder.ts';
3
+ export type PaddleMaskerOptions = {
4
+ readonly baseURL: string;
5
+ readonly appId: string;
6
+ readonly secretCode: string;
7
+ readonly timeoutMs?: number;
8
+ readonly maxInputBytes?: number;
9
+ readonly requireAllMatches?: boolean;
10
+ readonly fetch?: typeof fetch;
11
+ readonly findSensitiveWords?: SensitiveWordFinder;
12
+ };
13
+ /** 创建使用自建 Paddle OCR token 坐标的医疗图片脱敏器。 */
14
+ export declare const createPaddleMasker: (options: PaddleMaskerOptions) => Masker;
@@ -0,0 +1,252 @@
1
+ import { Buffer } from 'node:buffer';
2
+ import sharp from 'sharp';
3
+ import { z } from 'zod';
4
+ import { AppError } from '@onco-foundry/errors';
5
+ import { MaskingError } from './masking_error.js';
6
+ import { clampQuad, polygonSvg, scaleQuad } from './quad.js';
7
+ import { findSensitiveWordsByRules, } from './sensitive_word_finder.js';
8
+ const DEFAULT_TIMEOUT_MS = 120_000;
9
+ const DEFAULT_MAX_INPUT_BYTES = 30 * 1024 * 1024;
10
+ const MAX_INPUT_PIXELS = 40_000_000;
11
+ const TOKEN_SAFETY_SCALE = 1.3;
12
+ const RULE_TARGETS = ['id_number', 'phone'];
13
+ const TARGET_LABELS = {
14
+ patient_name: '姓名',
15
+ id_number: '证件号',
16
+ phone: '手机号',
17
+ address: '地址',
18
+ doctor_signature: '签字',
19
+ medical_record_number: '病历号',
20
+ };
21
+ const quad8Schema = z.array(z.number().finite()).length(8);
22
+ const tokenSchema = z.object({
23
+ text: z.string(),
24
+ score: z.number().finite().optional(),
25
+ quad: quad8Schema,
26
+ });
27
+ const lineSchema = z.object({
28
+ text: z.string(),
29
+ score: z.number().finite().optional(),
30
+ quad: quad8Schema,
31
+ tokens: z.array(tokenSchema).default([]),
32
+ });
33
+ const responseSchema = z.object({
34
+ code: z.number(),
35
+ message: z.string().default(''),
36
+ engine: z.object({
37
+ name: z.string().min(1),
38
+ runtime: z.string().min(1),
39
+ }).optional(),
40
+ pages: z.array(z.object({
41
+ width: z.number().positive(),
42
+ height: z.number().positive(),
43
+ lines: z.array(lineSchema).default([]),
44
+ })).optional(),
45
+ });
46
+ const toQuad = (position) => [
47
+ [position[0], position[1]],
48
+ [position[2], position[3]],
49
+ [position[4], position[5]],
50
+ [position[6], position[7]],
51
+ ];
52
+ const locateWordSpans = (lines, recognizedText, word) => {
53
+ const ranges = [];
54
+ let offset = 0;
55
+ for (const line of lines) {
56
+ ranges.push({ line, start: offset, end: offset + line.text.length });
57
+ offset += line.text.length + 1;
58
+ }
59
+ const occurrences = [];
60
+ let fromIndex = 0;
61
+ while (word.length > 0) {
62
+ const hit = recognizedText.indexOf(word, fromIndex);
63
+ if (hit < 0)
64
+ break;
65
+ fromIndex = hit + word.length;
66
+ const end = hit + word.length;
67
+ occurrences.push(ranges.flatMap((range) => {
68
+ const start = Math.max(hit, range.start);
69
+ const finish = Math.min(end, range.end);
70
+ return start < finish
71
+ ? [{ line: range.line, beginIndex: start - range.start, length: finish - start }]
72
+ : [];
73
+ }));
74
+ }
75
+ return occurrences;
76
+ };
77
+ /**
78
+ * 选择与字符区间相交的原生 token。结构不自洽时返回整行框,宁可多盖。
79
+ */
80
+ const locateSpanQuads = (span) => {
81
+ if (span.line.tokens.map((token) => token.text).join('') !== span.line.text) {
82
+ return [toQuad(span.line.quad)];
83
+ }
84
+ const end = span.beginIndex + span.length;
85
+ let offset = 0;
86
+ const tokens = [];
87
+ for (const token of span.line.tokens) {
88
+ const tokenEnd = offset + token.text.length;
89
+ if (offset < end && tokenEnd > span.beginIndex)
90
+ tokens.push(token);
91
+ offset = tokenEnd;
92
+ }
93
+ return tokens.length > 0
94
+ ? tokens.map((token) => toQuad(token.quad))
95
+ : [toQuad(span.line.quad)];
96
+ };
97
+ const scaledToImage = (quad, scaleX, scaleY) => quad.map(([x, y]) => [x * scaleX, y * scaleY]);
98
+ const validateQuad = (quad, width, height) => {
99
+ const xs = quad.map(([x]) => x);
100
+ const ys = quad.map(([, y]) => y);
101
+ const twiceArea = Math.abs(quad.reduce((sum, [x, y], index) => {
102
+ const next = quad[(index + 1) % quad.length];
103
+ return sum + x * next[1] - next[0] * y;
104
+ }, 0));
105
+ if (twiceArea < 1
106
+ || quad.flat().some((value) => !Number.isFinite(value))
107
+ || Math.min(...xs) < 0
108
+ || Math.min(...ys) < 0
109
+ || Math.max(...xs) > width
110
+ || Math.max(...ys) > height) {
111
+ throw new MaskingError('mask_coordinates_invalid', 'Paddle 脱敏坐标无效');
112
+ }
113
+ };
114
+ /** 创建使用自建 Paddle OCR token 坐标的医疗图片脱敏器。 */
115
+ export const createPaddleMasker = (options) => {
116
+ const parsed = z.object({
117
+ baseURL: z.url(),
118
+ appId: z.string().trim().min(1),
119
+ secretCode: z.string().trim().min(1),
120
+ }).safeParse(options);
121
+ if (!parsed.success)
122
+ throw new AppError('Paddle 脱敏服务配置非法', 500);
123
+ const baseUrl = options.baseURL.replace(/\/+$/u, '');
124
+ const doFetch = options.fetch ?? fetch;
125
+ const timeoutMs = options.timeoutMs ?? DEFAULT_TIMEOUT_MS;
126
+ const finder = options.findSensitiveWords ?? findSensitiveWordsByRules;
127
+ return {
128
+ async mask(request) {
129
+ if (request.targets.length === 0) {
130
+ return {
131
+ maskedImageBytes: request.imageBytes,
132
+ mapping: [],
133
+ processorVersion: { engine: 'paddle-token-masker-v1' },
134
+ };
135
+ }
136
+ if (options.findSensitiveWords === undefined) {
137
+ const unsupported = request.targets.filter((target) => !RULE_TARGETS.includes(target));
138
+ if (unsupported.length > 0) {
139
+ throw new AppError(`Paddle 脱敏内置规则只支持 ${RULE_TARGETS.join('、')},`
140
+ + `不能处理:${unsupported.join('、')}(可注入 findSensitiveWords 扩展)`, 400);
141
+ }
142
+ }
143
+ if (request.imageBytes.byteLength > (options.maxInputBytes ?? DEFAULT_MAX_INPUT_BYTES)) {
144
+ throw new AppError('Paddle 脱敏输入图片超过大小上限', 413);
145
+ }
146
+ const metadata = await sharp(request.imageBytes, { limitInputPixels: MAX_INPUT_PIXELS })
147
+ .metadata()
148
+ .catch(() => {
149
+ throw new MaskingError('image_decode_failed', 'Paddle 脱敏无法解码输入图片', 400);
150
+ });
151
+ if (!metadata.width || !metadata.height) {
152
+ throw new MaskingError('image_decode_failed', 'Paddle 脱敏无法读取图片尺寸', 400);
153
+ }
154
+ const rotation = request.rotationClockwiseDegrees ?? 0;
155
+ let response;
156
+ try {
157
+ response = await doFetch(`${baseUrl}/v1/ocr?rotation_hint=${rotation}`, {
158
+ method: 'POST',
159
+ headers: {
160
+ 'x-ti-app-id': options.appId,
161
+ 'x-ti-secret-code': options.secretCode,
162
+ 'Content-Type': 'application/octet-stream',
163
+ },
164
+ body: Buffer.from(request.imageBytes),
165
+ signal: AbortSignal.timeout(timeoutMs),
166
+ });
167
+ }
168
+ catch (error) {
169
+ throw new MaskingError('text_recognition_request_failed', `Paddle 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
170
+ }
171
+ if (!response.ok) {
172
+ throw new MaskingError('text_recognition_http_failed', `Paddle 脱敏 HTTP ${response.status}`);
173
+ }
174
+ const payload = responseSchema.safeParse(await response.json().catch(() => undefined));
175
+ if (!payload.success) {
176
+ throw new MaskingError('text_recognition_invalid_response', 'Paddle 脱敏响应结构校验失败');
177
+ }
178
+ if (payload.data.code !== 200 || !payload.data.pages) {
179
+ throw new MaskingError('text_recognition_provider_failed', `Paddle 脱敏接口返回业务错误(code=${payload.data.code})`);
180
+ }
181
+ if (payload.data.pages.length !== 1) {
182
+ throw new MaskingError('text_recognition_multiple_pages', 'Paddle 脱敏需要单页图片识别结果');
183
+ }
184
+ const page = payload.data.pages[0];
185
+ const lines = page.lines;
186
+ const recognizedText = lines.map((line) => line.text).join('\n');
187
+ if (options.requireAllMatches && !recognizedText.trim()) {
188
+ throw new MaskingError('text_recognition_empty', 'Paddle 脱敏未识别出文字');
189
+ }
190
+ let detected;
191
+ try {
192
+ detected = await finder(recognizedText);
193
+ }
194
+ catch (error) {
195
+ if (error instanceof MaskingError)
196
+ throw error;
197
+ throw new MaskingError('sensitive_word_failed', '敏感词判定器执行失败');
198
+ }
199
+ if (options.requireAllMatches && detected.some((word) => !word.text || !(word.target in TARGET_LABELS) || !recognizedText.includes(word.text))) {
200
+ throw new MaskingError('sensitive_word_not_locatable', 'Paddle 脱敏判定结果无法定位');
201
+ }
202
+ const scaleX = metadata.width / page.width;
203
+ const scaleY = metadata.height / page.height;
204
+ const quads = [];
205
+ const mapping = [];
206
+ const counters = new Map();
207
+ for (const word of detected.filter((item) => item.text !== '' && request.targets.includes(item.target))) {
208
+ let located = 0;
209
+ for (const spans of locateWordSpans(lines, recognizedText, word.text)) {
210
+ const occurrence = spans.flatMap(locateSpanQuads).map((quad) => scaledToImage(quad, scaleX, scaleY));
211
+ if (occurrence.length === 0)
212
+ continue;
213
+ for (const quad of occurrence) {
214
+ if (options.requireAllMatches)
215
+ validateQuad(quad, metadata.width, metadata.height);
216
+ quads.push(clampQuad(scaleQuad(quad, TOKEN_SAFETY_SCALE), metadata.width, metadata.height));
217
+ }
218
+ located++;
219
+ const count = (counters.get(word.target) ?? 0) + 1;
220
+ counters.set(word.target, count);
221
+ mapping.push({
222
+ placeholder: `[${TARGET_LABELS[word.target]}${count}]`,
223
+ target: word.target,
224
+ originalText: word.text,
225
+ });
226
+ }
227
+ if (options.requireAllMatches && located === 0) {
228
+ throw new MaskingError('sensitive_word_not_locatable', 'Paddle 脱敏敏感词无法定位');
229
+ }
230
+ }
231
+ let maskedImageBytes;
232
+ try {
233
+ maskedImageBytes = await sharp(request.imageBytes, { limitInputPixels: MAX_INPUT_PIXELS })
234
+ .composite([{ input: polygonSvg(metadata.width, metadata.height, quads), blend: 'over' }])
235
+ .jpeg({ quality: 94 })
236
+ .toBuffer();
237
+ }
238
+ catch {
239
+ throw new MaskingError('mask_render_failed', 'Paddle 脱敏图片遮盖失败');
240
+ }
241
+ const engine = payload.data.engine;
242
+ return {
243
+ maskedImageBytes: new Uint8Array(maskedImageBytes),
244
+ mapping,
245
+ processorVersion: {
246
+ engine: 'paddle-token-masker-v1',
247
+ ...(engine ? { model: `${engine.name}@${engine.runtime}` } : {}),
248
+ },
249
+ };
250
+ },
251
+ };
252
+ };
@@ -0,0 +1,10 @@
1
+ import type { MaskTarget } from './masker.ts';
2
+ /** 一个待遮盖的敏感词及其类别。判定哪些词敏感是业务判断,由调用方注入。 */
3
+ export type SensitiveWord = {
4
+ readonly text: string;
5
+ readonly target: MaskTarget;
6
+ };
7
+ /** 吃 OCR 原文全文,吐出必须逐字引用原文的敏感词。 */
8
+ export type SensitiveWordFinder = (text: string) => readonly SensitiveWord[] | Promise<readonly SensitiveWord[]>;
9
+ /** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
10
+ export declare const findSensitiveWordsByRules: (text: string) => readonly SensitiveWord[];
@@ -0,0 +1,11 @@
1
+ /** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
2
+ export const findSensitiveWordsByRules = (text) => {
3
+ const words = [];
4
+ for (const match of text.matchAll(/(?<!\d)\d{17}[\dXx](?!\d)/gu)) {
5
+ words.push({ text: match[0], target: 'id_number' });
6
+ }
7
+ for (const match of text.matchAll(/(?<!\d)1[3-9]\d{9}(?!\d)/gu)) {
8
+ words.push({ text: match[0], target: 'phone' });
9
+ }
10
+ return words;
11
+ };
@@ -9,14 +9,8 @@
9
9
  * 下游按 mapping 替换出脱敏文本后可省掉第二次 OCR。注意本引擎接触的是原文图片,
10
10
  * 装配它意味着「原文出域给 TextIn」这一策略决定已经做出。
11
11
  */
12
- import type { Masker, MaskTarget } from './masker.ts';
13
- /** 一个待遮盖的敏感词及其类别。判定「哪些词敏感」是业务判断,由调用方注入。 */
14
- export type SensitiveWord = {
15
- readonly text: string;
16
- readonly target: MaskTarget;
17
- };
18
- /** 敏感词判定器:吃 OCR 原文全文,吐出要遮盖的词。可同步可异步。 */
19
- export type SensitiveWordFinder = (text: string) => readonly SensitiveWord[] | Promise<readonly SensitiveWord[]>;
12
+ import type { Masker } from './masker.ts';
13
+ import { type SensitiveWordFinder } from './sensitive_word_finder.ts';
20
14
  export type TextInMaskerOptions = {
21
15
  readonly appId: string;
22
16
  readonly secretCode: string;
@@ -33,8 +27,6 @@ export type TextInMaskerOptions = {
33
27
  */
34
28
  readonly findSensitiveWords?: SensitiveWordFinder;
35
29
  };
36
- /** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
37
- export declare const findSensitiveWordsByRules: (text: string) => readonly SensitiveWord[];
38
30
  /**
39
31
  * 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
40
32
  * 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
@@ -13,7 +13,9 @@ import { Buffer } from 'node:buffer';
13
13
  import sharp from 'sharp';
14
14
  import { z } from 'zod';
15
15
  import { AppError } from '@onco-foundry/errors';
16
+ import { MaskingError } from './masking_error.js';
16
17
  import { clampQuad, polygonSvg, scaleQuad } from './quad.js';
18
+ import { findSensitiveWordsByRules, } from './sensitive_word_finder.js';
17
19
  const DEFAULT_TEXTIN_BASE_URL = 'https://api.textin.com';
18
20
  const DEFAULT_TIMEOUT_MS = 120_000;
19
21
  const DEFAULT_MAX_INPUT_BYTES = 30 * 1024 * 1024;
@@ -31,17 +33,6 @@ const TARGET_LABELS = {
31
33
  doctor_signature: '签字',
32
34
  medical_record_number: '病历号',
33
35
  };
34
- /** 内置规则判定器:身份证号(18 位)与手机号(1 开头 11 位)。 */
35
- export const findSensitiveWordsByRules = (text) => {
36
- const words = [];
37
- for (const match of text.matchAll(/(?<!\d)\d{17}[\dXx](?!\d)/gu)) {
38
- words.push({ text: match[0], target: 'id_number' });
39
- }
40
- for (const match of text.matchAll(/(?<!\d)1[3-9]\d{9}(?!\d)/gu)) {
41
- words.push({ text: match[0], target: 'phone' });
42
- }
43
- return words;
44
- };
45
36
  /** TextIn 行坐标:四边形 4 顶点 8 个数,顺序左上、右上、右下、左下。 */
46
37
  const quad8Schema = z.array(z.number()).length(8);
47
38
  const lineSchema = z.object({
@@ -156,7 +147,7 @@ export const createTextInMasker = (options) => {
156
147
  .jpeg({ quality: 92 })
157
148
  .toBuffer({ resolveWithObject: true })
158
149
  .catch(() => {
159
- throw new AppError('TextIn 脱敏无法解码输入图片', 400);
150
+ throw new MaskingError('image_decode_failed', 'TextIn 脱敏无法解码输入图片', 400);
160
151
  });
161
152
  const imageWidth = upright.info.width;
162
153
  const imageHeight = upright.info.height;
@@ -174,20 +165,20 @@ export const createTextInMasker = (options) => {
174
165
  });
175
166
  }
176
167
  catch (error) {
177
- throw new AppError(`TextIn 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`, 502);
168
+ throw new MaskingError('text_recognition_request_failed', `TextIn 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
178
169
  }
179
170
  if (!response.ok) {
180
- throw new AppError(`TextIn 脱敏 HTTP ${response.status}`, 502);
171
+ throw new MaskingError('text_recognition_http_failed', `TextIn 脱敏 HTTP ${response.status}`);
181
172
  }
182
173
  const payload = recognizeResponseSchema.safeParse(await response.json().catch(() => undefined));
183
174
  if (!payload.success) {
184
- throw new AppError('TextIn 脱敏响应结构校验失败', 502);
175
+ throw new MaskingError('text_recognition_invalid_response', 'TextIn 脱敏响应结构校验失败');
185
176
  }
186
177
  if (payload.data.code !== 200 || payload.data.result === undefined) {
187
- throw new AppError(`TextIn 脱敏接口报错:${payload.data.code} ${payload.data.message}`, 502);
178
+ throw new MaskingError('text_recognition_provider_failed', `TextIn 脱敏接口返回业务错误(code=${payload.data.code})`);
188
179
  }
189
180
  if (options.requireAllMatches && payload.data.result.pages.length !== 1) {
190
- throw new AppError('TextIn 脱敏需要单页图片识别结果', 502);
181
+ throw new MaskingError('text_recognition_multiple_pages', 'TextIn 脱敏需要单页图片识别结果');
191
182
  }
192
183
  const page = payload.data.result.pages[0];
193
184
  const lines = page.lines;
@@ -195,11 +186,19 @@ export const createTextInMasker = (options) => {
195
186
  const scaleX = page.width !== undefined && page.width > 0 ? imageWidth / page.width : 1;
196
187
  const scaleY = page.height !== undefined && page.height > 0 ? imageHeight / page.height : 1;
197
188
  if (options.requireAllMatches && !recognizedText.trim()) {
198
- throw new AppError('TextIn 脱敏未识别出文字,无法确认处理结果', 502);
189
+ throw new MaskingError('text_recognition_empty', 'TextIn 脱敏未识别出文字,无法确认处理结果');
190
+ }
191
+ let detected;
192
+ try {
193
+ detected = await finder(recognizedText);
194
+ }
195
+ catch (error) {
196
+ if (error instanceof MaskingError)
197
+ throw error;
198
+ throw new MaskingError('sensitive_word_failed', '敏感词判定器执行失败');
199
199
  }
200
- const detected = await finder(recognizedText);
201
200
  if (options.requireAllMatches && detected.some(word => !word.text || !(word.target in TARGET_LABELS) || !recognizedText.includes(word.text))) {
202
- throw new AppError('TextIn 脱敏判定结果无法定位', 502);
201
+ throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏判定结果无法定位');
203
202
  }
204
203
  const requested = detected
205
204
  .filter((word) => word.text !== '' && request.targets.includes(word.target));
@@ -217,7 +216,7 @@ export const createTextInMasker = (options) => {
217
216
  const quad = locateWordQuad(locatedLine, span.beginIndex, span.length);
218
217
  if (quad === undefined) {
219
218
  if (options.requireAllMatches)
220
- throw new AppError('TextIn 脱敏敏感词缺少坐标', 502);
219
+ throw new MaskingError('mask_coordinates_missing', 'TextIn 脱敏敏感词缺少坐标');
221
220
  complete = false;
222
221
  break;
223
222
  }
@@ -233,7 +232,7 @@ export const createTextInMasker = (options) => {
233
232
  Math.min(...xs) < 0 || Math.min(...ys) < 0 ||
234
233
  Math.max(...xs) > imageWidth || Math.max(...ys) > imageHeight ||
235
234
  Math.max(...xs) <= Math.min(...xs) || Math.max(...ys) <= Math.min(...ys)) {
236
- throw new AppError('TextIn 脱敏坐标无效', 502);
235
+ throw new MaskingError('mask_coordinates_invalid', 'TextIn 脱敏坐标无效');
237
236
  }
238
237
  }
239
238
  occurrenceQuads.push(clampQuad(scaleQuad(scaleQuadToImage(quad, scaleX, scaleY), SAFETY_SCALE), imageWidth, imageHeight));
@@ -251,17 +250,23 @@ export const createTextInMasker = (options) => {
251
250
  }
252
251
  }
253
252
  if (options.requireAllMatches && located === 0) {
254
- throw new AppError('TextIn 脱敏敏感词无法定位', 502);
253
+ throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏敏感词无法定位');
255
254
  }
256
255
  }
257
256
  const overlay = polygonSvg(imageWidth, imageHeight, finalQuads);
258
- const uprightMasked = await sharp(upright.data)
259
- .composite([{ input: overlay, blend: 'over' }])
260
- .jpeg({ quality: 94 })
261
- .toBuffer();
262
- const maskedImageBytes = rotation === 0
263
- ? uprightMasked
264
- : await sharp(uprightMasked).rotate((360 - rotation) % 360).jpeg({ quality: 94 }).toBuffer();
257
+ let maskedImageBytes;
258
+ try {
259
+ const uprightMasked = await sharp(upright.data)
260
+ .composite([{ input: overlay, blend: 'over' }])
261
+ .jpeg({ quality: 94 })
262
+ .toBuffer();
263
+ maskedImageBytes = rotation === 0
264
+ ? uprightMasked
265
+ : await sharp(uprightMasked).rotate((360 - rotation) % 360).jpeg({ quality: 94 }).toBuffer();
266
+ }
267
+ catch {
268
+ throw new MaskingError('mask_render_failed', 'TextIn 脱敏图片遮盖失败');
269
+ }
265
270
  return {
266
271
  maskedImageBytes: new Uint8Array(maskedImageBytes),
267
272
  mapping: mappings,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@onco-foundry/mask-port",
3
- "version": "0.1.5",
3
+ "version": "0.2.1",
4
4
  "type": "module",
5
5
  "files": [
6
6
  "dist",
@@ -16,10 +16,10 @@
16
16
  "agent-lattice": "0.25.0",
17
17
  "sharp": "^0.35.3",
18
18
  "zod": "^4.4.3",
19
- "@onco-foundry/errors": "0.1.1",
20
19
  "@onco-foundry/capability-registry": "0.1.1",
21
- "@onco-foundry/trace-port": "0.3.2",
22
- "@onco-foundry/resource-versioning": "0.3.0"
20
+ "@onco-foundry/resource-versioning": "0.3.0",
21
+ "@onco-foundry/errors": "0.1.1",
22
+ "@onco-foundry/trace-port": "0.3.3"
23
23
  },
24
24
  "publishConfig": {
25
25
  "access": "public"