@onco-foundry/mask-port 0.1.4 → 0.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -6,3 +6,4 @@ export { redactTextByMapping } from './redact_text.ts';
6
6
  export { createLlmSensitiveWordFinder, type LlmSensitiveWordFinderOptions, } from './llm_sensitive_word_finder.ts';
7
7
  export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, type QwenAgentMasker, type QwenAgentMaskerOptions, type QwenAgentNameMaskInput, } from './qwen_agent_masker.ts';
8
8
  export { createFakeMasker } from './fake_masker.ts';
9
+ export { MaskingError, type MaskingFailureReason } from './masking_error.ts';
package/dist/index.js CHANGED
@@ -6,3 +6,4 @@ export { redactTextByMapping } from './redact_text.js';
6
6
  export { createLlmSensitiveWordFinder, } from './llm_sensitive_word_finder.js';
7
7
  export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, } from './qwen_agent_masker.js';
8
8
  export { createFakeMasker } from './fake_masker.js';
9
+ export { MaskingError } from './masking_error.js';
@@ -9,6 +9,7 @@
9
9
  import { z } from 'zod';
10
10
  import { AppError } from '@onco-foundry/errors';
11
11
  import { MASK_TARGETS } from './masker.js';
12
+ import { MaskingError } from './masking_error.js';
12
13
  const DEFAULT_TIMEOUT_MS = 120_000;
13
14
  // prompt 是模块内置资产,不开放注入:本模块定位就是医疗文书脱敏,提取口径由模块自己保证。
14
15
  const SYSTEM_PROMPT = `你从医疗文书扫描件(病历、检查报告、住院单等)的 OCR 文本中提取隐私信息。只提取以下类别:
@@ -46,13 +47,13 @@ const extractJsonArray = (content) => {
46
47
  const start = cleaned.indexOf('[');
47
48
  const end = cleaned.lastIndexOf(']');
48
49
  if (start < 0 || end <= start) {
49
- throw new AppError('LLM 敏感词提取返回的结构不是 JSON 数组', 502);
50
+ throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构不是 JSON 数组');
50
51
  }
51
52
  try {
52
53
  return JSON.parse(cleaned.slice(start, end + 1));
53
54
  }
54
55
  catch {
55
- throw new AppError('LLM 敏感词提取返回的 JSON 无法解析', 502);
56
+ throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的 JSON 无法解析');
56
57
  }
57
58
  };
58
59
  /**
@@ -97,25 +98,25 @@ export const createLlmSensitiveWordFinder = (options) => {
97
98
  });
98
99
  }
99
100
  catch (error) {
100
- throw new AppError(`LLM 敏感词提取请求未完成:${error instanceof Error ? error.name : '未知错误'}`, 502);
101
+ throw new MaskingError('sensitive_word_request_failed', `LLM 敏感词提取请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
101
102
  }
102
103
  if (!response.ok) {
103
- throw new AppError(`LLM 敏感词提取 HTTP ${response.status}`, 502);
104
+ throw new MaskingError('sensitive_word_http_failed', `LLM 敏感词提取 HTTP ${response.status}`);
104
105
  }
105
106
  const payload = modelResponseSchema.safeParse(await response.json().catch(() => undefined));
106
107
  if (!payload.success) {
107
- throw new AppError('LLM 敏感词提取响应缺少模型输出', 502);
108
+ throw new MaskingError('sensitive_word_invalid_response', 'LLM 敏感词提取响应缺少模型输出');
108
109
  }
109
110
  // 带推理的模型会先回 thinking 块,取第一个 text 块。
110
111
  const textBlock = payload.data.content.find((block) => block.type === 'text' && block.text);
111
112
  if (textBlock?.text === undefined) {
112
113
  const blockTypes = payload.data.content.map((block) => block.type).join('、');
113
- throw new AppError(`LLM 敏感词提取响应缺少文本输出(stop_reason=${payload.data.stop_reason ?? '未知'},`
114
- + `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token 上限)`, 502);
114
+ throw new MaskingError('sensitive_word_output_missing', `LLM 敏感词提取响应缺少文本输出(stop_reason=${payload.data.stop_reason ?? '未知'},`
115
+ + `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token 上限)`);
115
116
  }
116
117
  const parsed = extractedSchema.safeParse(extractJsonArray(textBlock.text));
117
118
  if (!parsed.success) {
118
- throw new AppError('LLM 敏感词提取返回的结构校验失败', 502);
119
+ throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构校验失败');
119
120
  }
120
121
  const seen = new Set();
121
122
  const words = [];
@@ -0,0 +1,7 @@
1
+ import { AppError } from '@onco-foundry/errors';
2
+ /** Safe machine-readable categories. Values never contain source text or provider response bodies. */
3
+ export type MaskingFailureReason = 'image_decode_failed' | 'text_recognition_request_failed' | 'text_recognition_http_failed' | 'text_recognition_invalid_response' | 'text_recognition_provider_failed' | 'text_recognition_multiple_pages' | 'text_recognition_empty' | 'sensitive_word_request_failed' | 'sensitive_word_http_failed' | 'sensitive_word_invalid_response' | 'sensitive_word_invalid_output' | 'sensitive_word_output_missing' | 'sensitive_word_failed' | 'sensitive_word_not_locatable' | 'mask_coordinates_missing' | 'mask_coordinates_invalid' | 'mask_render_failed';
4
+ export declare class MaskingError extends AppError {
5
+ readonly reason: MaskingFailureReason;
6
+ constructor(reason: MaskingFailureReason, message: string, statusCode?: number);
7
+ }
@@ -0,0 +1,9 @@
1
+ import { AppError } from '@onco-foundry/errors';
2
+ export class MaskingError extends AppError {
3
+ reason;
4
+ constructor(reason, message, statusCode = 502) {
5
+ super(message, statusCode);
6
+ this.name = 'MaskingError';
7
+ this.reason = reason;
8
+ }
9
+ }
@@ -37,6 +37,6 @@ export type TextInMaskerOptions = {
37
37
  export declare const findSensitiveWordsByRules: (text: string) => readonly SensitiveWord[];
38
38
  /**
39
39
  * 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
40
- * 同一敏感词在一行出现多次时逐处遮盖、逐处登记 mapping(占位符按类别各自编号)。
40
+ * 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
41
41
  */
42
42
  export declare const createTextInMasker: (options: TextInMaskerOptions) => Masker;
@@ -13,6 +13,7 @@ import { Buffer } from 'node:buffer';
13
13
  import sharp from 'sharp';
14
14
  import { z } from 'zod';
15
15
  import { AppError } from '@onco-foundry/errors';
16
+ import { MaskingError } from './masking_error.js';
16
17
  import { clampQuad, polygonSvg, scaleQuad } from './quad.js';
17
18
  const DEFAULT_TEXTIN_BASE_URL = 'https://api.textin.com';
18
19
  const DEFAULT_TIMEOUT_MS = 120_000;
@@ -82,9 +83,39 @@ const locateWordQuad = (line, beginIndex, wordLength) => {
82
83
  return line.position === undefined ? undefined : toQuad(line.position);
83
84
  };
84
85
  const scaleQuadToImage = (quad, scaleX, scaleY) => quad.map(([x, y]) => [x * scaleX, y * scaleY]);
86
+ /**
87
+ * Find exact occurrences against the same newline-joined text seen by the finder,
88
+ * then split each occurrence into line-local spans for geometry lookup.
89
+ */
90
+ const locateWordSpans = (lines, recognizedText, word) => {
91
+ const lineRanges = [];
92
+ let offset = 0;
93
+ for (const line of lines) {
94
+ lineRanges.push({ line, start: offset, end: offset + line.text.length });
95
+ offset += line.text.length + 1;
96
+ }
97
+ const occurrences = [];
98
+ let fromIndex = 0;
99
+ while (word.length > 0) {
100
+ const hit = recognizedText.indexOf(word, fromIndex);
101
+ if (hit < 0)
102
+ break;
103
+ fromIndex = hit + word.length;
104
+ const end = hit + word.length;
105
+ const spans = lineRanges.flatMap(range => {
106
+ const spanStart = Math.max(hit, range.start);
107
+ const spanEnd = Math.min(end, range.end);
108
+ return spanStart < spanEnd
109
+ ? [{ line: range.line, beginIndex: spanStart - range.start, length: spanEnd - spanStart }]
110
+ : [];
111
+ });
112
+ occurrences.push(spans);
113
+ }
114
+ return occurrences;
115
+ };
85
116
  /**
86
117
  * 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
87
- * 同一敏感词在一行出现多次时逐处遮盖、逐处登记 mapping(占位符按类别各自编号)。
118
+ * 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
88
119
  */
89
120
  export const createTextInMasker = (options) => {
90
121
  const credentials = z.object({
@@ -126,7 +157,7 @@ export const createTextInMasker = (options) => {
126
157
  .jpeg({ quality: 92 })
127
158
  .toBuffer({ resolveWithObject: true })
128
159
  .catch(() => {
129
- throw new AppError('TextIn 脱敏无法解码输入图片', 400);
160
+ throw new MaskingError('image_decode_failed', 'TextIn 脱敏无法解码输入图片', 400);
130
161
  });
131
162
  const imageWidth = upright.info.width;
132
163
  const imageHeight = upright.info.height;
@@ -144,20 +175,20 @@ export const createTextInMasker = (options) => {
144
175
  });
145
176
  }
146
177
  catch (error) {
147
- throw new AppError(`TextIn 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`, 502);
178
+ throw new MaskingError('text_recognition_request_failed', `TextIn 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
148
179
  }
149
180
  if (!response.ok) {
150
- throw new AppError(`TextIn 脱敏 HTTP ${response.status}`, 502);
181
+ throw new MaskingError('text_recognition_http_failed', `TextIn 脱敏 HTTP ${response.status}`);
151
182
  }
152
183
  const payload = recognizeResponseSchema.safeParse(await response.json().catch(() => undefined));
153
184
  if (!payload.success) {
154
- throw new AppError('TextIn 脱敏响应结构校验失败', 502);
185
+ throw new MaskingError('text_recognition_invalid_response', 'TextIn 脱敏响应结构校验失败');
155
186
  }
156
187
  if (payload.data.code !== 200 || payload.data.result === undefined) {
157
- throw new AppError(`TextIn 脱敏接口报错:${payload.data.code} ${payload.data.message}`, 502);
188
+ throw new MaskingError('text_recognition_provider_failed', `TextIn 脱敏接口返回业务错误(code=${payload.data.code})`);
158
189
  }
159
190
  if (options.requireAllMatches && payload.data.result.pages.length !== 1) {
160
- throw new AppError('TextIn 脱敏需要单页图片识别结果', 502);
191
+ throw new MaskingError('text_recognition_multiple_pages', 'TextIn 脱敏需要单页图片识别结果');
161
192
  }
162
193
  const page = payload.data.result.pages[0];
163
194
  const lines = page.lines;
@@ -165,11 +196,19 @@ export const createTextInMasker = (options) => {
165
196
  const scaleX = page.width !== undefined && page.width > 0 ? imageWidth / page.width : 1;
166
197
  const scaleY = page.height !== undefined && page.height > 0 ? imageHeight / page.height : 1;
167
198
  if (options.requireAllMatches && !recognizedText.trim()) {
168
- throw new AppError('TextIn 脱敏未识别出文字,无法确认处理结果', 502);
199
+ throw new MaskingError('text_recognition_empty', 'TextIn 脱敏未识别出文字,无法确认处理结果');
200
+ }
201
+ let detected;
202
+ try {
203
+ detected = await finder(recognizedText);
204
+ }
205
+ catch (error) {
206
+ if (error instanceof MaskingError)
207
+ throw error;
208
+ throw new MaskingError('sensitive_word_failed', '敏感词判定器执行失败');
169
209
  }
170
- const detected = await finder(recognizedText);
171
210
  if (options.requireAllMatches && detected.some(word => !word.text || !(word.target in TARGET_LABELS) || !recognizedText.includes(word.text))) {
172
- throw new AppError('TextIn 脱敏判定结果无法定位', 502);
211
+ throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏判定结果无法定位');
173
212
  }
174
213
  const requested = detected
175
214
  .filter((word) => word.text !== '' && request.targets.includes(word.target));
@@ -178,20 +217,18 @@ export const createTextInMasker = (options) => {
178
217
  const counters = new Map();
179
218
  for (const word of requested) {
180
219
  let located = 0;
181
- for (const line of lines) {
182
- let fromIndex = 0;
183
- while (true) {
184
- const hit = line.text.indexOf(word.text, fromIndex);
185
- if (hit < 0)
186
- break;
187
- fromIndex = hit + word.text.length;
188
- const locatedLine = options.requireAllMatches && line.char_positions?.length !== line.text.length
189
- ? { ...line, char_positions: undefined } : line;
190
- const quad = locateWordQuad(locatedLine, hit, word.text.length);
220
+ for (const spans of locateWordSpans(lines, recognizedText, word.text)) {
221
+ const occurrenceQuads = [];
222
+ let complete = spans.length > 0;
223
+ for (const span of spans) {
224
+ const locatedLine = options.requireAllMatches && span.line.char_positions?.length !== span.line.text.length
225
+ ? { ...span.line, char_positions: undefined } : span.line;
226
+ const quad = locateWordQuad(locatedLine, span.beginIndex, span.length);
191
227
  if (quad === undefined) {
192
228
  if (options.requireAllMatches)
193
- throw new AppError('TextIn 脱敏敏感词缺少坐标', 502);
194
- continue;
229
+ throw new MaskingError('mask_coordinates_missing', 'TextIn 脱敏敏感词缺少坐标');
230
+ complete = false;
231
+ break;
195
232
  }
196
233
  if (options.requireAllMatches) {
197
234
  const scaled = scaleQuadToImage(quad, scaleX, scaleY);
@@ -205,11 +242,14 @@ export const createTextInMasker = (options) => {
205
242
  Math.min(...xs) < 0 || Math.min(...ys) < 0 ||
206
243
  Math.max(...xs) > imageWidth || Math.max(...ys) > imageHeight ||
207
244
  Math.max(...xs) <= Math.min(...xs) || Math.max(...ys) <= Math.min(...ys)) {
208
- throw new AppError('TextIn 脱敏坐标无效', 502);
245
+ throw new MaskingError('mask_coordinates_invalid', 'TextIn 脱敏坐标无效');
209
246
  }
210
247
  }
248
+ occurrenceQuads.push(clampQuad(scaleQuad(scaleQuadToImage(quad, scaleX, scaleY), SAFETY_SCALE), imageWidth, imageHeight));
249
+ }
250
+ if (complete && occurrenceQuads.length > 0) {
211
251
  located++;
212
- finalQuads.push(clampQuad(scaleQuad(scaleQuadToImage(quad, scaleX, scaleY), SAFETY_SCALE), imageWidth, imageHeight));
252
+ finalQuads.push(...occurrenceQuads);
213
253
  const count = (counters.get(word.target) ?? 0) + 1;
214
254
  counters.set(word.target, count);
215
255
  mappings.push({
@@ -220,17 +260,23 @@ export const createTextInMasker = (options) => {
220
260
  }
221
261
  }
222
262
  if (options.requireAllMatches && located === 0) {
223
- throw new AppError('TextIn 脱敏敏感词无法在单行定位', 502);
263
+ throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏敏感词无法定位');
224
264
  }
225
265
  }
226
266
  const overlay = polygonSvg(imageWidth, imageHeight, finalQuads);
227
- const uprightMasked = await sharp(upright.data)
228
- .composite([{ input: overlay, blend: 'over' }])
229
- .jpeg({ quality: 94 })
230
- .toBuffer();
231
- const maskedImageBytes = rotation === 0
232
- ? uprightMasked
233
- : await sharp(uprightMasked).rotate((360 - rotation) % 360).jpeg({ quality: 94 }).toBuffer();
267
+ let maskedImageBytes;
268
+ try {
269
+ const uprightMasked = await sharp(upright.data)
270
+ .composite([{ input: overlay, blend: 'over' }])
271
+ .jpeg({ quality: 94 })
272
+ .toBuffer();
273
+ maskedImageBytes = rotation === 0
274
+ ? uprightMasked
275
+ : await sharp(uprightMasked).rotate((360 - rotation) % 360).jpeg({ quality: 94 }).toBuffer();
276
+ }
277
+ catch {
278
+ throw new MaskingError('mask_render_failed', 'TextIn 脱敏图片遮盖失败');
279
+ }
234
280
  return {
235
281
  maskedImageBytes: new Uint8Array(maskedImageBytes),
236
282
  mapping: mappings,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@onco-foundry/mask-port",
3
- "version": "0.1.4",
3
+ "version": "0.1.6",
4
4
  "type": "module",
5
5
  "files": [
6
6
  "dist",
@@ -18,8 +18,8 @@
18
18
  "zod": "^4.4.3",
19
19
  "@onco-foundry/capability-registry": "0.1.1",
20
20
  "@onco-foundry/errors": "0.1.1",
21
- "@onco-foundry/trace-port": "0.3.2",
22
- "@onco-foundry/resource-versioning": "0.3.0"
21
+ "@onco-foundry/resource-versioning": "0.3.0",
22
+ "@onco-foundry/trace-port": "0.3.2"
23
23
  },
24
24
  "publishConfig": {
25
25
  "access": "public"