@onco-foundry/mask-port 0.1.5 → 0.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/llm_sensitive_word_finder.js +9 -8
- package/dist/masking_error.d.ts +7 -0
- package/dist/masking_error.js +9 -0
- package/dist/textin_masker.js +34 -19
- package/package.json +4 -4
package/dist/index.d.ts
CHANGED
|
@@ -6,3 +6,4 @@ export { redactTextByMapping } from './redact_text.ts';
|
|
|
6
6
|
export { createLlmSensitiveWordFinder, type LlmSensitiveWordFinderOptions, } from './llm_sensitive_word_finder.ts';
|
|
7
7
|
export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, type QwenAgentMasker, type QwenAgentMaskerOptions, type QwenAgentNameMaskInput, } from './qwen_agent_masker.ts';
|
|
8
8
|
export { createFakeMasker } from './fake_masker.ts';
|
|
9
|
+
export { MaskingError, type MaskingFailureReason } from './masking_error.ts';
|
package/dist/index.js
CHANGED
|
@@ -6,3 +6,4 @@ export { redactTextByMapping } from './redact_text.js';
|
|
|
6
6
|
export { createLlmSensitiveWordFinder, } from './llm_sensitive_word_finder.js';
|
|
7
7
|
export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, } from './qwen_agent_masker.js';
|
|
8
8
|
export { createFakeMasker } from './fake_masker.js';
|
|
9
|
+
export { MaskingError } from './masking_error.js';
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
import { z } from 'zod';
|
|
10
10
|
import { AppError } from '@onco-foundry/errors';
|
|
11
11
|
import { MASK_TARGETS } from './masker.js';
|
|
12
|
+
import { MaskingError } from './masking_error.js';
|
|
12
13
|
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
13
14
|
// prompt 是模块内置资产,不开放注入:本模块定位就是医疗文书脱敏,提取口径由模块自己保证。
|
|
14
15
|
const SYSTEM_PROMPT = `你从医疗文书扫描件(病历、检查报告、住院单等)的 OCR 文本中提取隐私信息。只提取以下类别:
|
|
@@ -46,13 +47,13 @@ const extractJsonArray = (content) => {
|
|
|
46
47
|
const start = cleaned.indexOf('[');
|
|
47
48
|
const end = cleaned.lastIndexOf(']');
|
|
48
49
|
if (start < 0 || end <= start) {
|
|
49
|
-
throw new
|
|
50
|
+
throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构不是 JSON 数组');
|
|
50
51
|
}
|
|
51
52
|
try {
|
|
52
53
|
return JSON.parse(cleaned.slice(start, end + 1));
|
|
53
54
|
}
|
|
54
55
|
catch {
|
|
55
|
-
throw new
|
|
56
|
+
throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的 JSON 无法解析');
|
|
56
57
|
}
|
|
57
58
|
};
|
|
58
59
|
/**
|
|
@@ -97,25 +98,25 @@ export const createLlmSensitiveWordFinder = (options) => {
|
|
|
97
98
|
});
|
|
98
99
|
}
|
|
99
100
|
catch (error) {
|
|
100
|
-
throw new
|
|
101
|
+
throw new MaskingError('sensitive_word_request_failed', `LLM 敏感词提取请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
|
|
101
102
|
}
|
|
102
103
|
if (!response.ok) {
|
|
103
|
-
throw new
|
|
104
|
+
throw new MaskingError('sensitive_word_http_failed', `LLM 敏感词提取 HTTP ${response.status}`);
|
|
104
105
|
}
|
|
105
106
|
const payload = modelResponseSchema.safeParse(await response.json().catch(() => undefined));
|
|
106
107
|
if (!payload.success) {
|
|
107
|
-
throw new
|
|
108
|
+
throw new MaskingError('sensitive_word_invalid_response', 'LLM 敏感词提取响应缺少模型输出');
|
|
108
109
|
}
|
|
109
110
|
// 带推理的模型会先回 thinking 块,取第一个 text 块。
|
|
110
111
|
const textBlock = payload.data.content.find((block) => block.type === 'text' && block.text);
|
|
111
112
|
if (textBlock?.text === undefined) {
|
|
112
113
|
const blockTypes = payload.data.content.map((block) => block.type).join('、');
|
|
113
|
-
throw new
|
|
114
|
-
+ `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token
|
|
114
|
+
throw new MaskingError('sensitive_word_output_missing', `LLM 敏感词提取响应缺少文本输出(stop_reason=${payload.data.stop_reason ?? '未知'},`
|
|
115
|
+
+ `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token 上限)`);
|
|
115
116
|
}
|
|
116
117
|
const parsed = extractedSchema.safeParse(extractJsonArray(textBlock.text));
|
|
117
118
|
if (!parsed.success) {
|
|
118
|
-
throw new
|
|
119
|
+
throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构校验失败');
|
|
119
120
|
}
|
|
120
121
|
const seen = new Set();
|
|
121
122
|
const words = [];
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { AppError } from '@onco-foundry/errors';
|
|
2
|
+
/** Safe machine-readable categories. Values never contain source text or provider response bodies. */
|
|
3
|
+
export type MaskingFailureReason = 'image_decode_failed' | 'text_recognition_request_failed' | 'text_recognition_http_failed' | 'text_recognition_invalid_response' | 'text_recognition_provider_failed' | 'text_recognition_multiple_pages' | 'text_recognition_empty' | 'sensitive_word_request_failed' | 'sensitive_word_http_failed' | 'sensitive_word_invalid_response' | 'sensitive_word_invalid_output' | 'sensitive_word_output_missing' | 'sensitive_word_failed' | 'sensitive_word_not_locatable' | 'mask_coordinates_missing' | 'mask_coordinates_invalid' | 'mask_render_failed';
|
|
4
|
+
export declare class MaskingError extends AppError {
|
|
5
|
+
readonly reason: MaskingFailureReason;
|
|
6
|
+
constructor(reason: MaskingFailureReason, message: string, statusCode?: number);
|
|
7
|
+
}
|
package/dist/textin_masker.js
CHANGED
|
@@ -13,6 +13,7 @@ import { Buffer } from 'node:buffer';
|
|
|
13
13
|
import sharp from 'sharp';
|
|
14
14
|
import { z } from 'zod';
|
|
15
15
|
import { AppError } from '@onco-foundry/errors';
|
|
16
|
+
import { MaskingError } from './masking_error.js';
|
|
16
17
|
import { clampQuad, polygonSvg, scaleQuad } from './quad.js';
|
|
17
18
|
const DEFAULT_TEXTIN_BASE_URL = 'https://api.textin.com';
|
|
18
19
|
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
@@ -156,7 +157,7 @@ export const createTextInMasker = (options) => {
|
|
|
156
157
|
.jpeg({ quality: 92 })
|
|
157
158
|
.toBuffer({ resolveWithObject: true })
|
|
158
159
|
.catch(() => {
|
|
159
|
-
throw new
|
|
160
|
+
throw new MaskingError('image_decode_failed', 'TextIn 脱敏无法解码输入图片', 400);
|
|
160
161
|
});
|
|
161
162
|
const imageWidth = upright.info.width;
|
|
162
163
|
const imageHeight = upright.info.height;
|
|
@@ -174,20 +175,20 @@ export const createTextInMasker = (options) => {
|
|
|
174
175
|
});
|
|
175
176
|
}
|
|
176
177
|
catch (error) {
|
|
177
|
-
throw new
|
|
178
|
+
throw new MaskingError('text_recognition_request_failed', `TextIn 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
|
|
178
179
|
}
|
|
179
180
|
if (!response.ok) {
|
|
180
|
-
throw new
|
|
181
|
+
throw new MaskingError('text_recognition_http_failed', `TextIn 脱敏 HTTP ${response.status}`);
|
|
181
182
|
}
|
|
182
183
|
const payload = recognizeResponseSchema.safeParse(await response.json().catch(() => undefined));
|
|
183
184
|
if (!payload.success) {
|
|
184
|
-
throw new
|
|
185
|
+
throw new MaskingError('text_recognition_invalid_response', 'TextIn 脱敏响应结构校验失败');
|
|
185
186
|
}
|
|
186
187
|
if (payload.data.code !== 200 || payload.data.result === undefined) {
|
|
187
|
-
throw new
|
|
188
|
+
throw new MaskingError('text_recognition_provider_failed', `TextIn 脱敏接口返回业务错误(code=${payload.data.code})`);
|
|
188
189
|
}
|
|
189
190
|
if (options.requireAllMatches && payload.data.result.pages.length !== 1) {
|
|
190
|
-
throw new
|
|
191
|
+
throw new MaskingError('text_recognition_multiple_pages', 'TextIn 脱敏需要单页图片识别结果');
|
|
191
192
|
}
|
|
192
193
|
const page = payload.data.result.pages[0];
|
|
193
194
|
const lines = page.lines;
|
|
@@ -195,11 +196,19 @@ export const createTextInMasker = (options) => {
|
|
|
195
196
|
const scaleX = page.width !== undefined && page.width > 0 ? imageWidth / page.width : 1;
|
|
196
197
|
const scaleY = page.height !== undefined && page.height > 0 ? imageHeight / page.height : 1;
|
|
197
198
|
if (options.requireAllMatches && !recognizedText.trim()) {
|
|
198
|
-
throw new
|
|
199
|
+
throw new MaskingError('text_recognition_empty', 'TextIn 脱敏未识别出文字,无法确认处理结果');
|
|
200
|
+
}
|
|
201
|
+
let detected;
|
|
202
|
+
try {
|
|
203
|
+
detected = await finder(recognizedText);
|
|
204
|
+
}
|
|
205
|
+
catch (error) {
|
|
206
|
+
if (error instanceof MaskingError)
|
|
207
|
+
throw error;
|
|
208
|
+
throw new MaskingError('sensitive_word_failed', '敏感词判定器执行失败');
|
|
199
209
|
}
|
|
200
|
-
const detected = await finder(recognizedText);
|
|
201
210
|
if (options.requireAllMatches && detected.some(word => !word.text || !(word.target in TARGET_LABELS) || !recognizedText.includes(word.text))) {
|
|
202
|
-
throw new
|
|
211
|
+
throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏判定结果无法定位');
|
|
203
212
|
}
|
|
204
213
|
const requested = detected
|
|
205
214
|
.filter((word) => word.text !== '' && request.targets.includes(word.target));
|
|
@@ -217,7 +226,7 @@ export const createTextInMasker = (options) => {
|
|
|
217
226
|
const quad = locateWordQuad(locatedLine, span.beginIndex, span.length);
|
|
218
227
|
if (quad === undefined) {
|
|
219
228
|
if (options.requireAllMatches)
|
|
220
|
-
throw new
|
|
229
|
+
throw new MaskingError('mask_coordinates_missing', 'TextIn 脱敏敏感词缺少坐标');
|
|
221
230
|
complete = false;
|
|
222
231
|
break;
|
|
223
232
|
}
|
|
@@ -233,7 +242,7 @@ export const createTextInMasker = (options) => {
|
|
|
233
242
|
Math.min(...xs) < 0 || Math.min(...ys) < 0 ||
|
|
234
243
|
Math.max(...xs) > imageWidth || Math.max(...ys) > imageHeight ||
|
|
235
244
|
Math.max(...xs) <= Math.min(...xs) || Math.max(...ys) <= Math.min(...ys)) {
|
|
236
|
-
throw new
|
|
245
|
+
throw new MaskingError('mask_coordinates_invalid', 'TextIn 脱敏坐标无效');
|
|
237
246
|
}
|
|
238
247
|
}
|
|
239
248
|
occurrenceQuads.push(clampQuad(scaleQuad(scaleQuadToImage(quad, scaleX, scaleY), SAFETY_SCALE), imageWidth, imageHeight));
|
|
@@ -251,17 +260,23 @@ export const createTextInMasker = (options) => {
|
|
|
251
260
|
}
|
|
252
261
|
}
|
|
253
262
|
if (options.requireAllMatches && located === 0) {
|
|
254
|
-
throw new
|
|
263
|
+
throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏敏感词无法定位');
|
|
255
264
|
}
|
|
256
265
|
}
|
|
257
266
|
const overlay = polygonSvg(imageWidth, imageHeight, finalQuads);
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
267
|
+
let maskedImageBytes;
|
|
268
|
+
try {
|
|
269
|
+
const uprightMasked = await sharp(upright.data)
|
|
270
|
+
.composite([{ input: overlay, blend: 'over' }])
|
|
271
|
+
.jpeg({ quality: 94 })
|
|
272
|
+
.toBuffer();
|
|
273
|
+
maskedImageBytes = rotation === 0
|
|
274
|
+
? uprightMasked
|
|
275
|
+
: await sharp(uprightMasked).rotate((360 - rotation) % 360).jpeg({ quality: 94 }).toBuffer();
|
|
276
|
+
}
|
|
277
|
+
catch {
|
|
278
|
+
throw new MaskingError('mask_render_failed', 'TextIn 脱敏图片遮盖失败');
|
|
279
|
+
}
|
|
265
280
|
return {
|
|
266
281
|
maskedImageBytes: new Uint8Array(maskedImageBytes),
|
|
267
282
|
mapping: mappings,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@onco-foundry/mask-port",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.6",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"files": [
|
|
6
6
|
"dist",
|
|
@@ -16,10 +16,10 @@
|
|
|
16
16
|
"agent-lattice": "0.25.0",
|
|
17
17
|
"sharp": "^0.35.3",
|
|
18
18
|
"zod": "^4.4.3",
|
|
19
|
-
"@onco-foundry/errors": "0.1.1",
|
|
20
19
|
"@onco-foundry/capability-registry": "0.1.1",
|
|
21
|
-
"@onco-foundry/
|
|
22
|
-
"@onco-foundry/resource-versioning": "0.3.0"
|
|
20
|
+
"@onco-foundry/errors": "0.1.1",
|
|
21
|
+
"@onco-foundry/resource-versioning": "0.3.0",
|
|
22
|
+
"@onco-foundry/trace-port": "0.3.2"
|
|
23
23
|
},
|
|
24
24
|
"publishConfig": {
|
|
25
25
|
"access": "public"
|