@onco-foundry/mask-port 0.1.4 → 0.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/llm_sensitive_word_finder.js +9 -8
- package/dist/masking_error.d.ts +7 -0
- package/dist/masking_error.js +9 -0
- package/dist/textin_masker.d.ts +1 -1
- package/dist/textin_masker.js +78 -32
- package/package.json +3 -3
package/dist/index.d.ts
CHANGED
|
@@ -6,3 +6,4 @@ export { redactTextByMapping } from './redact_text.ts';
|
|
|
6
6
|
export { createLlmSensitiveWordFinder, type LlmSensitiveWordFinderOptions, } from './llm_sensitive_word_finder.ts';
|
|
7
7
|
export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, type QwenAgentMasker, type QwenAgentMaskerOptions, type QwenAgentNameMaskInput, } from './qwen_agent_masker.ts';
|
|
8
8
|
export { createFakeMasker } from './fake_masker.ts';
|
|
9
|
+
export { MaskingError, type MaskingFailureReason } from './masking_error.ts';
|
package/dist/index.js
CHANGED
|
@@ -6,3 +6,4 @@ export { redactTextByMapping } from './redact_text.js';
|
|
|
6
6
|
export { createLlmSensitiveWordFinder, } from './llm_sensitive_word_finder.js';
|
|
7
7
|
export { createQwenAgentMasker, qwenAgentNameMaskCapabilityCard, qwenAgentNameMaskInputSchema, } from './qwen_agent_masker.js';
|
|
8
8
|
export { createFakeMasker } from './fake_masker.js';
|
|
9
|
+
export { MaskingError } from './masking_error.js';
|
|
@@ -9,6 +9,7 @@
|
|
|
9
9
|
import { z } from 'zod';
|
|
10
10
|
import { AppError } from '@onco-foundry/errors';
|
|
11
11
|
import { MASK_TARGETS } from './masker.js';
|
|
12
|
+
import { MaskingError } from './masking_error.js';
|
|
12
13
|
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
13
14
|
// prompt 是模块内置资产,不开放注入:本模块定位就是医疗文书脱敏,提取口径由模块自己保证。
|
|
14
15
|
const SYSTEM_PROMPT = `你从医疗文书扫描件(病历、检查报告、住院单等)的 OCR 文本中提取隐私信息。只提取以下类别:
|
|
@@ -46,13 +47,13 @@ const extractJsonArray = (content) => {
|
|
|
46
47
|
const start = cleaned.indexOf('[');
|
|
47
48
|
const end = cleaned.lastIndexOf(']');
|
|
48
49
|
if (start < 0 || end <= start) {
|
|
49
|
-
throw new
|
|
50
|
+
throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构不是 JSON 数组');
|
|
50
51
|
}
|
|
51
52
|
try {
|
|
52
53
|
return JSON.parse(cleaned.slice(start, end + 1));
|
|
53
54
|
}
|
|
54
55
|
catch {
|
|
55
|
-
throw new
|
|
56
|
+
throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的 JSON 无法解析');
|
|
56
57
|
}
|
|
57
58
|
};
|
|
58
59
|
/**
|
|
@@ -97,25 +98,25 @@ export const createLlmSensitiveWordFinder = (options) => {
|
|
|
97
98
|
});
|
|
98
99
|
}
|
|
99
100
|
catch (error) {
|
|
100
|
-
throw new
|
|
101
|
+
throw new MaskingError('sensitive_word_request_failed', `LLM 敏感词提取请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
|
|
101
102
|
}
|
|
102
103
|
if (!response.ok) {
|
|
103
|
-
throw new
|
|
104
|
+
throw new MaskingError('sensitive_word_http_failed', `LLM 敏感词提取 HTTP ${response.status}`);
|
|
104
105
|
}
|
|
105
106
|
const payload = modelResponseSchema.safeParse(await response.json().catch(() => undefined));
|
|
106
107
|
if (!payload.success) {
|
|
107
|
-
throw new
|
|
108
|
+
throw new MaskingError('sensitive_word_invalid_response', 'LLM 敏感词提取响应缺少模型输出');
|
|
108
109
|
}
|
|
109
110
|
// 带推理的模型会先回 thinking 块,取第一个 text 块。
|
|
110
111
|
const textBlock = payload.data.content.find((block) => block.type === 'text' && block.text);
|
|
111
112
|
if (textBlock?.text === undefined) {
|
|
112
113
|
const blockTypes = payload.data.content.map((block) => block.type).join('、');
|
|
113
|
-
throw new
|
|
114
|
-
+ `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token
|
|
114
|
+
throw new MaskingError('sensitive_word_output_missing', `LLM 敏感词提取响应缺少文本输出(stop_reason=${payload.data.stop_reason ?? '未知'},`
|
|
115
|
+
+ `块类型=${blockTypes};stop_reason=max_tokens 说明思考烧光了 token 上限)`);
|
|
115
116
|
}
|
|
116
117
|
const parsed = extractedSchema.safeParse(extractJsonArray(textBlock.text));
|
|
117
118
|
if (!parsed.success) {
|
|
118
|
-
throw new
|
|
119
|
+
throw new MaskingError('sensitive_word_invalid_output', 'LLM 敏感词提取返回的结构校验失败');
|
|
119
120
|
}
|
|
120
121
|
const seen = new Set();
|
|
121
122
|
const words = [];
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import { AppError } from '@onco-foundry/errors';
|
|
2
|
+
/** Safe machine-readable categories. Values never contain source text or provider response bodies. */
|
|
3
|
+
export type MaskingFailureReason = 'image_decode_failed' | 'text_recognition_request_failed' | 'text_recognition_http_failed' | 'text_recognition_invalid_response' | 'text_recognition_provider_failed' | 'text_recognition_multiple_pages' | 'text_recognition_empty' | 'sensitive_word_request_failed' | 'sensitive_word_http_failed' | 'sensitive_word_invalid_response' | 'sensitive_word_invalid_output' | 'sensitive_word_output_missing' | 'sensitive_word_failed' | 'sensitive_word_not_locatable' | 'mask_coordinates_missing' | 'mask_coordinates_invalid' | 'mask_render_failed';
|
|
4
|
+
export declare class MaskingError extends AppError {
|
|
5
|
+
readonly reason: MaskingFailureReason;
|
|
6
|
+
constructor(reason: MaskingFailureReason, message: string, statusCode?: number);
|
|
7
|
+
}
|
package/dist/textin_masker.d.ts
CHANGED
|
@@ -37,6 +37,6 @@ export type TextInMaskerOptions = {
|
|
|
37
37
|
export declare const findSensitiveWordsByRules: (text: string) => readonly SensitiveWord[];
|
|
38
38
|
/**
|
|
39
39
|
* 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
|
|
40
|
-
*
|
|
40
|
+
* 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
|
|
41
41
|
*/
|
|
42
42
|
export declare const createTextInMasker: (options: TextInMaskerOptions) => Masker;
|
package/dist/textin_masker.js
CHANGED
|
@@ -13,6 +13,7 @@ import { Buffer } from 'node:buffer';
|
|
|
13
13
|
import sharp from 'sharp';
|
|
14
14
|
import { z } from 'zod';
|
|
15
15
|
import { AppError } from '@onco-foundry/errors';
|
|
16
|
+
import { MaskingError } from './masking_error.js';
|
|
16
17
|
import { clampQuad, polygonSvg, scaleQuad } from './quad.js';
|
|
17
18
|
const DEFAULT_TEXTIN_BASE_URL = 'https://api.textin.com';
|
|
18
19
|
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
@@ -82,9 +83,39 @@ const locateWordQuad = (line, beginIndex, wordLength) => {
|
|
|
82
83
|
return line.position === undefined ? undefined : toQuad(line.position);
|
|
83
84
|
};
|
|
84
85
|
const scaleQuadToImage = (quad, scaleX, scaleY) => quad.map(([x, y]) => [x * scaleX, y * scaleY]);
|
|
86
|
+
/**
|
|
87
|
+
* Find exact occurrences against the same newline-joined text seen by the finder,
|
|
88
|
+
* then split each occurrence into line-local spans for geometry lookup.
|
|
89
|
+
*/
|
|
90
|
+
const locateWordSpans = (lines, recognizedText, word) => {
|
|
91
|
+
const lineRanges = [];
|
|
92
|
+
let offset = 0;
|
|
93
|
+
for (const line of lines) {
|
|
94
|
+
lineRanges.push({ line, start: offset, end: offset + line.text.length });
|
|
95
|
+
offset += line.text.length + 1;
|
|
96
|
+
}
|
|
97
|
+
const occurrences = [];
|
|
98
|
+
let fromIndex = 0;
|
|
99
|
+
while (word.length > 0) {
|
|
100
|
+
const hit = recognizedText.indexOf(word, fromIndex);
|
|
101
|
+
if (hit < 0)
|
|
102
|
+
break;
|
|
103
|
+
fromIndex = hit + word.length;
|
|
104
|
+
const end = hit + word.length;
|
|
105
|
+
const spans = lineRanges.flatMap(range => {
|
|
106
|
+
const spanStart = Math.max(hit, range.start);
|
|
107
|
+
const spanEnd = Math.min(end, range.end);
|
|
108
|
+
return spanStart < spanEnd
|
|
109
|
+
? [{ line: range.line, beginIndex: spanStart - range.start, length: spanEnd - spanStart }]
|
|
110
|
+
: [];
|
|
111
|
+
});
|
|
112
|
+
occurrences.push(spans);
|
|
113
|
+
}
|
|
114
|
+
return occurrences;
|
|
115
|
+
};
|
|
85
116
|
/**
|
|
86
117
|
* 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
|
|
87
|
-
*
|
|
118
|
+
* 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
|
|
88
119
|
*/
|
|
89
120
|
export const createTextInMasker = (options) => {
|
|
90
121
|
const credentials = z.object({
|
|
@@ -126,7 +157,7 @@ export const createTextInMasker = (options) => {
|
|
|
126
157
|
.jpeg({ quality: 92 })
|
|
127
158
|
.toBuffer({ resolveWithObject: true })
|
|
128
159
|
.catch(() => {
|
|
129
|
-
throw new
|
|
160
|
+
throw new MaskingError('image_decode_failed', 'TextIn 脱敏无法解码输入图片', 400);
|
|
130
161
|
});
|
|
131
162
|
const imageWidth = upright.info.width;
|
|
132
163
|
const imageHeight = upright.info.height;
|
|
@@ -144,20 +175,20 @@ export const createTextInMasker = (options) => {
|
|
|
144
175
|
});
|
|
145
176
|
}
|
|
146
177
|
catch (error) {
|
|
147
|
-
throw new
|
|
178
|
+
throw new MaskingError('text_recognition_request_failed', `TextIn 脱敏请求未完成:${error instanceof Error ? error.name : '未知错误'}`);
|
|
148
179
|
}
|
|
149
180
|
if (!response.ok) {
|
|
150
|
-
throw new
|
|
181
|
+
throw new MaskingError('text_recognition_http_failed', `TextIn 脱敏 HTTP ${response.status}`);
|
|
151
182
|
}
|
|
152
183
|
const payload = recognizeResponseSchema.safeParse(await response.json().catch(() => undefined));
|
|
153
184
|
if (!payload.success) {
|
|
154
|
-
throw new
|
|
185
|
+
throw new MaskingError('text_recognition_invalid_response', 'TextIn 脱敏响应结构校验失败');
|
|
155
186
|
}
|
|
156
187
|
if (payload.data.code !== 200 || payload.data.result === undefined) {
|
|
157
|
-
throw new
|
|
188
|
+
throw new MaskingError('text_recognition_provider_failed', `TextIn 脱敏接口返回业务错误(code=${payload.data.code})`);
|
|
158
189
|
}
|
|
159
190
|
if (options.requireAllMatches && payload.data.result.pages.length !== 1) {
|
|
160
|
-
throw new
|
|
191
|
+
throw new MaskingError('text_recognition_multiple_pages', 'TextIn 脱敏需要单页图片识别结果');
|
|
161
192
|
}
|
|
162
193
|
const page = payload.data.result.pages[0];
|
|
163
194
|
const lines = page.lines;
|
|
@@ -165,11 +196,19 @@ export const createTextInMasker = (options) => {
|
|
|
165
196
|
const scaleX = page.width !== undefined && page.width > 0 ? imageWidth / page.width : 1;
|
|
166
197
|
const scaleY = page.height !== undefined && page.height > 0 ? imageHeight / page.height : 1;
|
|
167
198
|
if (options.requireAllMatches && !recognizedText.trim()) {
|
|
168
|
-
throw new
|
|
199
|
+
throw new MaskingError('text_recognition_empty', 'TextIn 脱敏未识别出文字,无法确认处理结果');
|
|
200
|
+
}
|
|
201
|
+
let detected;
|
|
202
|
+
try {
|
|
203
|
+
detected = await finder(recognizedText);
|
|
204
|
+
}
|
|
205
|
+
catch (error) {
|
|
206
|
+
if (error instanceof MaskingError)
|
|
207
|
+
throw error;
|
|
208
|
+
throw new MaskingError('sensitive_word_failed', '敏感词判定器执行失败');
|
|
169
209
|
}
|
|
170
|
-
const detected = await finder(recognizedText);
|
|
171
210
|
if (options.requireAllMatches && detected.some(word => !word.text || !(word.target in TARGET_LABELS) || !recognizedText.includes(word.text))) {
|
|
172
|
-
throw new
|
|
211
|
+
throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏判定结果无法定位');
|
|
173
212
|
}
|
|
174
213
|
const requested = detected
|
|
175
214
|
.filter((word) => word.text !== '' && request.targets.includes(word.target));
|
|
@@ -178,20 +217,18 @@ export const createTextInMasker = (options) => {
|
|
|
178
217
|
const counters = new Map();
|
|
179
218
|
for (const word of requested) {
|
|
180
219
|
let located = 0;
|
|
181
|
-
for (const
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
const locatedLine = options.requireAllMatches && line.char_positions?.length !== line.text.length
|
|
189
|
-
? { ...line, char_positions: undefined } : line;
|
|
190
|
-
const quad = locateWordQuad(locatedLine, hit, word.text.length);
|
|
220
|
+
for (const spans of locateWordSpans(lines, recognizedText, word.text)) {
|
|
221
|
+
const occurrenceQuads = [];
|
|
222
|
+
let complete = spans.length > 0;
|
|
223
|
+
for (const span of spans) {
|
|
224
|
+
const locatedLine = options.requireAllMatches && span.line.char_positions?.length !== span.line.text.length
|
|
225
|
+
? { ...span.line, char_positions: undefined } : span.line;
|
|
226
|
+
const quad = locateWordQuad(locatedLine, span.beginIndex, span.length);
|
|
191
227
|
if (quad === undefined) {
|
|
192
228
|
if (options.requireAllMatches)
|
|
193
|
-
throw new
|
|
194
|
-
|
|
229
|
+
throw new MaskingError('mask_coordinates_missing', 'TextIn 脱敏敏感词缺少坐标');
|
|
230
|
+
complete = false;
|
|
231
|
+
break;
|
|
195
232
|
}
|
|
196
233
|
if (options.requireAllMatches) {
|
|
197
234
|
const scaled = scaleQuadToImage(quad, scaleX, scaleY);
|
|
@@ -205,11 +242,14 @@ export const createTextInMasker = (options) => {
|
|
|
205
242
|
Math.min(...xs) < 0 || Math.min(...ys) < 0 ||
|
|
206
243
|
Math.max(...xs) > imageWidth || Math.max(...ys) > imageHeight ||
|
|
207
244
|
Math.max(...xs) <= Math.min(...xs) || Math.max(...ys) <= Math.min(...ys)) {
|
|
208
|
-
throw new
|
|
245
|
+
throw new MaskingError('mask_coordinates_invalid', 'TextIn 脱敏坐标无效');
|
|
209
246
|
}
|
|
210
247
|
}
|
|
248
|
+
occurrenceQuads.push(clampQuad(scaleQuad(scaleQuadToImage(quad, scaleX, scaleY), SAFETY_SCALE), imageWidth, imageHeight));
|
|
249
|
+
}
|
|
250
|
+
if (complete && occurrenceQuads.length > 0) {
|
|
211
251
|
located++;
|
|
212
|
-
finalQuads.push(
|
|
252
|
+
finalQuads.push(...occurrenceQuads);
|
|
213
253
|
const count = (counters.get(word.target) ?? 0) + 1;
|
|
214
254
|
counters.set(word.target, count);
|
|
215
255
|
mappings.push({
|
|
@@ -220,17 +260,23 @@ export const createTextInMasker = (options) => {
|
|
|
220
260
|
}
|
|
221
261
|
}
|
|
222
262
|
if (options.requireAllMatches && located === 0) {
|
|
223
|
-
throw new
|
|
263
|
+
throw new MaskingError('sensitive_word_not_locatable', 'TextIn 脱敏敏感词无法定位');
|
|
224
264
|
}
|
|
225
265
|
}
|
|
226
266
|
const overlay = polygonSvg(imageWidth, imageHeight, finalQuads);
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
267
|
+
let maskedImageBytes;
|
|
268
|
+
try {
|
|
269
|
+
const uprightMasked = await sharp(upright.data)
|
|
270
|
+
.composite([{ input: overlay, blend: 'over' }])
|
|
271
|
+
.jpeg({ quality: 94 })
|
|
272
|
+
.toBuffer();
|
|
273
|
+
maskedImageBytes = rotation === 0
|
|
274
|
+
? uprightMasked
|
|
275
|
+
: await sharp(uprightMasked).rotate((360 - rotation) % 360).jpeg({ quality: 94 }).toBuffer();
|
|
276
|
+
}
|
|
277
|
+
catch {
|
|
278
|
+
throw new MaskingError('mask_render_failed', 'TextIn 脱敏图片遮盖失败');
|
|
279
|
+
}
|
|
234
280
|
return {
|
|
235
281
|
maskedImageBytes: new Uint8Array(maskedImageBytes),
|
|
236
282
|
mapping: mappings,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@onco-foundry/mask-port",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.6",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"files": [
|
|
6
6
|
"dist",
|
|
@@ -18,8 +18,8 @@
|
|
|
18
18
|
"zod": "^4.4.3",
|
|
19
19
|
"@onco-foundry/capability-registry": "0.1.1",
|
|
20
20
|
"@onco-foundry/errors": "0.1.1",
|
|
21
|
-
"@onco-foundry/
|
|
22
|
-
"@onco-foundry/
|
|
21
|
+
"@onco-foundry/resource-versioning": "0.3.0",
|
|
22
|
+
"@onco-foundry/trace-port": "0.3.2"
|
|
23
23
|
},
|
|
24
24
|
"publishConfig": {
|
|
25
25
|
"access": "public"
|