@onco-foundry/mask-port 0.1.3 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/textin_masker.d.ts +3 -1
- package/dist/textin_masker.js +78 -12
- package/package.json +4 -4
package/dist/textin_masker.d.ts
CHANGED
|
@@ -23,6 +23,8 @@ export type TextInMaskerOptions = {
|
|
|
23
23
|
readonly baseURL?: string;
|
|
24
24
|
readonly timeoutMs?: number;
|
|
25
25
|
readonly maxInputBytes?: number;
|
|
26
|
+
/** 自动处理链路使用:已发现的敏感词无法完整定位时拒绝交付。 */
|
|
27
|
+
readonly requireAllMatches?: boolean;
|
|
26
28
|
/** 测试注入缝:替换掉真实的 HTTP 层,不碰网络。 */
|
|
27
29
|
readonly fetch?: typeof fetch;
|
|
28
30
|
/**
|
|
@@ -35,6 +37,6 @@ export type TextInMaskerOptions = {
|
|
|
35
37
|
export declare const findSensitiveWordsByRules: (text: string) => readonly SensitiveWord[];
|
|
36
38
|
/**
|
|
37
39
|
* 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
|
|
38
|
-
*
|
|
40
|
+
* 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
|
|
39
41
|
*/
|
|
40
42
|
export declare const createTextInMasker: (options: TextInMaskerOptions) => Masker;
|
package/dist/textin_masker.js
CHANGED
|
@@ -82,9 +82,39 @@ const locateWordQuad = (line, beginIndex, wordLength) => {
|
|
|
82
82
|
return line.position === undefined ? undefined : toQuad(line.position);
|
|
83
83
|
};
|
|
84
84
|
const scaleQuadToImage = (quad, scaleX, scaleY) => quad.map(([x, y]) => [x * scaleX, y * scaleY]);
|
|
85
|
+
/**
|
|
86
|
+
* Find exact occurrences against the same newline-joined text seen by the finder,
|
|
87
|
+
* then split each occurrence into line-local spans for geometry lookup.
|
|
88
|
+
*/
|
|
89
|
+
const locateWordSpans = (lines, recognizedText, word) => {
|
|
90
|
+
const lineRanges = [];
|
|
91
|
+
let offset = 0;
|
|
92
|
+
for (const line of lines) {
|
|
93
|
+
lineRanges.push({ line, start: offset, end: offset + line.text.length });
|
|
94
|
+
offset += line.text.length + 1;
|
|
95
|
+
}
|
|
96
|
+
const occurrences = [];
|
|
97
|
+
let fromIndex = 0;
|
|
98
|
+
while (word.length > 0) {
|
|
99
|
+
const hit = recognizedText.indexOf(word, fromIndex);
|
|
100
|
+
if (hit < 0)
|
|
101
|
+
break;
|
|
102
|
+
fromIndex = hit + word.length;
|
|
103
|
+
const end = hit + word.length;
|
|
104
|
+
const spans = lineRanges.flatMap(range => {
|
|
105
|
+
const spanStart = Math.max(hit, range.start);
|
|
106
|
+
const spanEnd = Math.min(end, range.end);
|
|
107
|
+
return spanStart < spanEnd
|
|
108
|
+
? [{ line: range.line, beginIndex: spanStart - range.start, length: spanEnd - spanStart }]
|
|
109
|
+
: [];
|
|
110
|
+
});
|
|
111
|
+
occurrences.push(spans);
|
|
112
|
+
}
|
|
113
|
+
return occurrences;
|
|
114
|
+
};
|
|
85
115
|
/**
|
|
86
116
|
* 创建 TextIn 识别脱敏器。凭证在 create 时一次绑定,之后 mask 只传业务参数。
|
|
87
|
-
*
|
|
117
|
+
* 同一敏感词出现多次时逐处遮盖、逐处登记 mapping;跨行命中拆成多个框,仍登记为一个实体。
|
|
88
118
|
*/
|
|
89
119
|
export const createTextInMasker = (options) => {
|
|
90
120
|
const credentials = z.object({
|
|
@@ -156,28 +186,61 @@ export const createTextInMasker = (options) => {
|
|
|
156
186
|
if (payload.data.code !== 200 || payload.data.result === undefined) {
|
|
157
187
|
throw new AppError(`TextIn 脱敏接口报错:${payload.data.code} ${payload.data.message}`, 502);
|
|
158
188
|
}
|
|
189
|
+
if (options.requireAllMatches && payload.data.result.pages.length !== 1) {
|
|
190
|
+
throw new AppError('TextIn 脱敏需要单页图片识别结果', 502);
|
|
191
|
+
}
|
|
159
192
|
const page = payload.data.result.pages[0];
|
|
160
193
|
const lines = page.lines;
|
|
161
194
|
const recognizedText = lines.map((line) => line.text).join('\n');
|
|
162
195
|
const scaleX = page.width !== undefined && page.width > 0 ? imageWidth / page.width : 1;
|
|
163
196
|
const scaleY = page.height !== undefined && page.height > 0 ? imageHeight / page.height : 1;
|
|
164
|
-
|
|
197
|
+
if (options.requireAllMatches && !recognizedText.trim()) {
|
|
198
|
+
throw new AppError('TextIn 脱敏未识别出文字,无法确认处理结果', 502);
|
|
199
|
+
}
|
|
200
|
+
const detected = await finder(recognizedText);
|
|
201
|
+
if (options.requireAllMatches && detected.some(word => !word.text || !(word.target in TARGET_LABELS) || !recognizedText.includes(word.text))) {
|
|
202
|
+
throw new AppError('TextIn 脱敏判定结果无法定位', 502);
|
|
203
|
+
}
|
|
204
|
+
const requested = detected
|
|
165
205
|
.filter((word) => word.text !== '' && request.targets.includes(word.target));
|
|
166
206
|
const finalQuads = [];
|
|
167
207
|
const mappings = [];
|
|
168
208
|
const counters = new Map();
|
|
169
209
|
for (const word of requested) {
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
210
|
+
let located = 0;
|
|
211
|
+
for (const spans of locateWordSpans(lines, recognizedText, word.text)) {
|
|
212
|
+
const occurrenceQuads = [];
|
|
213
|
+
let complete = spans.length > 0;
|
|
214
|
+
for (const span of spans) {
|
|
215
|
+
const locatedLine = options.requireAllMatches && span.line.char_positions?.length !== span.line.text.length
|
|
216
|
+
? { ...span.line, char_positions: undefined } : span.line;
|
|
217
|
+
const quad = locateWordQuad(locatedLine, span.beginIndex, span.length);
|
|
218
|
+
if (quad === undefined) {
|
|
219
|
+
if (options.requireAllMatches)
|
|
220
|
+
throw new AppError('TextIn 脱敏敏感词缺少坐标', 502);
|
|
221
|
+
complete = false;
|
|
175
222
|
break;
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
223
|
+
}
|
|
224
|
+
if (options.requireAllMatches) {
|
|
225
|
+
const scaled = scaleQuadToImage(quad, scaleX, scaleY);
|
|
226
|
+
const xs = scaled.map(point => point[0]);
|
|
227
|
+
const ys = scaled.map(point => point[1]);
|
|
228
|
+
const area = Math.abs(scaled.reduce((sum, [x, y], index) => {
|
|
229
|
+
const next = scaled[(index + 1) % scaled.length];
|
|
230
|
+
return sum + x * next[1] - next[0] * y;
|
|
231
|
+
}, 0));
|
|
232
|
+
if (area < 1 || scaled.flat().some(n => !Number.isFinite(n)) ||
|
|
233
|
+
Math.min(...xs) < 0 || Math.min(...ys) < 0 ||
|
|
234
|
+
Math.max(...xs) > imageWidth || Math.max(...ys) > imageHeight ||
|
|
235
|
+
Math.max(...xs) <= Math.min(...xs) || Math.max(...ys) <= Math.min(...ys)) {
|
|
236
|
+
throw new AppError('TextIn 脱敏坐标无效', 502);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
occurrenceQuads.push(clampQuad(scaleQuad(scaleQuadToImage(quad, scaleX, scaleY), SAFETY_SCALE), imageWidth, imageHeight));
|
|
240
|
+
}
|
|
241
|
+
if (complete && occurrenceQuads.length > 0) {
|
|
242
|
+
located++;
|
|
243
|
+
finalQuads.push(...occurrenceQuads);
|
|
181
244
|
const count = (counters.get(word.target) ?? 0) + 1;
|
|
182
245
|
counters.set(word.target, count);
|
|
183
246
|
mappings.push({
|
|
@@ -187,6 +250,9 @@ export const createTextInMasker = (options) => {
|
|
|
187
250
|
});
|
|
188
251
|
}
|
|
189
252
|
}
|
|
253
|
+
if (options.requireAllMatches && located === 0) {
|
|
254
|
+
throw new AppError('TextIn 脱敏敏感词无法定位', 502);
|
|
255
|
+
}
|
|
190
256
|
}
|
|
191
257
|
const overlay = polygonSvg(imageWidth, imageHeight, finalQuads);
|
|
192
258
|
const uprightMasked = await sharp(upright.data)
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@onco-foundry/mask-port",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.5",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"files": [
|
|
6
6
|
"dist",
|
|
@@ -16,10 +16,10 @@
|
|
|
16
16
|
"agent-lattice": "0.25.0",
|
|
17
17
|
"sharp": "^0.35.3",
|
|
18
18
|
"zod": "^4.4.3",
|
|
19
|
-
"@onco-foundry/resource-versioning": "0.2.0",
|
|
20
|
-
"@onco-foundry/trace-port": "0.3.2",
|
|
21
19
|
"@onco-foundry/errors": "0.1.1",
|
|
22
|
-
"@onco-foundry/capability-registry": "0.1.1"
|
|
20
|
+
"@onco-foundry/capability-registry": "0.1.1",
|
|
21
|
+
"@onco-foundry/trace-port": "0.3.2",
|
|
22
|
+
"@onco-foundry/resource-versioning": "0.3.0"
|
|
23
23
|
},
|
|
24
24
|
"publishConfig": {
|
|
25
25
|
"access": "public"
|