@precisa-saude/fhir-ocr-utils 0.37.2 → 0.38.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/dist/cli.js +177 -31
- package/dist/index.cjs +177 -31
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +176 -30
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
package/dist/index.js
CHANGED
|
@@ -119,6 +119,34 @@ var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
|
|
|
119
119
|
"undetectable",
|
|
120
120
|
"yellow"
|
|
121
121
|
]);
|
|
122
|
+
var GENETIC_CONTEXT_PATTERNS = [
|
|
123
|
+
/\b[nx][mrpc]_\d{6,}/,
|
|
124
|
+
// RefSeq: NM_000384.2, NP_, NR_, XM_
|
|
125
|
+
/\bens[gtp]\d{6,}/,
|
|
126
|
+
// Ensembl: ENSG00000084674
|
|
127
|
+
/\bp\.[a-z]{3}\d/,
|
|
128
|
+
// HGVS proteína: p.Trp448*
|
|
129
|
+
/\bc\.\d+[acgt]?[>_+-]/,
|
|
130
|
+
// HGVS codificante: c.1234A>G, c.76_78del
|
|
131
|
+
/\brs\d{4,}\b/,
|
|
132
|
+
// dbSNP
|
|
133
|
+
/\bgenes?\b/,
|
|
134
|
+
/\bvariante?s?\b/,
|
|
135
|
+
/\bexons?\b/,
|
|
136
|
+
/\bzygosity\b/,
|
|
137
|
+
/\bzigosidade\b/,
|
|
138
|
+
/\balleles?\b/,
|
|
139
|
+
/\balelos?\b/,
|
|
140
|
+
/\bmutations?\b/,
|
|
141
|
+
/\bmutac(ao|oes)\b/,
|
|
142
|
+
/\bpathogenic/,
|
|
143
|
+
/\bpatogenic/,
|
|
144
|
+
/\bheterozyg/,
|
|
145
|
+
/\bhomozyg/,
|
|
146
|
+
/\bheterozigot/,
|
|
147
|
+
/\bhomozigot/,
|
|
148
|
+
/\bsequence change\b/
|
|
149
|
+
];
|
|
122
150
|
|
|
123
151
|
// src/body-region.ts
|
|
124
152
|
var WHOLE_BODY_COMPOSITION_CODES = /* @__PURE__ */ new Set([
|
|
@@ -235,6 +263,144 @@ function attachMethodVariants(matches, normalizedText, anchoredLineStarts, match
|
|
|
235
263
|
}
|
|
236
264
|
}
|
|
237
265
|
|
|
266
|
+
// src/urinalysis-section.ts
|
|
267
|
+
var HEADER_NAMES = String.raw`(?:urinalysis|urinalise|urina tipo i|rotina de urina|eas)(?![\p{L}\p{N}])`;
|
|
268
|
+
var URINALYSIS_HEADER = new RegExp(`^${HEADER_NAMES}`, "u");
|
|
269
|
+
var ANY_HEADER = new RegExp(String.raw`(?:^|\n)[^\S\n]*` + HEADER_NAMES, "u");
|
|
270
|
+
function mentionsUrinalysisHeader(normalizedText) {
|
|
271
|
+
return ANY_HEADER.test(normalizedText);
|
|
272
|
+
}
|
|
273
|
+
var URINALYSIS_SECTION_NAMES = /* @__PURE__ */ new Map([
|
|
274
|
+
["bacteria", "Bacteria_Urine"],
|
|
275
|
+
["bacterias", "Bacteria_Urine"],
|
|
276
|
+
["bilirrubina", "Bilirubin_Urine"],
|
|
277
|
+
["bilirubin", "Bilirubin_Urine"],
|
|
278
|
+
["cetonas", "Ketones_Urine"],
|
|
279
|
+
["color", "Color_Urine"],
|
|
280
|
+
["cor", "Color_Urine"],
|
|
281
|
+
["glicose", "Glucose_Urine"],
|
|
282
|
+
["glucose", "Glucose_Urine"],
|
|
283
|
+
["hemacias", "RBC_Urine"],
|
|
284
|
+
["ketones", "Ketones_Urine"],
|
|
285
|
+
["leucocitos", "Leukocytes_Urine"],
|
|
286
|
+
["nitrite", "Nitrite_Urine"],
|
|
287
|
+
["nitrito", "Nitrite_Urine"],
|
|
288
|
+
["ph", "pH_Urine"],
|
|
289
|
+
["protein", "Protein_Urine"],
|
|
290
|
+
["proteina", "Protein_Urine"],
|
|
291
|
+
["rbc", "RBC_Urine"],
|
|
292
|
+
["wbc", "Leukocytes_Urine"]
|
|
293
|
+
]);
|
|
294
|
+
function isSectionName(matched) {
|
|
295
|
+
return URINALYSIS_SECTION_NAMES.has(matched) || URINALYSIS_SECTION_NAMES.has(matched.replace(/s$/, ""));
|
|
296
|
+
}
|
|
297
|
+
var MENTIONS_URINE = /(?<![\p{L}\p{N}])urin[ae](?![\p{L}\p{N}])/u;
|
|
298
|
+
var SEDIMENT_FIELD_UNIT = /\/(?:hpf|lpf)(?![\p{L}\p{N}])/u;
|
|
299
|
+
function hasUrineCue(line) {
|
|
300
|
+
return MENTIONS_URINE.test(line) || SEDIMENT_FIELD_UNIT.test(line);
|
|
301
|
+
}
|
|
302
|
+
function kindOf(line) {
|
|
303
|
+
const text = line.text.trim();
|
|
304
|
+
if (!text) return "blank";
|
|
305
|
+
if (URINALYSIS_HEADER.test(text) && !/\d/.test(text)) return "header";
|
|
306
|
+
if (line.foreign) return "foreign";
|
|
307
|
+
if (line.urine) return "urine";
|
|
308
|
+
return line.known || line.hasValue ? "neutral" : "unknown";
|
|
309
|
+
}
|
|
310
|
+
function urinalysisLineIndexes(lines) {
|
|
311
|
+
const kinds = lines.map(kindOf);
|
|
312
|
+
const continuesAsUrine = (from) => {
|
|
313
|
+
const next = kinds.findIndex(
|
|
314
|
+
(kind, index) => index > from && (kind === "urine" || kind === "foreign" || kind === "header")
|
|
315
|
+
);
|
|
316
|
+
return next !== -1 && kinds[next] === "urine";
|
|
317
|
+
};
|
|
318
|
+
const inside = /* @__PURE__ */ new Set();
|
|
319
|
+
let open = false;
|
|
320
|
+
kinds.forEach((kind, index) => {
|
|
321
|
+
if (kind === "header") {
|
|
322
|
+
open = true;
|
|
323
|
+
return;
|
|
324
|
+
}
|
|
325
|
+
if (!open || kind === "blank") {
|
|
326
|
+
return;
|
|
327
|
+
}
|
|
328
|
+
if (kind === "foreign" || kind === "unknown" && !continuesAsUrine(index)) {
|
|
329
|
+
open = false;
|
|
330
|
+
return;
|
|
331
|
+
}
|
|
332
|
+
inside.add(index);
|
|
333
|
+
});
|
|
334
|
+
return inside;
|
|
335
|
+
}
|
|
336
|
+
function sectionDepsFrom(catalog, buildPattern, hasValue, resolve) {
|
|
337
|
+
const loincByCode = new Map(catalog.map((p) => [p.code, p.loinc]));
|
|
338
|
+
const isUrine = (p) => (Array.isArray(p.category) ? p.category : [p.category]).includes("urina");
|
|
339
|
+
return {
|
|
340
|
+
hasValue,
|
|
341
|
+
patterns: [...URINALYSIS_SECTION_NAMES].map(([name, code]) => {
|
|
342
|
+
const loinc = loincByCode.get(code);
|
|
343
|
+
const entry = { ambiguous: false, code, ...loinc && { loinc }, original: name };
|
|
344
|
+
return { entry, name, regex: buildPattern(name) };
|
|
345
|
+
}),
|
|
346
|
+
resolve,
|
|
347
|
+
urineCodes: new Set(catalog.filter(isUrine).map((p) => p.code))
|
|
348
|
+
};
|
|
349
|
+
}
|
|
350
|
+
function* matchesIn(text, name, regex) {
|
|
351
|
+
if (!text.includes(name)) return;
|
|
352
|
+
regex.lastIndex = 0;
|
|
353
|
+
for (let match = regex.exec(text); match !== null; match = regex.exec(text)) {
|
|
354
|
+
yield match;
|
|
355
|
+
}
|
|
356
|
+
}
|
|
357
|
+
function applyUrinalysisSection(normalizedText, candidates, deps) {
|
|
358
|
+
const resolved = deps.resolve(candidates);
|
|
359
|
+
if (!mentionsUrinalysisHeader(normalizedText)) {
|
|
360
|
+
return resolved;
|
|
361
|
+
}
|
|
362
|
+
const lineStarts = [0];
|
|
363
|
+
for (let i = normalizedText.indexOf("\n"); i !== -1; i = normalizedText.indexOf("\n", i + 1)) {
|
|
364
|
+
lineStarts.push(i + 1);
|
|
365
|
+
}
|
|
366
|
+
const lines = lineStarts.map((start, index) => {
|
|
367
|
+
const end = index + 1 < lineStarts.length ? lineStarts[index + 1] - 1 : normalizedText.length;
|
|
368
|
+
const text = normalizedText.slice(start, end);
|
|
369
|
+
const known = deps.patterns.some(
|
|
370
|
+
({ name, regex }) => !matchesIn(text, name, regex).next().done
|
|
371
|
+
);
|
|
372
|
+
return { foreign: false, hasValue: deps.hasValue(text), known, text, urine: hasUrineCue(text) };
|
|
373
|
+
});
|
|
374
|
+
for (const candidate of resolved) {
|
|
375
|
+
let index = lineStarts.length - 1;
|
|
376
|
+
while (lineStarts[index] > candidate.start) index -= 1;
|
|
377
|
+
const line = lines[index];
|
|
378
|
+
line.known = true;
|
|
379
|
+
const matched = normalizedText.slice(candidate.start, candidate.end);
|
|
380
|
+
if (candidate.entries.some((e) => deps.urineCodes.has(e.code))) {
|
|
381
|
+
line.urine = true;
|
|
382
|
+
} else if (!isSectionName(matched)) {
|
|
383
|
+
line.foreign = true;
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
const inside = urinalysisLineIndexes(lines);
|
|
387
|
+
if (!inside.size) {
|
|
388
|
+
return resolved;
|
|
389
|
+
}
|
|
390
|
+
const sectionCandidates = [];
|
|
391
|
+
for (const index of inside) {
|
|
392
|
+
const start = lineStarts[index];
|
|
393
|
+
for (const { entry, name, regex } of deps.patterns) {
|
|
394
|
+
for (const match of matchesIn(lines[index].text, name, regex)) {
|
|
395
|
+
const at = start + match.index;
|
|
396
|
+
sectionCandidates.push({ end: at + match[0].length, entries: [entry], start: at });
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
const replaced = (c) => sectionCandidates.some((s) => s.start === c.start && s.end === c.end);
|
|
401
|
+
return deps.resolve([...candidates.filter((c) => !replaced(c)), ...sectionCandidates]);
|
|
402
|
+
}
|
|
403
|
+
|
|
238
404
|
// src/anchor.ts
|
|
239
405
|
var CONFIDENCE_VALUE_ADJACENT = 1;
|
|
240
406
|
var CONFIDENCE_NAME_ONLY = 0.7;
|
|
@@ -286,40 +452,13 @@ function foldCommas(text) {
|
|
|
286
452
|
(comma, space, offset) => startsWithCatalogName(text.slice(offset + 1 + space.length)) ? comma : " "
|
|
287
453
|
);
|
|
288
454
|
}
|
|
289
|
-
var GENETIC_CONTEXT_PATTERNS = [
|
|
290
|
-
/\b[nx][mrpc]_\d{6,}/,
|
|
291
|
-
// RefSeq: NM_000384.2, NP_, NR_, XM_
|
|
292
|
-
/\bens[gtp]\d{6,}/,
|
|
293
|
-
// Ensembl: ENSG00000084674
|
|
294
|
-
/\bp\.[a-z]{3}\d/,
|
|
295
|
-
// HGVS proteína: p.Trp448*
|
|
296
|
-
/\bc\.\d+[acgt]?[>_+-]/,
|
|
297
|
-
// HGVS codificante: c.1234A>G, c.76_78del
|
|
298
|
-
/\brs\d{4,}\b/,
|
|
299
|
-
// dbSNP
|
|
300
|
-
/\bgenes?\b/,
|
|
301
|
-
/\bvariante?s?\b/,
|
|
302
|
-
/\bexons?\b/,
|
|
303
|
-
/\bzygosity\b/,
|
|
304
|
-
/\bzigosidade\b/,
|
|
305
|
-
/\balleles?\b/,
|
|
306
|
-
/\balelos?\b/,
|
|
307
|
-
/\bmutations?\b/,
|
|
308
|
-
/\bmutac(ao|oes)\b/,
|
|
309
|
-
/\bpathogenic/,
|
|
310
|
-
/\bpatogenic/,
|
|
311
|
-
/\bheterozyg/,
|
|
312
|
-
/\bhomozyg/,
|
|
313
|
-
/\bheterozigot/,
|
|
314
|
-
/\bhomozigot/,
|
|
315
|
-
/\bsequence change\b/
|
|
316
|
-
];
|
|
317
455
|
var DIGIT_PATTERN = /\d/;
|
|
318
456
|
var cachedUnitTokens = null;
|
|
319
457
|
function getUnitTokens() {
|
|
320
458
|
if (!cachedUnitTokens) {
|
|
459
|
+
const isExamName = (t) => URINALYSIS_SECTION_NAMES.has(t) || getNamePatterns().has(t);
|
|
321
460
|
cachedUnitTokens = new Set(
|
|
322
|
-
Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter(
|
|
461
|
+
Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter((unit) => unit && !isExamName(unit))
|
|
323
462
|
);
|
|
324
463
|
}
|
|
325
464
|
return cachedUnitTokens;
|
|
@@ -481,6 +620,13 @@ function resolveOverlaps(candidates) {
|
|
|
481
620
|
}
|
|
482
621
|
return accepted;
|
|
483
622
|
}
|
|
623
|
+
var cachedSectionDeps = null;
|
|
624
|
+
var getSectionDeps = () => cachedSectionDeps ??= sectionDepsFrom(
|
|
625
|
+
getPatterns(),
|
|
626
|
+
buildNamePattern,
|
|
627
|
+
hasValueEvidence,
|
|
628
|
+
resolveOverlaps
|
|
629
|
+
);
|
|
484
630
|
function findBiomarkersInText(ocrText) {
|
|
485
631
|
const startTime = Date.now();
|
|
486
632
|
const normalizedText = normalize(ocrText);
|
|
@@ -491,7 +637,7 @@ function findBiomarkersInText(ocrText) {
|
|
|
491
637
|
...collectCandidates(normalizedText),
|
|
492
638
|
...collectWrappedCandidates(normalizedText)
|
|
493
639
|
];
|
|
494
|
-
for (const candidate of
|
|
640
|
+
for (const candidate of applyUrinalysisSection(normalizedText, candidates, getSectionDeps())) {
|
|
495
641
|
const lineStart = getLineBounds(normalizedText, candidate.start).start;
|
|
496
642
|
const lineEnd = getLineBounds(normalizedText, candidate.end - 1).end;
|
|
497
643
|
const key = `${lineStart}:${lineEnd}`;
|