@precisa-saude/fhir-ocr-utils 0.37.2 → 0.38.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/dist/cli.js +177 -31
- package/dist/index.cjs +177 -31
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +176 -30
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -83,6 +83,20 @@ Por isso a correspondência é conservadora:
|
|
|
83
83
|
a linha carregar um valor: um número, uma unidade conhecida, ou um termo
|
|
84
84
|
qualitativo esperado ("Negativo", "Ausente", "Amarelo Citrino"). "Specimen
|
|
85
85
|
type: Blood" não ancora; "Sangue Oculto: Negativo" ancora.
|
|
86
|
+
- **Seção de urinálise** — debaixo de um cabeçalho de urinálise
|
|
87
|
+
("URINALYSIS", "Urina tipo I", "Rotina de urina", "EAS", "Urinálise"), os
|
|
88
|
+
nomes nus do exame de urina passam a ser o analito da urina: "GLUCOSE"
|
|
89
|
+
ancora `Glucose_Urine`, "WBC" ancora `Leukocytes_Urine`, "RBC" ancora
|
|
90
|
+
`RBC_Urine`, "PH" ancora `pH_Urine`, e "COLOR", "KETONES", "PROTEIN" e
|
|
91
|
+
companhia ancoram mesmo sem valor na linha (texto em colunas). O código do
|
|
92
|
+
sangue não ancora pela mesma linha. A seção acaba na primeira linha com
|
|
93
|
+
exame de outro painel, no cabeçalho de outro painel ou no fim do texto. Uma
|
|
94
|
+
linha sem valor e sem nome conhecido pode ser as duas coisas ("COMPREHENSIVE
|
|
95
|
+
METABOLIC PANEL" ou "MUCUS"), e quem decide é a próxima linha decisiva: se só
|
|
96
|
+
pode ser da urina (código da urina, a palavra urina, ou /HPF e /LPF), a seção
|
|
97
|
+
segue; se é exame de outro painel, ou se o texto acaba sem nada decisivo, a
|
|
98
|
+
seção acaba ali, e o nome volta ao sentido de fora da seção. Fora dela nada
|
|
99
|
+
muda.
|
|
86
100
|
- **Contexto genético é descartado** — símbolos de gene colidem com nomes de
|
|
87
101
|
biomarcador (o gene `APOB` vs. a lipoproteína `ApoB`). Linhas com acesso
|
|
88
102
|
RefSeq (`NM_000384.2`), notação HGVS (`p.Trp448*`, `c.1234A>G`), `rs` do dbSNP
|
package/dist/cli.js
CHANGED
|
@@ -135,6 +135,34 @@ var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
|
|
|
135
135
|
"undetectable",
|
|
136
136
|
"yellow"
|
|
137
137
|
]);
|
|
138
|
+
var GENETIC_CONTEXT_PATTERNS = [
|
|
139
|
+
/\b[nx][mrpc]_\d{6,}/,
|
|
140
|
+
// RefSeq: NM_000384.2, NP_, NR_, XM_
|
|
141
|
+
/\bens[gtp]\d{6,}/,
|
|
142
|
+
// Ensembl: ENSG00000084674
|
|
143
|
+
/\bp\.[a-z]{3}\d/,
|
|
144
|
+
// HGVS proteína: p.Trp448*
|
|
145
|
+
/\bc\.\d+[acgt]?[>_+-]/,
|
|
146
|
+
// HGVS codificante: c.1234A>G, c.76_78del
|
|
147
|
+
/\brs\d{4,}\b/,
|
|
148
|
+
// dbSNP
|
|
149
|
+
/\bgenes?\b/,
|
|
150
|
+
/\bvariante?s?\b/,
|
|
151
|
+
/\bexons?\b/,
|
|
152
|
+
/\bzygosity\b/,
|
|
153
|
+
/\bzigosidade\b/,
|
|
154
|
+
/\balleles?\b/,
|
|
155
|
+
/\balelos?\b/,
|
|
156
|
+
/\bmutations?\b/,
|
|
157
|
+
/\bmutac(ao|oes)\b/,
|
|
158
|
+
/\bpathogenic/,
|
|
159
|
+
/\bpatogenic/,
|
|
160
|
+
/\bheterozyg/,
|
|
161
|
+
/\bhomozyg/,
|
|
162
|
+
/\bheterozigot/,
|
|
163
|
+
/\bhomozigot/,
|
|
164
|
+
/\bsequence change\b/
|
|
165
|
+
];
|
|
138
166
|
|
|
139
167
|
// src/body-region.ts
|
|
140
168
|
var WHOLE_BODY_COMPOSITION_CODES = /* @__PURE__ */ new Set([
|
|
@@ -251,6 +279,144 @@ function attachMethodVariants(matches, normalizedText, anchoredLineStarts, match
|
|
|
251
279
|
}
|
|
252
280
|
}
|
|
253
281
|
|
|
282
|
+
// src/urinalysis-section.ts
|
|
283
|
+
var HEADER_NAMES = String.raw`(?:urinalysis|urinalise|urina tipo i|rotina de urina|eas)(?![\p{L}\p{N}])`;
|
|
284
|
+
var URINALYSIS_HEADER = new RegExp(`^${HEADER_NAMES}`, "u");
|
|
285
|
+
var ANY_HEADER = new RegExp(String.raw`(?:^|\n)[^\S\n]*` + HEADER_NAMES, "u");
|
|
286
|
+
function mentionsUrinalysisHeader(normalizedText) {
|
|
287
|
+
return ANY_HEADER.test(normalizedText);
|
|
288
|
+
}
|
|
289
|
+
var URINALYSIS_SECTION_NAMES = /* @__PURE__ */ new Map([
|
|
290
|
+
["bacteria", "Bacteria_Urine"],
|
|
291
|
+
["bacterias", "Bacteria_Urine"],
|
|
292
|
+
["bilirrubina", "Bilirubin_Urine"],
|
|
293
|
+
["bilirubin", "Bilirubin_Urine"],
|
|
294
|
+
["cetonas", "Ketones_Urine"],
|
|
295
|
+
["color", "Color_Urine"],
|
|
296
|
+
["cor", "Color_Urine"],
|
|
297
|
+
["glicose", "Glucose_Urine"],
|
|
298
|
+
["glucose", "Glucose_Urine"],
|
|
299
|
+
["hemacias", "RBC_Urine"],
|
|
300
|
+
["ketones", "Ketones_Urine"],
|
|
301
|
+
["leucocitos", "Leukocytes_Urine"],
|
|
302
|
+
["nitrite", "Nitrite_Urine"],
|
|
303
|
+
["nitrito", "Nitrite_Urine"],
|
|
304
|
+
["ph", "pH_Urine"],
|
|
305
|
+
["protein", "Protein_Urine"],
|
|
306
|
+
["proteina", "Protein_Urine"],
|
|
307
|
+
["rbc", "RBC_Urine"],
|
|
308
|
+
["wbc", "Leukocytes_Urine"]
|
|
309
|
+
]);
|
|
310
|
+
function isSectionName(matched) {
|
|
311
|
+
return URINALYSIS_SECTION_NAMES.has(matched) || URINALYSIS_SECTION_NAMES.has(matched.replace(/s$/, ""));
|
|
312
|
+
}
|
|
313
|
+
var MENTIONS_URINE = /(?<![\p{L}\p{N}])urin[ae](?![\p{L}\p{N}])/u;
|
|
314
|
+
var SEDIMENT_FIELD_UNIT = /\/(?:hpf|lpf)(?![\p{L}\p{N}])/u;
|
|
315
|
+
function hasUrineCue(line) {
|
|
316
|
+
return MENTIONS_URINE.test(line) || SEDIMENT_FIELD_UNIT.test(line);
|
|
317
|
+
}
|
|
318
|
+
function kindOf(line) {
|
|
319
|
+
const text = line.text.trim();
|
|
320
|
+
if (!text) return "blank";
|
|
321
|
+
if (URINALYSIS_HEADER.test(text) && !/\d/.test(text)) return "header";
|
|
322
|
+
if (line.foreign) return "foreign";
|
|
323
|
+
if (line.urine) return "urine";
|
|
324
|
+
return line.known || line.hasValue ? "neutral" : "unknown";
|
|
325
|
+
}
|
|
326
|
+
function urinalysisLineIndexes(lines) {
|
|
327
|
+
const kinds = lines.map(kindOf);
|
|
328
|
+
const continuesAsUrine = (from) => {
|
|
329
|
+
const next = kinds.findIndex(
|
|
330
|
+
(kind, index) => index > from && (kind === "urine" || kind === "foreign" || kind === "header")
|
|
331
|
+
);
|
|
332
|
+
return next !== -1 && kinds[next] === "urine";
|
|
333
|
+
};
|
|
334
|
+
const inside = /* @__PURE__ */ new Set();
|
|
335
|
+
let open = false;
|
|
336
|
+
kinds.forEach((kind, index) => {
|
|
337
|
+
if (kind === "header") {
|
|
338
|
+
open = true;
|
|
339
|
+
return;
|
|
340
|
+
}
|
|
341
|
+
if (!open || kind === "blank") {
|
|
342
|
+
return;
|
|
343
|
+
}
|
|
344
|
+
if (kind === "foreign" || kind === "unknown" && !continuesAsUrine(index)) {
|
|
345
|
+
open = false;
|
|
346
|
+
return;
|
|
347
|
+
}
|
|
348
|
+
inside.add(index);
|
|
349
|
+
});
|
|
350
|
+
return inside;
|
|
351
|
+
}
|
|
352
|
+
function sectionDepsFrom(catalog, buildPattern, hasValue, resolve) {
|
|
353
|
+
const loincByCode = new Map(catalog.map((p) => [p.code, p.loinc]));
|
|
354
|
+
const isUrine = (p) => (Array.isArray(p.category) ? p.category : [p.category]).includes("urina");
|
|
355
|
+
return {
|
|
356
|
+
hasValue,
|
|
357
|
+
patterns: [...URINALYSIS_SECTION_NAMES].map(([name, code]) => {
|
|
358
|
+
const loinc = loincByCode.get(code);
|
|
359
|
+
const entry = { ambiguous: false, code, ...loinc && { loinc }, original: name };
|
|
360
|
+
return { entry, name, regex: buildPattern(name) };
|
|
361
|
+
}),
|
|
362
|
+
resolve,
|
|
363
|
+
urineCodes: new Set(catalog.filter(isUrine).map((p) => p.code))
|
|
364
|
+
};
|
|
365
|
+
}
|
|
366
|
+
function* matchesIn(text, name, regex) {
|
|
367
|
+
if (!text.includes(name)) return;
|
|
368
|
+
regex.lastIndex = 0;
|
|
369
|
+
for (let match = regex.exec(text); match !== null; match = regex.exec(text)) {
|
|
370
|
+
yield match;
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
function applyUrinalysisSection(normalizedText, candidates, deps) {
|
|
374
|
+
const resolved = deps.resolve(candidates);
|
|
375
|
+
if (!mentionsUrinalysisHeader(normalizedText)) {
|
|
376
|
+
return resolved;
|
|
377
|
+
}
|
|
378
|
+
const lineStarts = [0];
|
|
379
|
+
for (let i = normalizedText.indexOf("\n"); i !== -1; i = normalizedText.indexOf("\n", i + 1)) {
|
|
380
|
+
lineStarts.push(i + 1);
|
|
381
|
+
}
|
|
382
|
+
const lines = lineStarts.map((start, index) => {
|
|
383
|
+
const end = index + 1 < lineStarts.length ? lineStarts[index + 1] - 1 : normalizedText.length;
|
|
384
|
+
const text = normalizedText.slice(start, end);
|
|
385
|
+
const known = deps.patterns.some(
|
|
386
|
+
({ name, regex }) => !matchesIn(text, name, regex).next().done
|
|
387
|
+
);
|
|
388
|
+
return { foreign: false, hasValue: deps.hasValue(text), known, text, urine: hasUrineCue(text) };
|
|
389
|
+
});
|
|
390
|
+
for (const candidate of resolved) {
|
|
391
|
+
let index = lineStarts.length - 1;
|
|
392
|
+
while (lineStarts[index] > candidate.start) index -= 1;
|
|
393
|
+
const line = lines[index];
|
|
394
|
+
line.known = true;
|
|
395
|
+
const matched = normalizedText.slice(candidate.start, candidate.end);
|
|
396
|
+
if (candidate.entries.some((e) => deps.urineCodes.has(e.code))) {
|
|
397
|
+
line.urine = true;
|
|
398
|
+
} else if (!isSectionName(matched)) {
|
|
399
|
+
line.foreign = true;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
const inside = urinalysisLineIndexes(lines);
|
|
403
|
+
if (!inside.size) {
|
|
404
|
+
return resolved;
|
|
405
|
+
}
|
|
406
|
+
const sectionCandidates = [];
|
|
407
|
+
for (const index of inside) {
|
|
408
|
+
const start = lineStarts[index];
|
|
409
|
+
for (const { entry, name, regex } of deps.patterns) {
|
|
410
|
+
for (const match of matchesIn(lines[index].text, name, regex)) {
|
|
411
|
+
const at = start + match.index;
|
|
412
|
+
sectionCandidates.push({ end: at + match[0].length, entries: [entry], start: at });
|
|
413
|
+
}
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
const replaced = (c) => sectionCandidates.some((s) => s.start === c.start && s.end === c.end);
|
|
417
|
+
return deps.resolve([...candidates.filter((c) => !replaced(c)), ...sectionCandidates]);
|
|
418
|
+
}
|
|
419
|
+
|
|
254
420
|
// src/anchor.ts
|
|
255
421
|
var CONFIDENCE_VALUE_ADJACENT = 1;
|
|
256
422
|
var CONFIDENCE_NAME_ONLY = 0.7;
|
|
@@ -302,40 +468,13 @@ function foldCommas(text) {
|
|
|
302
468
|
(comma, space, offset) => startsWithCatalogName(text.slice(offset + 1 + space.length)) ? comma : " "
|
|
303
469
|
);
|
|
304
470
|
}
|
|
305
|
-
var GENETIC_CONTEXT_PATTERNS = [
|
|
306
|
-
/\b[nx][mrpc]_\d{6,}/,
|
|
307
|
-
// RefSeq: NM_000384.2, NP_, NR_, XM_
|
|
308
|
-
/\bens[gtp]\d{6,}/,
|
|
309
|
-
// Ensembl: ENSG00000084674
|
|
310
|
-
/\bp\.[a-z]{3}\d/,
|
|
311
|
-
// HGVS proteína: p.Trp448*
|
|
312
|
-
/\bc\.\d+[acgt]?[>_+-]/,
|
|
313
|
-
// HGVS codificante: c.1234A>G, c.76_78del
|
|
314
|
-
/\brs\d{4,}\b/,
|
|
315
|
-
// dbSNP
|
|
316
|
-
/\bgenes?\b/,
|
|
317
|
-
/\bvariante?s?\b/,
|
|
318
|
-
/\bexons?\b/,
|
|
319
|
-
/\bzygosity\b/,
|
|
320
|
-
/\bzigosidade\b/,
|
|
321
|
-
/\balleles?\b/,
|
|
322
|
-
/\balelos?\b/,
|
|
323
|
-
/\bmutations?\b/,
|
|
324
|
-
/\bmutac(ao|oes)\b/,
|
|
325
|
-
/\bpathogenic/,
|
|
326
|
-
/\bpatogenic/,
|
|
327
|
-
/\bheterozyg/,
|
|
328
|
-
/\bhomozyg/,
|
|
329
|
-
/\bheterozigot/,
|
|
330
|
-
/\bhomozigot/,
|
|
331
|
-
/\bsequence change\b/
|
|
332
|
-
];
|
|
333
471
|
var DIGIT_PATTERN = /\d/;
|
|
334
472
|
var cachedUnitTokens = null;
|
|
335
473
|
function getUnitTokens() {
|
|
336
474
|
if (!cachedUnitTokens) {
|
|
475
|
+
const isExamName = (t) => URINALYSIS_SECTION_NAMES.has(t) || getNamePatterns().has(t);
|
|
337
476
|
cachedUnitTokens = new Set(
|
|
338
|
-
Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter(
|
|
477
|
+
Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter((unit) => unit && !isExamName(unit))
|
|
339
478
|
);
|
|
340
479
|
}
|
|
341
480
|
return cachedUnitTokens;
|
|
@@ -497,6 +636,13 @@ function resolveOverlaps(candidates) {
|
|
|
497
636
|
}
|
|
498
637
|
return accepted;
|
|
499
638
|
}
|
|
639
|
+
var cachedSectionDeps = null;
|
|
640
|
+
var getSectionDeps = () => cachedSectionDeps ??= sectionDepsFrom(
|
|
641
|
+
getPatterns(),
|
|
642
|
+
buildNamePattern,
|
|
643
|
+
hasValueEvidence,
|
|
644
|
+
resolveOverlaps
|
|
645
|
+
);
|
|
500
646
|
function findBiomarkersInText(ocrText) {
|
|
501
647
|
const startTime = Date.now();
|
|
502
648
|
const normalizedText = normalize(ocrText);
|
|
@@ -507,7 +653,7 @@ function findBiomarkersInText(ocrText) {
|
|
|
507
653
|
...collectCandidates(normalizedText),
|
|
508
654
|
...collectWrappedCandidates(normalizedText)
|
|
509
655
|
];
|
|
510
|
-
for (const candidate of
|
|
656
|
+
for (const candidate of applyUrinalysisSection(normalizedText, candidates, getSectionDeps())) {
|
|
511
657
|
const lineStart = getLineBounds(normalizedText, candidate.start).start;
|
|
512
658
|
const lineEnd = getLineBounds(normalizedText, candidate.end - 1).end;
|
|
513
659
|
const key = `${lineStart}:${lineEnd}`;
|
|
@@ -1093,7 +1239,7 @@ async function main() {
|
|
|
1093
1239
|
strict: false
|
|
1094
1240
|
});
|
|
1095
1241
|
if (values.version) {
|
|
1096
|
-
process.stdout.write(`${"0.
|
|
1242
|
+
process.stdout.write(`${"0.38.1"}
|
|
1097
1243
|
`);
|
|
1098
1244
|
return;
|
|
1099
1245
|
}
|
package/dist/index.cjs
CHANGED
|
@@ -119,6 +119,34 @@ var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
|
|
|
119
119
|
"undetectable",
|
|
120
120
|
"yellow"
|
|
121
121
|
]);
|
|
122
|
+
var GENETIC_CONTEXT_PATTERNS = [
|
|
123
|
+
/\b[nx][mrpc]_\d{6,}/,
|
|
124
|
+
// RefSeq: NM_000384.2, NP_, NR_, XM_
|
|
125
|
+
/\bens[gtp]\d{6,}/,
|
|
126
|
+
// Ensembl: ENSG00000084674
|
|
127
|
+
/\bp\.[a-z]{3}\d/,
|
|
128
|
+
// HGVS proteína: p.Trp448*
|
|
129
|
+
/\bc\.\d+[acgt]?[>_+-]/,
|
|
130
|
+
// HGVS codificante: c.1234A>G, c.76_78del
|
|
131
|
+
/\brs\d{4,}\b/,
|
|
132
|
+
// dbSNP
|
|
133
|
+
/\bgenes?\b/,
|
|
134
|
+
/\bvariante?s?\b/,
|
|
135
|
+
/\bexons?\b/,
|
|
136
|
+
/\bzygosity\b/,
|
|
137
|
+
/\bzigosidade\b/,
|
|
138
|
+
/\balleles?\b/,
|
|
139
|
+
/\balelos?\b/,
|
|
140
|
+
/\bmutations?\b/,
|
|
141
|
+
/\bmutac(ao|oes)\b/,
|
|
142
|
+
/\bpathogenic/,
|
|
143
|
+
/\bpatogenic/,
|
|
144
|
+
/\bheterozyg/,
|
|
145
|
+
/\bhomozyg/,
|
|
146
|
+
/\bheterozigot/,
|
|
147
|
+
/\bhomozigot/,
|
|
148
|
+
/\bsequence change\b/
|
|
149
|
+
];
|
|
122
150
|
|
|
123
151
|
// src/body-region.ts
|
|
124
152
|
var WHOLE_BODY_COMPOSITION_CODES = /* @__PURE__ */ new Set([
|
|
@@ -235,6 +263,144 @@ function attachMethodVariants(matches, normalizedText, anchoredLineStarts, match
|
|
|
235
263
|
}
|
|
236
264
|
}
|
|
237
265
|
|
|
266
|
+
// src/urinalysis-section.ts
|
|
267
|
+
var HEADER_NAMES = String.raw`(?:urinalysis|urinalise|urina tipo i|rotina de urina|eas)(?![\p{L}\p{N}])`;
|
|
268
|
+
var URINALYSIS_HEADER = new RegExp(`^${HEADER_NAMES}`, "u");
|
|
269
|
+
var ANY_HEADER = new RegExp(String.raw`(?:^|\n)[^\S\n]*` + HEADER_NAMES, "u");
|
|
270
|
+
function mentionsUrinalysisHeader(normalizedText) {
|
|
271
|
+
return ANY_HEADER.test(normalizedText);
|
|
272
|
+
}
|
|
273
|
+
var URINALYSIS_SECTION_NAMES = /* @__PURE__ */ new Map([
|
|
274
|
+
["bacteria", "Bacteria_Urine"],
|
|
275
|
+
["bacterias", "Bacteria_Urine"],
|
|
276
|
+
["bilirrubina", "Bilirubin_Urine"],
|
|
277
|
+
["bilirubin", "Bilirubin_Urine"],
|
|
278
|
+
["cetonas", "Ketones_Urine"],
|
|
279
|
+
["color", "Color_Urine"],
|
|
280
|
+
["cor", "Color_Urine"],
|
|
281
|
+
["glicose", "Glucose_Urine"],
|
|
282
|
+
["glucose", "Glucose_Urine"],
|
|
283
|
+
["hemacias", "RBC_Urine"],
|
|
284
|
+
["ketones", "Ketones_Urine"],
|
|
285
|
+
["leucocitos", "Leukocytes_Urine"],
|
|
286
|
+
["nitrite", "Nitrite_Urine"],
|
|
287
|
+
["nitrito", "Nitrite_Urine"],
|
|
288
|
+
["ph", "pH_Urine"],
|
|
289
|
+
["protein", "Protein_Urine"],
|
|
290
|
+
["proteina", "Protein_Urine"],
|
|
291
|
+
["rbc", "RBC_Urine"],
|
|
292
|
+
["wbc", "Leukocytes_Urine"]
|
|
293
|
+
]);
|
|
294
|
+
function isSectionName(matched) {
|
|
295
|
+
return URINALYSIS_SECTION_NAMES.has(matched) || URINALYSIS_SECTION_NAMES.has(matched.replace(/s$/, ""));
|
|
296
|
+
}
|
|
297
|
+
var MENTIONS_URINE = /(?<![\p{L}\p{N}])urin[ae](?![\p{L}\p{N}])/u;
|
|
298
|
+
var SEDIMENT_FIELD_UNIT = /\/(?:hpf|lpf)(?![\p{L}\p{N}])/u;
|
|
299
|
+
function hasUrineCue(line) {
|
|
300
|
+
return MENTIONS_URINE.test(line) || SEDIMENT_FIELD_UNIT.test(line);
|
|
301
|
+
}
|
|
302
|
+
function kindOf(line) {
|
|
303
|
+
const text = line.text.trim();
|
|
304
|
+
if (!text) return "blank";
|
|
305
|
+
if (URINALYSIS_HEADER.test(text) && !/\d/.test(text)) return "header";
|
|
306
|
+
if (line.foreign) return "foreign";
|
|
307
|
+
if (line.urine) return "urine";
|
|
308
|
+
return line.known || line.hasValue ? "neutral" : "unknown";
|
|
309
|
+
}
|
|
310
|
+
function urinalysisLineIndexes(lines) {
|
|
311
|
+
const kinds = lines.map(kindOf);
|
|
312
|
+
const continuesAsUrine = (from) => {
|
|
313
|
+
const next = kinds.findIndex(
|
|
314
|
+
(kind, index) => index > from && (kind === "urine" || kind === "foreign" || kind === "header")
|
|
315
|
+
);
|
|
316
|
+
return next !== -1 && kinds[next] === "urine";
|
|
317
|
+
};
|
|
318
|
+
const inside = /* @__PURE__ */ new Set();
|
|
319
|
+
let open = false;
|
|
320
|
+
kinds.forEach((kind, index) => {
|
|
321
|
+
if (kind === "header") {
|
|
322
|
+
open = true;
|
|
323
|
+
return;
|
|
324
|
+
}
|
|
325
|
+
if (!open || kind === "blank") {
|
|
326
|
+
return;
|
|
327
|
+
}
|
|
328
|
+
if (kind === "foreign" || kind === "unknown" && !continuesAsUrine(index)) {
|
|
329
|
+
open = false;
|
|
330
|
+
return;
|
|
331
|
+
}
|
|
332
|
+
inside.add(index);
|
|
333
|
+
});
|
|
334
|
+
return inside;
|
|
335
|
+
}
|
|
336
|
+
function sectionDepsFrom(catalog, buildPattern, hasValue, resolve) {
|
|
337
|
+
const loincByCode = new Map(catalog.map((p) => [p.code, p.loinc]));
|
|
338
|
+
const isUrine = (p) => (Array.isArray(p.category) ? p.category : [p.category]).includes("urina");
|
|
339
|
+
return {
|
|
340
|
+
hasValue,
|
|
341
|
+
patterns: [...URINALYSIS_SECTION_NAMES].map(([name, code]) => {
|
|
342
|
+
const loinc = loincByCode.get(code);
|
|
343
|
+
const entry = { ambiguous: false, code, ...loinc && { loinc }, original: name };
|
|
344
|
+
return { entry, name, regex: buildPattern(name) };
|
|
345
|
+
}),
|
|
346
|
+
resolve,
|
|
347
|
+
urineCodes: new Set(catalog.filter(isUrine).map((p) => p.code))
|
|
348
|
+
};
|
|
349
|
+
}
|
|
350
|
+
function* matchesIn(text, name, regex) {
|
|
351
|
+
if (!text.includes(name)) return;
|
|
352
|
+
regex.lastIndex = 0;
|
|
353
|
+
for (let match = regex.exec(text); match !== null; match = regex.exec(text)) {
|
|
354
|
+
yield match;
|
|
355
|
+
}
|
|
356
|
+
}
|
|
357
|
+
function applyUrinalysisSection(normalizedText, candidates, deps) {
|
|
358
|
+
const resolved = deps.resolve(candidates);
|
|
359
|
+
if (!mentionsUrinalysisHeader(normalizedText)) {
|
|
360
|
+
return resolved;
|
|
361
|
+
}
|
|
362
|
+
const lineStarts = [0];
|
|
363
|
+
for (let i = normalizedText.indexOf("\n"); i !== -1; i = normalizedText.indexOf("\n", i + 1)) {
|
|
364
|
+
lineStarts.push(i + 1);
|
|
365
|
+
}
|
|
366
|
+
const lines = lineStarts.map((start, index) => {
|
|
367
|
+
const end = index + 1 < lineStarts.length ? lineStarts[index + 1] - 1 : normalizedText.length;
|
|
368
|
+
const text = normalizedText.slice(start, end);
|
|
369
|
+
const known = deps.patterns.some(
|
|
370
|
+
({ name, regex }) => !matchesIn(text, name, regex).next().done
|
|
371
|
+
);
|
|
372
|
+
return { foreign: false, hasValue: deps.hasValue(text), known, text, urine: hasUrineCue(text) };
|
|
373
|
+
});
|
|
374
|
+
for (const candidate of resolved) {
|
|
375
|
+
let index = lineStarts.length - 1;
|
|
376
|
+
while (lineStarts[index] > candidate.start) index -= 1;
|
|
377
|
+
const line = lines[index];
|
|
378
|
+
line.known = true;
|
|
379
|
+
const matched = normalizedText.slice(candidate.start, candidate.end);
|
|
380
|
+
if (candidate.entries.some((e) => deps.urineCodes.has(e.code))) {
|
|
381
|
+
line.urine = true;
|
|
382
|
+
} else if (!isSectionName(matched)) {
|
|
383
|
+
line.foreign = true;
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
const inside = urinalysisLineIndexes(lines);
|
|
387
|
+
if (!inside.size) {
|
|
388
|
+
return resolved;
|
|
389
|
+
}
|
|
390
|
+
const sectionCandidates = [];
|
|
391
|
+
for (const index of inside) {
|
|
392
|
+
const start = lineStarts[index];
|
|
393
|
+
for (const { entry, name, regex } of deps.patterns) {
|
|
394
|
+
for (const match of matchesIn(lines[index].text, name, regex)) {
|
|
395
|
+
const at = start + match.index;
|
|
396
|
+
sectionCandidates.push({ end: at + match[0].length, entries: [entry], start: at });
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
const replaced = (c) => sectionCandidates.some((s) => s.start === c.start && s.end === c.end);
|
|
401
|
+
return deps.resolve([...candidates.filter((c) => !replaced(c)), ...sectionCandidates]);
|
|
402
|
+
}
|
|
403
|
+
|
|
238
404
|
// src/anchor.ts
|
|
239
405
|
var CONFIDENCE_VALUE_ADJACENT = 1;
|
|
240
406
|
var CONFIDENCE_NAME_ONLY = 0.7;
|
|
@@ -286,40 +452,13 @@ function foldCommas(text) {
|
|
|
286
452
|
(comma, space, offset) => startsWithCatalogName(text.slice(offset + 1 + space.length)) ? comma : " "
|
|
287
453
|
);
|
|
288
454
|
}
|
|
289
|
-
var GENETIC_CONTEXT_PATTERNS = [
|
|
290
|
-
/\b[nx][mrpc]_\d{6,}/,
|
|
291
|
-
// RefSeq: NM_000384.2, NP_, NR_, XM_
|
|
292
|
-
/\bens[gtp]\d{6,}/,
|
|
293
|
-
// Ensembl: ENSG00000084674
|
|
294
|
-
/\bp\.[a-z]{3}\d/,
|
|
295
|
-
// HGVS proteína: p.Trp448*
|
|
296
|
-
/\bc\.\d+[acgt]?[>_+-]/,
|
|
297
|
-
// HGVS codificante: c.1234A>G, c.76_78del
|
|
298
|
-
/\brs\d{4,}\b/,
|
|
299
|
-
// dbSNP
|
|
300
|
-
/\bgenes?\b/,
|
|
301
|
-
/\bvariante?s?\b/,
|
|
302
|
-
/\bexons?\b/,
|
|
303
|
-
/\bzygosity\b/,
|
|
304
|
-
/\bzigosidade\b/,
|
|
305
|
-
/\balleles?\b/,
|
|
306
|
-
/\balelos?\b/,
|
|
307
|
-
/\bmutations?\b/,
|
|
308
|
-
/\bmutac(ao|oes)\b/,
|
|
309
|
-
/\bpathogenic/,
|
|
310
|
-
/\bpatogenic/,
|
|
311
|
-
/\bheterozyg/,
|
|
312
|
-
/\bhomozyg/,
|
|
313
|
-
/\bheterozigot/,
|
|
314
|
-
/\bhomozigot/,
|
|
315
|
-
/\bsequence change\b/
|
|
316
|
-
];
|
|
317
455
|
var DIGIT_PATTERN = /\d/;
|
|
318
456
|
var cachedUnitTokens = null;
|
|
319
457
|
function getUnitTokens() {
|
|
320
458
|
if (!cachedUnitTokens) {
|
|
459
|
+
const isExamName = (t) => URINALYSIS_SECTION_NAMES.has(t) || getNamePatterns().has(t);
|
|
321
460
|
cachedUnitTokens = new Set(
|
|
322
|
-
Object.keys(_fhir.UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter(
|
|
461
|
+
Object.keys(_fhir.UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter((unit) => unit && !isExamName(unit))
|
|
323
462
|
);
|
|
324
463
|
}
|
|
325
464
|
return cachedUnitTokens;
|
|
@@ -481,6 +620,13 @@ function resolveOverlaps(candidates) {
|
|
|
481
620
|
}
|
|
482
621
|
return accepted;
|
|
483
622
|
}
|
|
623
|
+
var cachedSectionDeps = null;
|
|
624
|
+
var getSectionDeps = () => cachedSectionDeps ??= sectionDepsFrom(
|
|
625
|
+
getPatterns(),
|
|
626
|
+
buildNamePattern,
|
|
627
|
+
hasValueEvidence,
|
|
628
|
+
resolveOverlaps
|
|
629
|
+
);
|
|
484
630
|
function findBiomarkersInText(ocrText) {
|
|
485
631
|
const startTime = Date.now();
|
|
486
632
|
const normalizedText = normalize(ocrText);
|
|
@@ -491,7 +637,7 @@ function findBiomarkersInText(ocrText) {
|
|
|
491
637
|
...collectCandidates(normalizedText),
|
|
492
638
|
...collectWrappedCandidates(normalizedText)
|
|
493
639
|
];
|
|
494
|
-
for (const candidate of
|
|
640
|
+
for (const candidate of applyUrinalysisSection(normalizedText, candidates, getSectionDeps())) {
|
|
495
641
|
const lineStart = getLineBounds(normalizedText, candidate.start).start;
|
|
496
642
|
const lineEnd = getLineBounds(normalizedText, candidate.end - 1).end;
|
|
497
643
|
const key = `${lineStart}:${lineEnd}`;
|
|
@@ -881,7 +1027,7 @@ async function extractWithModel(text, options) {
|
|
|
881
1027
|
const tookMs = Date.now() - startedAt;
|
|
882
1028
|
try {
|
|
883
1029
|
return { payload: JSON.parse(stripFence(raw)), raw, tookMs };
|
|
884
|
-
} catch (
|
|
1030
|
+
} catch (e2) {
|
|
885
1031
|
throw new Error(`O modelo n\xE3o devolveu JSON analis\xE1vel. Resposta crua:
|
|
886
1032
|
${raw.slice(0, 500)}`);
|
|
887
1033
|
}
|