@precisa-saude/fhir-ocr-utils 0.37.2 → 0.38.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -83,6 +83,16 @@ Por isso a correspondência é conservadora:
83
83
  a linha carregar um valor: um número, uma unidade conhecida, ou um termo
84
84
  qualitativo esperado ("Negativo", "Ausente", "Amarelo Citrino"). "Specimen
85
85
  type: Blood" não ancora; "Sangue Oculto: Negativo" ancora.
86
+ - **Seção de urinálise** — debaixo de um cabeçalho de urinálise
87
+ ("URINALYSIS", "Urina tipo I", "Rotina de urina", "EAS", "Urinálise"), os
88
+ nomes nus do exame de urina passam a ser o analito da urina: "GLUCOSE"
89
+ ancora `Glucose_Urine`, "WBC" ancora `Leukocytes_Urine`, "RBC" ancora
90
+ `RBC_Urine`, "PH" ancora `pH_Urine`, e "COLOR", "KETONES", "PROTEIN" e
91
+ companhia ancoram mesmo sem valor na linha (texto em colunas). O código do
92
+ sangue não ancora pela mesma linha. A seção acaba na primeira linha com
93
+ exame de outro painel ou na primeira linha sem valor e sem nome conhecido
94
+ (o jeito de um cabeçalho); na dúvida ela acaba, e o nome volta ao sentido de
95
+ fora da seção. Fora dela nada muda.
86
96
  - **Contexto genético é descartado** — símbolos de gene colidem com nomes de
87
97
  biomarcador (o gene `APOB` vs. a lipoproteína `ApoB`). Linhas com acesso
88
98
  RefSeq (`NM_000384.2`), notação HGVS (`p.Trp448*`, `c.1234A>G`), `rs` do dbSNP
package/dist/cli.js CHANGED
@@ -251,6 +251,124 @@ function attachMethodVariants(matches, normalizedText, anchoredLineStarts, match
251
251
  }
252
252
  }
253
253
 
254
+ // src/urinalysis-section.ts
255
+ var HEADER_NAMES = String.raw`(?:urinalysis|urinalise|urina tipo i|rotina de urina|eas)(?![\p{L}\p{N}])`;
256
+ var URINALYSIS_HEADER = new RegExp(`^${HEADER_NAMES}`, "u");
257
+ var ANY_HEADER = new RegExp(String.raw`(?:^|\n)[^\S\n]*` + HEADER_NAMES, "u");
258
+ function mentionsUrinalysisHeader(normalizedText) {
259
+ return ANY_HEADER.test(normalizedText);
260
+ }
261
+ var URINALYSIS_SECTION_NAMES = /* @__PURE__ */ new Map([
262
+ ["bacteria", "Bacteria_Urine"],
263
+ ["bacterias", "Bacteria_Urine"],
264
+ ["bilirrubina", "Bilirubin_Urine"],
265
+ ["bilirubin", "Bilirubin_Urine"],
266
+ ["cetonas", "Ketones_Urine"],
267
+ ["color", "Color_Urine"],
268
+ ["cor", "Color_Urine"],
269
+ ["glicose", "Glucose_Urine"],
270
+ ["glucose", "Glucose_Urine"],
271
+ ["hemacias", "RBC_Urine"],
272
+ ["ketones", "Ketones_Urine"],
273
+ ["leucocitos", "Leukocytes_Urine"],
274
+ ["nitrite", "Nitrite_Urine"],
275
+ ["nitrito", "Nitrite_Urine"],
276
+ ["ph", "pH_Urine"],
277
+ ["protein", "Protein_Urine"],
278
+ ["proteina", "Protein_Urine"],
279
+ ["rbc", "RBC_Urine"],
280
+ ["wbc", "Leukocytes_Urine"]
281
+ ]);
282
+ function isSectionName(matched) {
283
+ return URINALYSIS_SECTION_NAMES.has(matched) || URINALYSIS_SECTION_NAMES.has(matched.replace(/s$/, ""));
284
+ }
285
+ var MENTIONS_URINE = /(?<![\p{L}\p{N}])urin[ae](?![\p{L}\p{N}])/u;
286
+ function urinalysisLineIndexes(lines) {
287
+ const inside = /* @__PURE__ */ new Set();
288
+ let open = false;
289
+ lines.forEach((line, index) => {
290
+ const text = line.text.trim();
291
+ if (URINALYSIS_HEADER.test(text) && !/\d/.test(text)) {
292
+ open = true;
293
+ return;
294
+ }
295
+ if (!open || !text) {
296
+ return;
297
+ }
298
+ if (line.foreign || !line.known && !line.hasValue && !MENTIONS_URINE.test(text)) {
299
+ open = false;
300
+ return;
301
+ }
302
+ inside.add(index);
303
+ });
304
+ return inside;
305
+ }
306
+ function sectionDepsFrom(catalog, buildPattern, hasValue, resolve) {
307
+ const loincByCode = new Map(catalog.map((p) => [p.code, p.loinc]));
308
+ const isUrine = (p) => (Array.isArray(p.category) ? p.category : [p.category]).includes("urina");
309
+ return {
310
+ hasValue,
311
+ patterns: [...URINALYSIS_SECTION_NAMES].map(([name, code]) => {
312
+ const loinc = loincByCode.get(code);
313
+ const entry = { ambiguous: false, code, ...loinc && { loinc }, original: name };
314
+ return { entry, name, regex: buildPattern(name) };
315
+ }),
316
+ resolve,
317
+ urineCodes: new Set(catalog.filter(isUrine).map((p) => p.code))
318
+ };
319
+ }
320
+ function* matchesIn(text, name, regex) {
321
+ if (!text.includes(name)) return;
322
+ regex.lastIndex = 0;
323
+ for (let match = regex.exec(text); match !== null; match = regex.exec(text)) {
324
+ yield match;
325
+ }
326
+ }
327
+ function applyUrinalysisSection(normalizedText, candidates, deps) {
328
+ const resolved = deps.resolve(candidates);
329
+ if (!mentionsUrinalysisHeader(normalizedText)) {
330
+ return resolved;
331
+ }
332
+ const lineStarts = [0];
333
+ for (let i = normalizedText.indexOf("\n"); i !== -1; i = normalizedText.indexOf("\n", i + 1)) {
334
+ lineStarts.push(i + 1);
335
+ }
336
+ const lines = lineStarts.map((start, index) => {
337
+ const end = index + 1 < lineStarts.length ? lineStarts[index + 1] - 1 : normalizedText.length;
338
+ const text = normalizedText.slice(start, end);
339
+ const known = deps.patterns.some(
340
+ ({ name, regex }) => !matchesIn(text, name, regex).next().done
341
+ );
342
+ return { foreign: false, hasValue: deps.hasValue(text), known, text };
343
+ });
344
+ for (const candidate of resolved) {
345
+ let index = lineStarts.length - 1;
346
+ while (lineStarts[index] > candidate.start) index -= 1;
347
+ const line = lines[index];
348
+ line.known = true;
349
+ const matched = normalizedText.slice(candidate.start, candidate.end);
350
+ if (!isSectionName(matched) && !candidate.entries.some((e) => deps.urineCodes.has(e.code))) {
351
+ line.foreign = true;
352
+ }
353
+ }
354
+ const inside = urinalysisLineIndexes(lines);
355
+ if (!inside.size) {
356
+ return resolved;
357
+ }
358
+ const sectionCandidates = [];
359
+ for (const index of inside) {
360
+ const start = lineStarts[index];
361
+ for (const { entry, name, regex } of deps.patterns) {
362
+ for (const match of matchesIn(lines[index].text, name, regex)) {
363
+ const at = start + match.index;
364
+ sectionCandidates.push({ end: at + match[0].length, entries: [entry], start: at });
365
+ }
366
+ }
367
+ }
368
+ const replaced = (c) => sectionCandidates.some((s) => s.start === c.start && s.end === c.end);
369
+ return deps.resolve([...candidates.filter((c) => !replaced(c)), ...sectionCandidates]);
370
+ }
371
+
254
372
  // src/anchor.ts
255
373
  var CONFIDENCE_VALUE_ADJACENT = 1;
256
374
  var CONFIDENCE_NAME_ONLY = 0.7;
@@ -497,6 +615,13 @@ function resolveOverlaps(candidates) {
497
615
  }
498
616
  return accepted;
499
617
  }
618
+ var cachedSectionDeps = null;
619
+ var getSectionDeps = () => cachedSectionDeps ??= sectionDepsFrom(
620
+ getPatterns(),
621
+ buildNamePattern,
622
+ hasValueEvidence,
623
+ resolveOverlaps
624
+ );
500
625
  function findBiomarkersInText(ocrText) {
501
626
  const startTime = Date.now();
502
627
  const normalizedText = normalize(ocrText);
@@ -507,7 +632,7 @@ function findBiomarkersInText(ocrText) {
507
632
  ...collectCandidates(normalizedText),
508
633
  ...collectWrappedCandidates(normalizedText)
509
634
  ];
510
- for (const candidate of resolveOverlaps(candidates)) {
635
+ for (const candidate of applyUrinalysisSection(normalizedText, candidates, getSectionDeps())) {
511
636
  const lineStart = getLineBounds(normalizedText, candidate.start).start;
512
637
  const lineEnd = getLineBounds(normalizedText, candidate.end - 1).end;
513
638
  const key = `${lineStart}:${lineEnd}`;
@@ -1093,7 +1218,7 @@ async function main() {
1093
1218
  strict: false
1094
1219
  });
1095
1220
  if (values.version) {
1096
- process.stdout.write(`${"0.37.2"}
1221
+ process.stdout.write(`${"0.38.0"}
1097
1222
  `);
1098
1223
  return;
1099
1224
  }
package/dist/index.cjs CHANGED
@@ -235,6 +235,124 @@ function attachMethodVariants(matches, normalizedText, anchoredLineStarts, match
235
235
  }
236
236
  }
237
237
 
238
+ // src/urinalysis-section.ts
239
+ var HEADER_NAMES = String.raw`(?:urinalysis|urinalise|urina tipo i|rotina de urina|eas)(?![\p{L}\p{N}])`;
240
+ var URINALYSIS_HEADER = new RegExp(`^${HEADER_NAMES}`, "u");
241
+ var ANY_HEADER = new RegExp(String.raw`(?:^|\n)[^\S\n]*` + HEADER_NAMES, "u");
242
+ function mentionsUrinalysisHeader(normalizedText) {
243
+ return ANY_HEADER.test(normalizedText);
244
+ }
245
+ var URINALYSIS_SECTION_NAMES = /* @__PURE__ */ new Map([
246
+ ["bacteria", "Bacteria_Urine"],
247
+ ["bacterias", "Bacteria_Urine"],
248
+ ["bilirrubina", "Bilirubin_Urine"],
249
+ ["bilirubin", "Bilirubin_Urine"],
250
+ ["cetonas", "Ketones_Urine"],
251
+ ["color", "Color_Urine"],
252
+ ["cor", "Color_Urine"],
253
+ ["glicose", "Glucose_Urine"],
254
+ ["glucose", "Glucose_Urine"],
255
+ ["hemacias", "RBC_Urine"],
256
+ ["ketones", "Ketones_Urine"],
257
+ ["leucocitos", "Leukocytes_Urine"],
258
+ ["nitrite", "Nitrite_Urine"],
259
+ ["nitrito", "Nitrite_Urine"],
260
+ ["ph", "pH_Urine"],
261
+ ["protein", "Protein_Urine"],
262
+ ["proteina", "Protein_Urine"],
263
+ ["rbc", "RBC_Urine"],
264
+ ["wbc", "Leukocytes_Urine"]
265
+ ]);
266
+ function isSectionName(matched) {
267
+ return URINALYSIS_SECTION_NAMES.has(matched) || URINALYSIS_SECTION_NAMES.has(matched.replace(/s$/, ""));
268
+ }
269
+ var MENTIONS_URINE = /(?<![\p{L}\p{N}])urin[ae](?![\p{L}\p{N}])/u;
270
+ function urinalysisLineIndexes(lines) {
271
+ const inside = /* @__PURE__ */ new Set();
272
+ let open = false;
273
+ lines.forEach((line, index) => {
274
+ const text = line.text.trim();
275
+ if (URINALYSIS_HEADER.test(text) && !/\d/.test(text)) {
276
+ open = true;
277
+ return;
278
+ }
279
+ if (!open || !text) {
280
+ return;
281
+ }
282
+ if (line.foreign || !line.known && !line.hasValue && !MENTIONS_URINE.test(text)) {
283
+ open = false;
284
+ return;
285
+ }
286
+ inside.add(index);
287
+ });
288
+ return inside;
289
+ }
290
+ function sectionDepsFrom(catalog, buildPattern, hasValue, resolve) {
291
+ const loincByCode = new Map(catalog.map((p) => [p.code, p.loinc]));
292
+ const isUrine = (p) => (Array.isArray(p.category) ? p.category : [p.category]).includes("urina");
293
+ return {
294
+ hasValue,
295
+ patterns: [...URINALYSIS_SECTION_NAMES].map(([name, code]) => {
296
+ const loinc = loincByCode.get(code);
297
+ const entry = { ambiguous: false, code, ...loinc && { loinc }, original: name };
298
+ return { entry, name, regex: buildPattern(name) };
299
+ }),
300
+ resolve,
301
+ urineCodes: new Set(catalog.filter(isUrine).map((p) => p.code))
302
+ };
303
+ }
304
+ function* matchesIn(text, name, regex) {
305
+ if (!text.includes(name)) return;
306
+ regex.lastIndex = 0;
307
+ for (let match = regex.exec(text); match !== null; match = regex.exec(text)) {
308
+ yield match;
309
+ }
310
+ }
311
+ function applyUrinalysisSection(normalizedText, candidates, deps) {
312
+ const resolved = deps.resolve(candidates);
313
+ if (!mentionsUrinalysisHeader(normalizedText)) {
314
+ return resolved;
315
+ }
316
+ const lineStarts = [0];
317
+ for (let i = normalizedText.indexOf("\n"); i !== -1; i = normalizedText.indexOf("\n", i + 1)) {
318
+ lineStarts.push(i + 1);
319
+ }
320
+ const lines = lineStarts.map((start, index) => {
321
+ const end = index + 1 < lineStarts.length ? lineStarts[index + 1] - 1 : normalizedText.length;
322
+ const text = normalizedText.slice(start, end);
323
+ const known = deps.patterns.some(
324
+ ({ name, regex }) => !matchesIn(text, name, regex).next().done
325
+ );
326
+ return { foreign: false, hasValue: deps.hasValue(text), known, text };
327
+ });
328
+ for (const candidate of resolved) {
329
+ let index = lineStarts.length - 1;
330
+ while (lineStarts[index] > candidate.start) index -= 1;
331
+ const line = lines[index];
332
+ line.known = true;
333
+ const matched = normalizedText.slice(candidate.start, candidate.end);
334
+ if (!isSectionName(matched) && !candidate.entries.some((e) => deps.urineCodes.has(e.code))) {
335
+ line.foreign = true;
336
+ }
337
+ }
338
+ const inside = urinalysisLineIndexes(lines);
339
+ if (!inside.size) {
340
+ return resolved;
341
+ }
342
+ const sectionCandidates = [];
343
+ for (const index of inside) {
344
+ const start = lineStarts[index];
345
+ for (const { entry, name, regex } of deps.patterns) {
346
+ for (const match of matchesIn(lines[index].text, name, regex)) {
347
+ const at = start + match.index;
348
+ sectionCandidates.push({ end: at + match[0].length, entries: [entry], start: at });
349
+ }
350
+ }
351
+ }
352
+ const replaced = (c) => sectionCandidates.some((s) => s.start === c.start && s.end === c.end);
353
+ return deps.resolve([...candidates.filter((c) => !replaced(c)), ...sectionCandidates]);
354
+ }
355
+
238
356
  // src/anchor.ts
239
357
  var CONFIDENCE_VALUE_ADJACENT = 1;
240
358
  var CONFIDENCE_NAME_ONLY = 0.7;
@@ -481,6 +599,13 @@ function resolveOverlaps(candidates) {
481
599
  }
482
600
  return accepted;
483
601
  }
602
+ var cachedSectionDeps = null;
603
+ var getSectionDeps = () => cachedSectionDeps ??= sectionDepsFrom(
604
+ getPatterns(),
605
+ buildNamePattern,
606
+ hasValueEvidence,
607
+ resolveOverlaps
608
+ );
484
609
  function findBiomarkersInText(ocrText) {
485
610
  const startTime = Date.now();
486
611
  const normalizedText = normalize(ocrText);
@@ -491,7 +616,7 @@ function findBiomarkersInText(ocrText) {
491
616
  ...collectCandidates(normalizedText),
492
617
  ...collectWrappedCandidates(normalizedText)
493
618
  ];
494
- for (const candidate of resolveOverlaps(candidates)) {
619
+ for (const candidate of applyUrinalysisSection(normalizedText, candidates, getSectionDeps())) {
495
620
  const lineStart = getLineBounds(normalizedText, candidate.start).start;
496
621
  const lineEnd = getLineBounds(normalizedText, candidate.end - 1).end;
497
622
  const key = `${lineStart}:${lineEnd}`;
@@ -881,7 +1006,7 @@ async function extractWithModel(text, options) {
881
1006
  const tookMs = Date.now() - startedAt;
882
1007
  try {
883
1008
  return { payload: JSON.parse(stripFence(raw)), raw, tookMs };
884
- } catch (e) {
1009
+ } catch (e2) {
885
1010
  throw new Error(`O modelo n\xE3o devolveu JSON analis\xE1vel. Resposta crua:
886
1011
  ${raw.slice(0, 500)}`);
887
1012
  }