@precisa-saude/fhir-ocr-utils 0.37.2 → 0.38.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -119,6 +119,34 @@ var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
119
119
  "undetectable",
120
120
  "yellow"
121
121
  ]);
122
+ var GENETIC_CONTEXT_PATTERNS = [
123
+ /\b[nx][mrpc]_\d{6,}/,
124
+ // RefSeq: NM_000384.2, NP_, NR_, XM_
125
+ /\bens[gtp]\d{6,}/,
126
+ // Ensembl: ENSG00000084674
127
+ /\bp\.[a-z]{3}\d/,
128
+ // HGVS proteína: p.Trp448*
129
+ /\bc\.\d+[acgt]?[>_+-]/,
130
+ // HGVS codificante: c.1234A>G, c.76_78del
131
+ /\brs\d{4,}\b/,
132
+ // dbSNP
133
+ /\bgenes?\b/,
134
+ /\bvariante?s?\b/,
135
+ /\bexons?\b/,
136
+ /\bzygosity\b/,
137
+ /\bzigosidade\b/,
138
+ /\balleles?\b/,
139
+ /\balelos?\b/,
140
+ /\bmutations?\b/,
141
+ /\bmutac(ao|oes)\b/,
142
+ /\bpathogenic/,
143
+ /\bpatogenic/,
144
+ /\bheterozyg/,
145
+ /\bhomozyg/,
146
+ /\bheterozigot/,
147
+ /\bhomozigot/,
148
+ /\bsequence change\b/
149
+ ];
122
150
 
123
151
  // src/body-region.ts
124
152
  var WHOLE_BODY_COMPOSITION_CODES = /* @__PURE__ */ new Set([
@@ -235,6 +263,144 @@ function attachMethodVariants(matches, normalizedText, anchoredLineStarts, match
235
263
  }
236
264
  }
237
265
 
266
+ // src/urinalysis-section.ts
267
+ var HEADER_NAMES = String.raw`(?:urinalysis|urinalise|urina tipo i|rotina de urina|eas)(?![\p{L}\p{N}])`;
268
+ var URINALYSIS_HEADER = new RegExp(`^${HEADER_NAMES}`, "u");
269
+ var ANY_HEADER = new RegExp(String.raw`(?:^|\n)[^\S\n]*` + HEADER_NAMES, "u");
270
+ function mentionsUrinalysisHeader(normalizedText) {
271
+ return ANY_HEADER.test(normalizedText);
272
+ }
273
+ var URINALYSIS_SECTION_NAMES = /* @__PURE__ */ new Map([
274
+ ["bacteria", "Bacteria_Urine"],
275
+ ["bacterias", "Bacteria_Urine"],
276
+ ["bilirrubina", "Bilirubin_Urine"],
277
+ ["bilirubin", "Bilirubin_Urine"],
278
+ ["cetonas", "Ketones_Urine"],
279
+ ["color", "Color_Urine"],
280
+ ["cor", "Color_Urine"],
281
+ ["glicose", "Glucose_Urine"],
282
+ ["glucose", "Glucose_Urine"],
283
+ ["hemacias", "RBC_Urine"],
284
+ ["ketones", "Ketones_Urine"],
285
+ ["leucocitos", "Leukocytes_Urine"],
286
+ ["nitrite", "Nitrite_Urine"],
287
+ ["nitrito", "Nitrite_Urine"],
288
+ ["ph", "pH_Urine"],
289
+ ["protein", "Protein_Urine"],
290
+ ["proteina", "Protein_Urine"],
291
+ ["rbc", "RBC_Urine"],
292
+ ["wbc", "Leukocytes_Urine"]
293
+ ]);
294
+ function isSectionName(matched) {
295
+ return URINALYSIS_SECTION_NAMES.has(matched) || URINALYSIS_SECTION_NAMES.has(matched.replace(/s$/, ""));
296
+ }
297
+ var MENTIONS_URINE = /(?<![\p{L}\p{N}])urin[ae](?![\p{L}\p{N}])/u;
298
+ var SEDIMENT_FIELD_UNIT = /\/(?:hpf|lpf)(?![\p{L}\p{N}])/u;
299
+ function hasUrineCue(line) {
300
+ return MENTIONS_URINE.test(line) || SEDIMENT_FIELD_UNIT.test(line);
301
+ }
302
+ function kindOf(line) {
303
+ const text = line.text.trim();
304
+ if (!text) return "blank";
305
+ if (URINALYSIS_HEADER.test(text) && !/\d/.test(text)) return "header";
306
+ if (line.foreign) return "foreign";
307
+ if (line.urine) return "urine";
308
+ return line.known || line.hasValue ? "neutral" : "unknown";
309
+ }
310
+ function urinalysisLineIndexes(lines) {
311
+ const kinds = lines.map(kindOf);
312
+ const continuesAsUrine = (from) => {
313
+ const next = kinds.findIndex(
314
+ (kind, index) => index > from && (kind === "urine" || kind === "foreign" || kind === "header")
315
+ );
316
+ return next !== -1 && kinds[next] === "urine";
317
+ };
318
+ const inside = /* @__PURE__ */ new Set();
319
+ let open = false;
320
+ kinds.forEach((kind, index) => {
321
+ if (kind === "header") {
322
+ open = true;
323
+ return;
324
+ }
325
+ if (!open || kind === "blank") {
326
+ return;
327
+ }
328
+ if (kind === "foreign" || kind === "unknown" && !continuesAsUrine(index)) {
329
+ open = false;
330
+ return;
331
+ }
332
+ inside.add(index);
333
+ });
334
+ return inside;
335
+ }
336
+ function sectionDepsFrom(catalog, buildPattern, hasValue, resolve) {
337
+ const loincByCode = new Map(catalog.map((p) => [p.code, p.loinc]));
338
+ const isUrine = (p) => (Array.isArray(p.category) ? p.category : [p.category]).includes("urina");
339
+ return {
340
+ hasValue,
341
+ patterns: [...URINALYSIS_SECTION_NAMES].map(([name, code]) => {
342
+ const loinc = loincByCode.get(code);
343
+ const entry = { ambiguous: false, code, ...loinc && { loinc }, original: name };
344
+ return { entry, name, regex: buildPattern(name) };
345
+ }),
346
+ resolve,
347
+ urineCodes: new Set(catalog.filter(isUrine).map((p) => p.code))
348
+ };
349
+ }
350
+ function* matchesIn(text, name, regex) {
351
+ if (!text.includes(name)) return;
352
+ regex.lastIndex = 0;
353
+ for (let match = regex.exec(text); match !== null; match = regex.exec(text)) {
354
+ yield match;
355
+ }
356
+ }
357
+ function applyUrinalysisSection(normalizedText, candidates, deps) {
358
+ const resolved = deps.resolve(candidates);
359
+ if (!mentionsUrinalysisHeader(normalizedText)) {
360
+ return resolved;
361
+ }
362
+ const lineStarts = [0];
363
+ for (let i = normalizedText.indexOf("\n"); i !== -1; i = normalizedText.indexOf("\n", i + 1)) {
364
+ lineStarts.push(i + 1);
365
+ }
366
+ const lines = lineStarts.map((start, index) => {
367
+ const end = index + 1 < lineStarts.length ? lineStarts[index + 1] - 1 : normalizedText.length;
368
+ const text = normalizedText.slice(start, end);
369
+ const known = deps.patterns.some(
370
+ ({ name, regex }) => !matchesIn(text, name, regex).next().done
371
+ );
372
+ return { foreign: false, hasValue: deps.hasValue(text), known, text, urine: hasUrineCue(text) };
373
+ });
374
+ for (const candidate of resolved) {
375
+ let index = lineStarts.length - 1;
376
+ while (lineStarts[index] > candidate.start) index -= 1;
377
+ const line = lines[index];
378
+ line.known = true;
379
+ const matched = normalizedText.slice(candidate.start, candidate.end);
380
+ if (candidate.entries.some((e) => deps.urineCodes.has(e.code))) {
381
+ line.urine = true;
382
+ } else if (!isSectionName(matched)) {
383
+ line.foreign = true;
384
+ }
385
+ }
386
+ const inside = urinalysisLineIndexes(lines);
387
+ if (!inside.size) {
388
+ return resolved;
389
+ }
390
+ const sectionCandidates = [];
391
+ for (const index of inside) {
392
+ const start = lineStarts[index];
393
+ for (const { entry, name, regex } of deps.patterns) {
394
+ for (const match of matchesIn(lines[index].text, name, regex)) {
395
+ const at = start + match.index;
396
+ sectionCandidates.push({ end: at + match[0].length, entries: [entry], start: at });
397
+ }
398
+ }
399
+ }
400
+ const replaced = (c) => sectionCandidates.some((s) => s.start === c.start && s.end === c.end);
401
+ return deps.resolve([...candidates.filter((c) => !replaced(c)), ...sectionCandidates]);
402
+ }
403
+
238
404
  // src/anchor.ts
239
405
  var CONFIDENCE_VALUE_ADJACENT = 1;
240
406
  var CONFIDENCE_NAME_ONLY = 0.7;
@@ -286,40 +452,13 @@ function foldCommas(text) {
286
452
  (comma, space, offset) => startsWithCatalogName(text.slice(offset + 1 + space.length)) ? comma : " "
287
453
  );
288
454
  }
289
- var GENETIC_CONTEXT_PATTERNS = [
290
- /\b[nx][mrpc]_\d{6,}/,
291
- // RefSeq: NM_000384.2, NP_, NR_, XM_
292
- /\bens[gtp]\d{6,}/,
293
- // Ensembl: ENSG00000084674
294
- /\bp\.[a-z]{3}\d/,
295
- // HGVS proteína: p.Trp448*
296
- /\bc\.\d+[acgt]?[>_+-]/,
297
- // HGVS codificante: c.1234A>G, c.76_78del
298
- /\brs\d{4,}\b/,
299
- // dbSNP
300
- /\bgenes?\b/,
301
- /\bvariante?s?\b/,
302
- /\bexons?\b/,
303
- /\bzygosity\b/,
304
- /\bzigosidade\b/,
305
- /\balleles?\b/,
306
- /\balelos?\b/,
307
- /\bmutations?\b/,
308
- /\bmutac(ao|oes)\b/,
309
- /\bpathogenic/,
310
- /\bpatogenic/,
311
- /\bheterozyg/,
312
- /\bhomozyg/,
313
- /\bheterozigot/,
314
- /\bhomozigot/,
315
- /\bsequence change\b/
316
- ];
317
455
  var DIGIT_PATTERN = /\d/;
318
456
  var cachedUnitTokens = null;
319
457
  function getUnitTokens() {
320
458
  if (!cachedUnitTokens) {
459
+ const isExamName = (t) => URINALYSIS_SECTION_NAMES.has(t) || getNamePatterns().has(t);
321
460
  cachedUnitTokens = new Set(
322
- Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter(Boolean)
461
+ Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter((unit) => unit && !isExamName(unit))
323
462
  );
324
463
  }
325
464
  return cachedUnitTokens;
@@ -481,6 +620,13 @@ function resolveOverlaps(candidates) {
481
620
  }
482
621
  return accepted;
483
622
  }
623
+ var cachedSectionDeps = null;
624
+ var getSectionDeps = () => cachedSectionDeps ??= sectionDepsFrom(
625
+ getPatterns(),
626
+ buildNamePattern,
627
+ hasValueEvidence,
628
+ resolveOverlaps
629
+ );
484
630
  function findBiomarkersInText(ocrText) {
485
631
  const startTime = Date.now();
486
632
  const normalizedText = normalize(ocrText);
@@ -491,7 +637,7 @@ function findBiomarkersInText(ocrText) {
491
637
  ...collectCandidates(normalizedText),
492
638
  ...collectWrappedCandidates(normalizedText)
493
639
  ];
494
- for (const candidate of resolveOverlaps(candidates)) {
640
+ for (const candidate of applyUrinalysisSection(normalizedText, candidates, getSectionDeps())) {
495
641
  const lineStart = getLineBounds(normalizedText, candidate.start).start;
496
642
  const lineEnd = getLineBounds(normalizedText, candidate.end - 1).end;
497
643
  const key = `${lineStart}:${lineEnd}`;