@precisa-saude/fhir-ocr-utils 0.20.3 → 0.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -70,8 +70,22 @@ var CONTEXT_REQUIRED_NAMES = /* @__PURE__ */ new Set([
70
70
  // ProthrombinTime — "tap" em inglês
71
71
  "volume",
72
72
  // VATVolume
73
- "weight"
73
+ "weight",
74
74
  // TotalMass
75
+ // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,
76
+ // e aparecem em prosa: num laudo de DEXA real, "hips and thighs" e
77
+ // "abdominal region" ancoravam dobra cutânea que o documento não tem.
78
+ // Exigir valor na linha separa a tabela do parágrafo.
79
+ "abdominal",
80
+ "chest",
81
+ "coxa",
82
+ "peitoral",
83
+ "subescapular",
84
+ "subscapular",
85
+ "suprailiac",
86
+ "thigh",
87
+ "triceps",
88
+ "tricipital"
75
89
  ]);
76
90
  var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
77
91
  "absent",
@@ -163,11 +177,12 @@ var GIRTH_CONTEXT_PATTERNS = [
163
177
  /\bgirth\b/
164
178
  ];
165
179
  var SKINFOLD_CONTEXT_PATTERNS = [/\bdobras?\b/, /\bskin ?folds?\b/, /\bpregas?\b/];
180
+ var CENTIMETRE_VALUE = /\d\s*(?:,\d+\s*)?cm\b/;
166
181
  function hasGirthContext(line) {
167
182
  if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {
168
183
  return false;
169
184
  }
170
- return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line));
185
+ return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);
171
186
  }
172
187
  var DIGIT_PATTERN = /\d/;
173
188
  var cachedUnitTokens = null;
@@ -434,7 +449,7 @@ async function main() {
434
449
  strict: false
435
450
  });
436
451
  if (values.version) {
437
- process.stdout.write(`${"0.20.3"}
452
+ process.stdout.write(`${"0.21.0"}
438
453
  `);
439
454
  return;
440
455
  }
package/dist/index.cjs CHANGED
@@ -61,8 +61,22 @@ var CONTEXT_REQUIRED_NAMES = /* @__PURE__ */ new Set([
61
61
  // ProthrombinTime — "tap" em inglês
62
62
  "volume",
63
63
  // VATVolume
64
- "weight"
64
+ "weight",
65
65
  // TotalMass
66
+ // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,
67
+ // e aparecem em prosa: num laudo de DEXA real, "hips and thighs" e
68
+ // "abdominal region" ancoravam dobra cutânea que o documento não tem.
69
+ // Exigir valor na linha separa a tabela do parágrafo.
70
+ "abdominal",
71
+ "chest",
72
+ "coxa",
73
+ "peitoral",
74
+ "subescapular",
75
+ "subscapular",
76
+ "suprailiac",
77
+ "thigh",
78
+ "triceps",
79
+ "tricipital"
66
80
  ]);
67
81
  var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
68
82
  "absent",
@@ -154,11 +168,12 @@ var GIRTH_CONTEXT_PATTERNS = [
154
168
  /\bgirth\b/
155
169
  ];
156
170
  var SKINFOLD_CONTEXT_PATTERNS = [/\bdobras?\b/, /\bskin ?folds?\b/, /\bpregas?\b/];
171
+ var CENTIMETRE_VALUE = /\d\s*(?:,\d+\s*)?cm\b/;
157
172
  function hasGirthContext(line) {
158
173
  if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {
159
174
  return false;
160
175
  }
161
- return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line));
176
+ return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);
162
177
  }
163
178
  var DIGIT_PATTERN = /\d/;
164
179
  var cachedUnitTokens = null;
@@ -1 +1 @@
1
- {"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts"],"names":[],"mappings":"AAAA;ACaA;AAEE;AACA;AACA;AAAA,2CACK;AAwBA,IAAM,0BAAA,EAA4B,CAAA;AAMlC,IAAM,qBAAA,EAAuB,GAAA;AAM7B,IAAM,qBAAA,EAAuB,GAAA;AAGpC,IAAM,yBAAA,EAA2B,CAAA;AASjC,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,WAAA,EAAa,GAAG,CAAA;AAC7B;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAQD,IAAM,uBAAA,kBAAyB,IAAI,GAAA,CAAI;AAAA,EACrC,UAAA;AAAA;AAAA,EACA,WAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,YAAA;AAAA;AAAA,EACA,KAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA,EACA;AAAA;AACF,CAAC,CAAA;AAOD,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,QAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,OAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,cAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,OAAA;AAAA,EACA,UAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,QAAA;AAAA,EACA,OAAA;AAAA,EACA,cAAA;AAAA,EACA;AACF,CAAC,CAAA;AASD,IAAM,yBAAA,EAAqC;AAAA,EACzC,qBAAA;AAAA;AAAA,EACA,kBAAA;AAAA;AAAA,EACA,iBAAA;AAAA;AAAA,EACA,uBAAA;AAAA;AAAA,EACA,cAAA;AAAA;AAAA,EACA,YAAA;AAAA,EACA,iBAAA;AAAA,EACA,YAAA;AAAA,EACA,cAAA;AAAA,EACA,gBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,gBAAA;AAAA,EACA,mBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,aAAA;AAAA,EACA,WAAA;AAAA,EACA,eAAA;AAAA,EACA,aAAA;AAAA,EACA;AACF,CAAA;AASA,IAAM,oBAAA,kBAAsB,IAAI,GAAA,CAAI;AAAA,EAClC,mBAAA;AAAA,EACA,eAAA;AAAA,EACA,qBAAA;AAAA,EACA,qBAAA;AAAA,EACA,oBAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAC,CAAA;AAGD,IAAM,uBAAA,EAAmC;AAAA,EACvC,mBAAA;AAAA,EACA,sBAAA;AAAA,EACA,iBAAA;AAAA,EACA;AACF,CAAA;AAOA,IAAM,0BAAA,EAAsC,CAAC,aAAA,EAAe,kBAAA,EAAoB,aAAa,CAAA;AAE7F,SAAS,eAAA,CAAgB,IAAA,EAAuB;AAC9C,EAAA,GAAA,CAAI,yBAAA,CAA0B,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,CAAA,EAAG;AACzD,IAAA,OAAO,KAAA;AAAA,EACT;AACA,EAAA,OAAO,sBAAA,CAAuB,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,CAAA;AAC1D;AAEA,IAAM,cAAA,EAAgB,IAAA;AAGtB,IAAI,iBAAA,EAAuC,IAAA;AAE3C,SAAS,aAAA,CAAA,EAA6B;AACpC,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,iBAAA,EAAmB,IAAI,GAAA;AAAA,MACrB,MAAA,CAAO,IAAA,CAAK,kBAAY,CAAA,CACrB,GAAA,CAAI,CAAC,IAAA,EAAA,GAAS,SAAA,CAAU,IAAI,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,CACpC,MAAA,CAAO,OAAO;AAAA,IACnB,CAAA;AAAA,EACF;AACA,EAAA,OAAO,gBAAA;AACT;AAqBA,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,mBAAA,EAAsD,IAAA;AAE1D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,YAAA,CAAa,IAAA,EAAsB;AAC1C,EAAA,OAAO,IAAA,CAAK,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AACnD;AAkBA,SAAS,gBAAA,CAAiB,cAAA,EAAgC;AACxD,EAAA,MAAM,KAAA,EAAO,cAAA,CAAe,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,YAAY,CAAA,CAAE,IAAA,CAAK,YAAY,CAAA;AAC1E,EAAA,MAAM,OAAA,EAAS,SAAA,CAAU,IAAA,CAAK,cAAc,EAAA,EAAI,KAAA,EAAO,EAAA;AACvD,EAAA,OAAO,IAAI,MAAA,CAAO,CAAA,mBAAA,EAAsB,IAAI,CAAA,EAAA;AAC9C;AAEsE;AAC3B,EAAA;AACD,EAAA;AAC1C;AAOiD;AACb,EAAA;AACzB,IAAA;AACT,EAAA;AACkC,EAAA;AACpC;AAEqD;AAC1B,EAAA;AACkB,IAAA;AACJ,IAAA;AACD,MAAA;AACG,QAAA;AAClB,QAAA;AACf,UAAA;AACF,QAAA;AAC8B,QAAA;AAC5B,UAAA;AACF,QAAA;AAC6B,QAAA;AAClB,QAAA;AACyB,UAAA;AACV,UAAA;AAC1B,QAAA;AACkB,QAAA;AACW,UAAA;AACb,UAAA;AACwB,UAAA;AAC5B,UAAA;AACX,QAAA;AACH,MAAA;AACF,IAAA;AACqB,IAAA;AACvB,EAAA;AACO,EAAA;AACT;AAEuF;AACxC,EAAA;AACA,EAAA;AACP,EAAA;AACxC;AAEkD;AACV,EAAA;AACxC;AAMiD;AACjB,EAAA;AACrB,IAAA;AACT,EAAA;AACiC,EAAA;AACF,EAAA;AACU,IAAA;AAC9B,MAAA;AACT,IAAA;AACF,EAAA;AACO,EAAA;AACT;AASgE;AAC7B,EAAA;AACU,EAAA;AACL,IAAA;AAClC,MAAA;AACF,IAAA;AACoB,IAAA;AACU,IAAA;AACZ,IAAA;AACA,IAAA;AACmB,IAAA;AACE,IAAA;AACA,MAAA;AACtB,MAAA;AACkB,MAAA;AACnC,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAa+D;AAC9B,EAAA;AACU,IAAA;AACzC,EAAA;AAC+B,EAAA;AACC,EAAA;AACW,IAAA;AACd,IAAA;AAEE,MAAA;AAG7B,IAAA;AACgB,IAAA;AACS,MAAA;AACzB,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAUoE;AACvC,EAAA;AACa,EAAA;AACQ,EAAA;AACF,EAAA;AACF,EAAA;AACA,EAAA;AAEJ,EAAA;AACK,IAAA;AAEH,IAAA;AACb,IAAA;AACG,MAAA;AACO,MAAA;AACrC,IAAA;AACa,IAAA;AACX,MAAA;AACF,IAAA;AAEuC,IAAA;AACX,IAAA;AACE,MAAA;AACM,MAAA;AACpC,IAAA;AAEoC,IAAA;AACX,IAAA;AACgB,MAAA;AACR,MAAA;AACjC,IAAA;AAEuC,IAAA;AACH,MAAA;AAChC,QAAA;AACF,MAAA;AAEqC,MAAA;AACnC,QAAA;AACF,MAAA;AAEiB,MAAA;AACI,MAAA;AACN,QAAA;AACM,MAAA;AACN,QAAA;AACf,MAAA;AAEsC,MAAA;AAGvB,MAAA;AAEH,MAAA;AACiB,QAAA;AACb,UAAA;AACZ,UAAA;AACa,UAAA;AACM,UAAA;AACC,UAAA;AACrB,QAAA;AACH,MAAA;AACF,IAAA;AACF,EAAA;AAE6C,EAAA;AACb,EAAA;AAEzB,EAAA;AACc,IAAA;AACnB,IAAA;AACO,IAAA;AACiB,MAAA;AACtB,MAAA;AAC6B,MAAA;AAC/B,IAAA;AACF,EAAA;AACF;AAKgE;AACvB,EAAA;AACzC;ADlM+C;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line));\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"]}
1
+ {"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts"],"names":[],"mappings":"AAAA;ACaA;AAEE;AACA;AACA;AAAA,2CACK;AAwBA,IAAM,0BAAA,EAA4B,CAAA;AAMlC,IAAM,qBAAA,EAAuB,GAAA;AAM7B,IAAM,qBAAA,EAAuB,GAAA;AAGpC,IAAM,yBAAA,EAA2B,CAAA;AASjC,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,WAAA,EAAa,GAAG,CAAA;AAC7B;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAQD,IAAM,uBAAA,kBAAyB,IAAI,GAAA,CAAI;AAAA,EACrC,UAAA;AAAA;AAAA,EACA,WAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,YAAA;AAAA;AAAA,EACA,KAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,UAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,YAAA;AAAA,EACA,OAAA;AAAA,EACA,SAAA;AAAA,EACA;AACF,CAAC,CAAA;AAOD,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,QAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,OAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,cAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,OAAA;AAAA,EACA,UAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,QAAA;AAAA,EACA,OAAA;AAAA,EACA,cAAA;AAAA,EACA;AACF,CAAC,CAAA;AASD,IAAM,yBAAA,EAAqC;AAAA,EACzC,qBAAA;AAAA;AAAA,EACA,kBAAA;AAAA;AAAA,EACA,iBAAA;AAAA;AAAA,EACA,uBAAA;AAAA;AAAA,EACA,cAAA;AAAA;AAAA,EACA,YAAA;AAAA,EACA,iBAAA;AAAA,EACA,YAAA;AAAA,EACA,cAAA;AAAA,EACA,gBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,gBAAA;AAAA,EACA,mBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,aAAA;AAAA,EACA,WAAA;AAAA,EACA,eAAA;AAAA,EACA,aAAA;AAAA,EACA;AACF,CAAA;AASA,IAAM,oBAAA,kBAAsB,IAAI,GAAA,CAAI;AAAA,EAClC,mBAAA;AAAA,EACA,eAAA;AAAA,EACA,qBAAA;AAAA,EACA,qBAAA;AAAA,EACA,oBAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAC,CAAA;AAGD,IAAM,uBAAA,EAAmC;AAAA,EACvC,mBAAA;AAAA,EACA,sBAAA;AAAA,EACA,iBAAA;AAAA,EACA;AACF,CAAA;AAOA,IAAM,0BAAA,EAAsC,CAAC,aAAA,EAAe,kBAAA,EAAoB,aAAa,CAAA;AAa7F,IAAM,iBAAA,EAAmB,uBAAA;AAEzB,SAAS,eAAA,CAAgB,IAAA,EAAuB;AAC9C,EAAA,GAAA,CAAI,yBAAA,CAA0B,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,CAAA,EAAG;AACzD,IAAA,OAAO,KAAA;AAAA,EACT;AACA,EAAA,OAAO,sBAAA,CAAuB,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,EAAA,GAAK,gBAAA,CAAiB,IAAA,CAAK,IAAI,CAAA;AACzF;AAEA,IAAM,cAAA,EAAgB,IAAA;AAGtB,IAAI,iBAAA,EAAuC,IAAA;AAE3C,SAAS,aAAA,CAAA,EAA6B;AACpC,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,iBAAA,EAAmB,IAAI,GAAA;AAAA,MACrB,MAAA,CAAO,IAAA,CAAK,kBAAY,CAAA,CACrB,GAAA,CAAI,CAAC,IAAA,EAAA,GAAS,SAAA,CAAU,IAAI,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,CACpC,MAAA,CAAO,OAAO;AAAA,IACnB,CAAA;AAAA,EACF;AACA,EAAA,OAAO,gBAAA;AACT;AAqBA,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,mBAAA,EAAsD,IAAA;AAE1D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,YAAA,CAAa,IAAA,EAAsB;AAC1C,EAAA,OAAO,IAAA,CAAK,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AACnD;AAkBA,SAAS,gBAAA,CAAiB,cAAA,EAAgC;AACxD,EAAA,MAAM,KAAA,EAAO,cAAA,CAAe,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,YAAY,CAAA,CAAE,IAAA,CAAK,YAAY,CAAA;AAC1E,EAAA,MAAM,OAAA,EAAS,SAAA,CAAU,IAAA,CAAK,cAAc,EAAA,EAAI,KAAA,EAAO,EAAA;AACvD,EAAA,OAAO,IAAI,MAAA,CAAO,CAAA,mBAAA,EAAsB,IAAI,CAAA,EAAA;AAC9C;AAEsE;AAC3B,EAAA;AACD,EAAA;AAC1C;AAOiD;AACb,EAAA;AACzB,IAAA;AACT,EAAA;AACkC,EAAA;AACpC;AAEqD;AAC1B,EAAA;AACkB,IAAA;AACJ,IAAA;AACD,MAAA;AACG,QAAA;AAClB,QAAA;AACf,UAAA;AACF,QAAA;AAC8B,QAAA;AAC5B,UAAA;AACF,QAAA;AAC6B,QAAA;AAClB,QAAA;AACyB,UAAA;AACV,UAAA;AAC1B,QAAA;AACkB,QAAA;AACW,UAAA;AACb,UAAA;AACwB,UAAA;AAC5B,UAAA;AACX,QAAA;AACH,MAAA;AACF,IAAA;AACqB,IAAA;AACvB,EAAA;AACO,EAAA;AACT;AAEuF;AACxC,EAAA;AACA,EAAA;AACP,EAAA;AACxC;AAEkD;AACV,EAAA;AACxC;AAMiD;AACjB,EAAA;AACrB,IAAA;AACT,EAAA;AACiC,EAAA;AACF,EAAA;AACU,IAAA;AAC9B,MAAA;AACT,IAAA;AACF,EAAA;AACO,EAAA;AACT;AASgE;AAC7B,EAAA;AACU,EAAA;AACL,IAAA;AAClC,MAAA;AACF,IAAA;AACoB,IAAA;AACU,IAAA;AACZ,IAAA;AACA,IAAA;AACmB,IAAA;AACE,IAAA;AACA,MAAA;AACtB,MAAA;AACkB,MAAA;AACnC,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAa+D;AAC9B,EAAA;AACU,IAAA;AACzC,EAAA;AAC+B,EAAA;AACC,EAAA;AACW,IAAA;AACd,IAAA;AAEE,MAAA;AAG7B,IAAA;AACgB,IAAA;AACS,MAAA;AACzB,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAUoE;AACvC,EAAA;AACa,EAAA;AACQ,EAAA;AACF,EAAA;AACF,EAAA;AACA,EAAA;AAEJ,EAAA;AACK,IAAA;AAEH,IAAA;AACb,IAAA;AACG,MAAA;AACO,MAAA;AACrC,IAAA;AACa,IAAA;AACX,MAAA;AACF,IAAA;AAEuC,IAAA;AACX,IAAA;AACE,MAAA;AACM,MAAA;AACpC,IAAA;AAEoC,IAAA;AACX,IAAA;AACgB,MAAA;AACR,MAAA;AACjC,IAAA;AAEuC,IAAA;AACH,MAAA;AAChC,QAAA;AACF,MAAA;AAEqC,MAAA;AACnC,QAAA;AACF,MAAA;AAEiB,MAAA;AACI,MAAA;AACN,QAAA;AACM,MAAA;AACN,QAAA;AACf,MAAA;AAEsC,MAAA;AAGvB,MAAA;AAEH,MAAA;AACiB,QAAA;AACb,UAAA;AACZ,UAAA;AACa,UAAA;AACM,UAAA;AACC,UAAA;AACrB,QAAA;AACH,MAAA;AACF,IAAA;AACF,EAAA;AAE6C,EAAA;AACb,EAAA;AAEzB,EAAA;AACc,IAAA;AACnB,IAAA;AACO,IAAA;AACiB,MAAA;AACtB,MAAA;AAC6B,MAAA;AAC/B,IAAA;AACF,EAAA;AACF;AAKgE;AACvB,EAAA;AACzC;AD9M+C;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"]}
package/dist/index.js CHANGED
@@ -61,8 +61,22 @@ var CONTEXT_REQUIRED_NAMES = /* @__PURE__ */ new Set([
61
61
  // ProthrombinTime — "tap" em inglês
62
62
  "volume",
63
63
  // VATVolume
64
- "weight"
64
+ "weight",
65
65
  // TotalMass
66
+ // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,
67
+ // e aparecem em prosa: num laudo de DEXA real, "hips and thighs" e
68
+ // "abdominal region" ancoravam dobra cutânea que o documento não tem.
69
+ // Exigir valor na linha separa a tabela do parágrafo.
70
+ "abdominal",
71
+ "chest",
72
+ "coxa",
73
+ "peitoral",
74
+ "subescapular",
75
+ "subscapular",
76
+ "suprailiac",
77
+ "thigh",
78
+ "triceps",
79
+ "tricipital"
66
80
  ]);
67
81
  var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
68
82
  "absent",
@@ -154,11 +168,12 @@ var GIRTH_CONTEXT_PATTERNS = [
154
168
  /\bgirth\b/
155
169
  ];
156
170
  var SKINFOLD_CONTEXT_PATTERNS = [/\bdobras?\b/, /\bskin ?folds?\b/, /\bpregas?\b/];
171
+ var CENTIMETRE_VALUE = /\d\s*(?:,\d+\s*)?cm\b/;
157
172
  function hasGirthContext(line) {
158
173
  if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {
159
174
  return false;
160
175
  }
161
- return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line));
176
+ return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);
162
177
  }
163
178
  var DIGIT_PATTERN = /\d/;
164
179
  var cachedUnitTokens = null;
package/dist/index.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/anchor.ts"],"sourcesContent":["/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line));\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"],"mappings":";AAaA;AAAA,EAEE;AAAA,EACA;AAAA,EACA;AAAA,OACK;AAwBA,IAAM,4BAA4B;AAMlC,IAAM,uBAAuB;AAM7B,IAAM,uBAAuB;AAGpC,IAAM,2BAA2B;AASjC,SAAS,UAAU,MAAsB;AACvC,SAAO,KACJ,UAAU,KAAK,EACf,QAAQ,oBAAoB,EAAE,EAC9B,YAAY,EACZ,QAAQ,aAAa,GAAG;AAC7B;AAEA,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAQD,IAAM,yBAAyB,oBAAI,IAAI;AAAA,EACrC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AACF,CAAC;AAOD,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AASD,IAAM,2BAAqC;AAAA,EACzC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AASA,IAAM,sBAAsB,oBAAI,IAAI;AAAA,EAClC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAGD,IAAM,yBAAmC;AAAA,EACvC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAOA,IAAM,4BAAsC,CAAC,eAAe,oBAAoB,aAAa;AAE7F,SAAS,gBAAgB,MAAuB;AAC9C,MAAI,0BAA0B,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC,GAAG;AACzD,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC;AAC1D;AAEA,IAAM,gBAAgB;AAGtB,IAAI,mBAAuC;AAE3C,SAAS,gBAA6B;AACpC,MAAI,CAAC,kBAAkB;AACrB,uBAAmB,IAAI;AAAA,MACrB,OAAO,KAAK,YAAY,EACrB,IAAI,CAAC,SAAS,UAAU,IAAI,EAAE,KAAK,CAAC,EACpC,OAAO,OAAO;AAAA,IACnB;AAAA,EACF;AACA,SAAO;AACT;AAqBA,IAAI,iBAAkD;AACtD,IAAI,qBAAsD;AAE1D,SAAS,cAAwC;AAC/C,MAAI,CAAC,gBAAgB;AACnB,qBAAiB,qBAAqB;AAAA,EACxC;AACA,SAAO;AACT;AAEA,SAAS,aAAa,MAAsB;AAC1C,SAAO,KAAK,QAAQ,uBAAuB,MAAM;AACnD;AAkBA,SAAS,iBAAiB,gBAAgC;AACxD,QAAM,OAAO,eAAe,MAAM,GAAG,EAAE,IAAI,YAAY,EAAE,KAAK,YAAY;AAC1E,QAAM,SAAS,UAAU,KAAK,cAAc,IAAI,OAAO;AACvD,SAAO,IAAI,OAAO,sBAAsB,IAAI,GAAG,MAAM,sBAAsB,IAAI;AACjF;AAEA,SAAS,mBAAmB,SAA0C;AACpE,QAAM,aAAa,MAAM,QAAQ,QAAQ,QAAQ,IAAI,QAAQ,WAAW,CAAC,QAAQ,QAAQ;AACzF,SAAO,WAAW,SAAS,OAAO,KAAK,CAAC,QAAQ;AAClD;AAOA,SAAS,gBAAgB,gBAAwB,SAA0C;AACzF,MAAI,eAAe,SAAS,GAAG,GAAG;AAChC,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,IAAI,cAAc,KAAK,mBAAmB,OAAO;AACjF;AAEA,SAAS,kBAA4C;AACnD,MAAI,CAAC,oBAAoB;AACvB,UAAM,MAAM,oBAAI,IAAyB;AACzC,eAAW,WAAW,YAAY,GAAG;AACnC,iBAAW,QAAQ,QAAQ,OAAO;AAChC,cAAM,aAAa,UAAU,IAAI,EAAE,KAAK;AACxC,YAAI,CAAC,YAAY;AACf;AAAA,QACF;AACA,YAAI,WAAW,SAAS,KAAK,CAAC,wBAAwB,IAAI,UAAU,GAAG;AACrE;AAAA,QACF;AACA,YAAI,OAAO,IAAI,IAAI,UAAU;AAC7B,YAAI,CAAC,MAAM;AACT,iBAAO,EAAE,SAAS,CAAC,GAAG,OAAO,KAAK;AAClC,cAAI,IAAI,YAAY,IAAI;AAAA,QAC1B;AACA,aAAK,QAAQ,KAAK;AAAA,UAChB,WAAW,gBAAgB,YAAY,OAAO;AAAA,UAC9C,MAAM,QAAQ;AAAA,UACd,GAAI,QAAQ,SAAS,EAAE,OAAO,QAAQ,MAAM;AAAA,UAC5C,UAAU;AAAA,QACZ,CAAC;AAAA,MACH;AAAA,IACF;AACA,yBAAqB;AAAA,EACvB;AACA,SAAO;AACT;AAEA,SAAS,cAAc,MAAc,UAAkD;AACrF,QAAM,QAAQ,KAAK,YAAY,MAAM,QAAQ,IAAI;AACjD,QAAM,YAAY,KAAK,QAAQ,MAAM,QAAQ;AAC7C,SAAO,EAAE,KAAK,cAAc,KAAK,KAAK,SAAS,WAAW,MAAM;AAClE;AAEA,SAAS,kBAAkB,MAAuB;AAChD,SAAO,yBAAyB,KAAK,CAAC,YAAY,QAAQ,KAAK,IAAI,CAAC;AACtE;AAMA,SAAS,iBAAiB,MAAuB;AAC/C,MAAI,cAAc,KAAK,IAAI,GAAG;AAC5B,WAAO;AAAA,EACT;AACA,QAAM,aAAa,cAAc;AACjC,aAAW,SAAS,KAAK,MAAM,mBAAmB,GAAG;AACnD,QAAI,UAAU,WAAW,IAAI,KAAK,KAAK,wBAAwB,IAAI,KAAK,IAAI;AAC1E,aAAO;AAAA,IACT;AAAA,EACF;AACA,SAAO;AACT;AASA,SAAS,kBAAkB,gBAAqC;AAC9D,QAAM,aAA0B,CAAC;AACjC,aAAW,CAAC,MAAM,IAAI,KAAK,gBAAgB,GAAG;AAC5C,QAAI,CAAC,eAAe,SAAS,IAAI,GAAG;AAClC;AAAA,IACF;AACA,UAAM,EAAE,QAAQ,IAAI;AACpB,UAAM,QAAS,KAAK,UAAU,iBAAiB,IAAI;AACnD,UAAM,YAAY;AAClB,QAAI,cAAc;AAClB,QAAI,QAAQ,MAAM,KAAK,cAAc;AACrC,WAAO,UAAU,QAAQ,cAAc,0BAA0B;AAC/D,iBAAW,KAAK,EAAE,KAAK,MAAM,QAAQ,MAAM,CAAC,EAAE,QAAQ,SAAS,OAAO,MAAM,MAAM,CAAC;AACnF,qBAAe;AACf,cAAQ,MAAM,KAAK,cAAc;AAAA,IACnC;AAAA,EACF;AACA,SAAO;AACT;AAaA,SAAS,gBAAgB,YAAsC;AAC7D,QAAM,SAAS,CAAC,GAAG,UAAU,EAAE;AAAA,IAC7B,CAAC,GAAG,MAAM,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,UAAU,EAAE,QAAQ,EAAE;AAAA,EAC/D;AACA,QAAM,WAAwB,CAAC;AAC/B,aAAW,aAAa,QAAQ;AAC9B,UAAM,SAAS,UAAU,MAAM,UAAU;AACzC,UAAM,YAAY,SAAS;AAAA,MACzB,CAAC,UACC,MAAM,SAAS,UAAU,SACzB,UAAU,OAAO,MAAM,OACvB,MAAM,MAAM,MAAM,QAAQ;AAAA,IAC9B;AACA,QAAI,CAAC,WAAW;AACd,eAAS,KAAK,SAAS;AAAA,IACzB;AAAA,EACF;AACA,SAAO;AACT;AAUO,SAAS,qBAAqB,SAA+B;AAClE,QAAM,YAAY,KAAK,IAAI;AAC3B,QAAM,iBAAiB,UAAU,OAAO;AACxC,QAAM,aAAa,oBAAI,IAAyB;AAChD,QAAM,eAAe,oBAAI,IAAqB;AAC9C,QAAM,aAAa,oBAAI,IAAqB;AAC5C,QAAM,aAAa,oBAAI,IAAqB;AAE5C,aAAW,aAAa,gBAAgB,kBAAkB,cAAc,CAAC,GAAG;AAC1E,UAAM,EAAE,KAAK,SAAS,OAAO,UAAU,IAAI,cAAc,gBAAgB,UAAU,KAAK;AAExF,QAAI,UAAU,aAAa,IAAI,SAAS;AACxC,QAAI,YAAY,QAAW;AACzB,gBAAU,kBAAkB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,mBAAa,IAAI,WAAW,OAAO;AAAA,IACrC;AACA,QAAI,SAAS;AACX;AAAA,IACF;AAEA,QAAI,WAAW,WAAW,IAAI,SAAS;AACvC,QAAI,aAAa,QAAW;AAC1B,iBAAW,iBAAiB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,iBAAW,IAAI,WAAW,QAAQ;AAAA,IACpC;AAEA,QAAI,QAAQ,WAAW,IAAI,SAAS;AACpC,QAAI,UAAU,QAAW;AACvB,cAAQ,gBAAgB,eAAe,MAAM,WAAW,OAAO,CAAC;AAChE,iBAAW,IAAI,WAAW,KAAK;AAAA,IACjC;AAEA,eAAW,SAAS,UAAU,SAAS;AACrC,UAAI,MAAM,aAAa,CAAC,UAAU;AAChC;AAAA,MACF;AAEA,UAAI,SAAS,oBAAoB,IAAI,MAAM,IAAI,GAAG;AAChD;AAAA,MACF;AAEA,UAAI,aAAa;AACjB,UAAI,MAAM,WAAW;AACnB,qBAAa;AAAA,MACf,WAAW,UAAU;AACnB,qBAAa;AAAA,MACf;AAEA,YAAM,WAAW,WAAW,IAAI,MAAM,IAAI;AAC1C,YAAM,SACJ,CAAC,YACD,aAAa,SAAS,cACrB,eAAe,SAAS,cAAc,UAAU,QAAQ,SAAS;AACpE,UAAI,QAAQ;AACV,mBAAW,IAAI,MAAM,MAAM;AAAA,UACzB,MAAM,MAAM;AAAA,UACZ;AAAA,UACA,OAAO,MAAM;AAAA,UACb,aAAa,MAAM;AAAA,UACnB,UAAU,UAAU;AAAA,QACtB,CAAC;AAAA,MACH;AAAA,IACF;AAAA,EACF;AAEA,QAAM,UAAU,MAAM,KAAK,WAAW,OAAO,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,WAAW,EAAE,QAAQ;AACtF,QAAM,aAAa,KAAK,IAAI,IAAI;AAEhC,SAAO;AAAA,IACL,mBAAmB,6BAA6B,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI,CAAC;AAAA,IAC1E;AAAA,IACA,OAAO;AAAA,MACL,cAAc,QAAQ;AAAA,MACtB;AAAA,MACA,eAAe,YAAY,EAAE;AAAA,IAC/B;AAAA,EACF;AACF;AAKO,SAAS,gBAAgB,QAAgC;AAC9D,SAAO,OAAO,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI;AACzC;","names":[]}
1
+ {"version":3,"sources":["../src/anchor.ts"],"sourcesContent":["/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"],"mappings":";AAaA;AAAA,EAEE;AAAA,EACA;AAAA,EACA;AAAA,OACK;AAwBA,IAAM,4BAA4B;AAMlC,IAAM,uBAAuB;AAM7B,IAAM,uBAAuB;AAGpC,IAAM,2BAA2B;AASjC,SAAS,UAAU,MAAsB;AACvC,SAAO,KACJ,UAAU,KAAK,EACf,QAAQ,oBAAoB,EAAE,EAC9B,YAAY,EACZ,QAAQ,aAAa,GAAG;AAC7B;AAEA,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAQD,IAAM,yBAAyB,oBAAI,IAAI;AAAA,EACrC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAOD,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AASD,IAAM,2BAAqC;AAAA,EACzC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AASA,IAAM,sBAAsB,oBAAI,IAAI;AAAA,EAClC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAGD,IAAM,yBAAmC;AAAA,EACvC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAOA,IAAM,4BAAsC,CAAC,eAAe,oBAAoB,aAAa;AAa7F,IAAM,mBAAmB;AAEzB,SAAS,gBAAgB,MAAuB;AAC9C,MAAI,0BAA0B,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC,GAAG;AACzD,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC,KAAK,iBAAiB,KAAK,IAAI;AACzF;AAEA,IAAM,gBAAgB;AAGtB,IAAI,mBAAuC;AAE3C,SAAS,gBAA6B;AACpC,MAAI,CAAC,kBAAkB;AACrB,uBAAmB,IAAI;AAAA,MACrB,OAAO,KAAK,YAAY,EACrB,IAAI,CAAC,SAAS,UAAU,IAAI,EAAE,KAAK,CAAC,EACpC,OAAO,OAAO;AAAA,IACnB;AAAA,EACF;AACA,SAAO;AACT;AAqBA,IAAI,iBAAkD;AACtD,IAAI,qBAAsD;AAE1D,SAAS,cAAwC;AAC/C,MAAI,CAAC,gBAAgB;AACnB,qBAAiB,qBAAqB;AAAA,EACxC;AACA,SAAO;AACT;AAEA,SAAS,aAAa,MAAsB;AAC1C,SAAO,KAAK,QAAQ,uBAAuB,MAAM;AACnD;AAkBA,SAAS,iBAAiB,gBAAgC;AACxD,QAAM,OAAO,eAAe,MAAM,GAAG,EAAE,IAAI,YAAY,EAAE,KAAK,YAAY;AAC1E,QAAM,SAAS,UAAU,KAAK,cAAc,IAAI,OAAO;AACvD,SAAO,IAAI,OAAO,sBAAsB,IAAI,GAAG,MAAM,sBAAsB,IAAI;AACjF;AAEA,SAAS,mBAAmB,SAA0C;AACpE,QAAM,aAAa,MAAM,QAAQ,QAAQ,QAAQ,IAAI,QAAQ,WAAW,CAAC,QAAQ,QAAQ;AACzF,SAAO,WAAW,SAAS,OAAO,KAAK,CAAC,QAAQ;AAClD;AAOA,SAAS,gBAAgB,gBAAwB,SAA0C;AACzF,MAAI,eAAe,SAAS,GAAG,GAAG;AAChC,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,IAAI,cAAc,KAAK,mBAAmB,OAAO;AACjF;AAEA,SAAS,kBAA4C;AACnD,MAAI,CAAC,oBAAoB;AACvB,UAAM,MAAM,oBAAI,IAAyB;AACzC,eAAW,WAAW,YAAY,GAAG;AACnC,iBAAW,QAAQ,QAAQ,OAAO;AAChC,cAAM,aAAa,UAAU,IAAI,EAAE,KAAK;AACxC,YAAI,CAAC,YAAY;AACf;AAAA,QACF;AACA,YAAI,WAAW,SAAS,KAAK,CAAC,wBAAwB,IAAI,UAAU,GAAG;AACrE;AAAA,QACF;AACA,YAAI,OAAO,IAAI,IAAI,UAAU;AAC7B,YAAI,CAAC,MAAM;AACT,iBAAO,EAAE,SAAS,CAAC,GAAG,OAAO,KAAK;AAClC,cAAI,IAAI,YAAY,IAAI;AAAA,QAC1B;AACA,aAAK,QAAQ,KAAK;AAAA,UAChB,WAAW,gBAAgB,YAAY,OAAO;AAAA,UAC9C,MAAM,QAAQ;AAAA,UACd,GAAI,QAAQ,SAAS,EAAE,OAAO,QAAQ,MAAM;AAAA,UAC5C,UAAU;AAAA,QACZ,CAAC;AAAA,MACH;AAAA,IACF;AACA,yBAAqB;AAAA,EACvB;AACA,SAAO;AACT;AAEA,SAAS,cAAc,MAAc,UAAkD;AACrF,QAAM,QAAQ,KAAK,YAAY,MAAM,QAAQ,IAAI;AACjD,QAAM,YAAY,KAAK,QAAQ,MAAM,QAAQ;AAC7C,SAAO,EAAE,KAAK,cAAc,KAAK,KAAK,SAAS,WAAW,MAAM;AAClE;AAEA,SAAS,kBAAkB,MAAuB;AAChD,SAAO,yBAAyB,KAAK,CAAC,YAAY,QAAQ,KAAK,IAAI,CAAC;AACtE;AAMA,SAAS,iBAAiB,MAAuB;AAC/C,MAAI,cAAc,KAAK,IAAI,GAAG;AAC5B,WAAO;AAAA,EACT;AACA,QAAM,aAAa,cAAc;AACjC,aAAW,SAAS,KAAK,MAAM,mBAAmB,GAAG;AACnD,QAAI,UAAU,WAAW,IAAI,KAAK,KAAK,wBAAwB,IAAI,KAAK,IAAI;AAC1E,aAAO;AAAA,IACT;AAAA,EACF;AACA,SAAO;AACT;AASA,SAAS,kBAAkB,gBAAqC;AAC9D,QAAM,aAA0B,CAAC;AACjC,aAAW,CAAC,MAAM,IAAI,KAAK,gBAAgB,GAAG;AAC5C,QAAI,CAAC,eAAe,SAAS,IAAI,GAAG;AAClC;AAAA,IACF;AACA,UAAM,EAAE,QAAQ,IAAI;AACpB,UAAM,QAAS,KAAK,UAAU,iBAAiB,IAAI;AACnD,UAAM,YAAY;AAClB,QAAI,cAAc;AAClB,QAAI,QAAQ,MAAM,KAAK,cAAc;AACrC,WAAO,UAAU,QAAQ,cAAc,0BAA0B;AAC/D,iBAAW,KAAK,EAAE,KAAK,MAAM,QAAQ,MAAM,CAAC,EAAE,QAAQ,SAAS,OAAO,MAAM,MAAM,CAAC;AACnF,qBAAe;AACf,cAAQ,MAAM,KAAK,cAAc;AAAA,IACnC;AAAA,EACF;AACA,SAAO;AACT;AAaA,SAAS,gBAAgB,YAAsC;AAC7D,QAAM,SAAS,CAAC,GAAG,UAAU,EAAE;AAAA,IAC7B,CAAC,GAAG,MAAM,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,UAAU,EAAE,QAAQ,EAAE;AAAA,EAC/D;AACA,QAAM,WAAwB,CAAC;AAC/B,aAAW,aAAa,QAAQ;AAC9B,UAAM,SAAS,UAAU,MAAM,UAAU;AACzC,UAAM,YAAY,SAAS;AAAA,MACzB,CAAC,UACC,MAAM,SAAS,UAAU,SACzB,UAAU,OAAO,MAAM,OACvB,MAAM,MAAM,MAAM,QAAQ;AAAA,IAC9B;AACA,QAAI,CAAC,WAAW;AACd,eAAS,KAAK,SAAS;AAAA,IACzB;AAAA,EACF;AACA,SAAO;AACT;AAUO,SAAS,qBAAqB,SAA+B;AAClE,QAAM,YAAY,KAAK,IAAI;AAC3B,QAAM,iBAAiB,UAAU,OAAO;AACxC,QAAM,aAAa,oBAAI,IAAyB;AAChD,QAAM,eAAe,oBAAI,IAAqB;AAC9C,QAAM,aAAa,oBAAI,IAAqB;AAC5C,QAAM,aAAa,oBAAI,IAAqB;AAE5C,aAAW,aAAa,gBAAgB,kBAAkB,cAAc,CAAC,GAAG;AAC1E,UAAM,EAAE,KAAK,SAAS,OAAO,UAAU,IAAI,cAAc,gBAAgB,UAAU,KAAK;AAExF,QAAI,UAAU,aAAa,IAAI,SAAS;AACxC,QAAI,YAAY,QAAW;AACzB,gBAAU,kBAAkB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,mBAAa,IAAI,WAAW,OAAO;AAAA,IACrC;AACA,QAAI,SAAS;AACX;AAAA,IACF;AAEA,QAAI,WAAW,WAAW,IAAI,SAAS;AACvC,QAAI,aAAa,QAAW;AAC1B,iBAAW,iBAAiB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,iBAAW,IAAI,WAAW,QAAQ;AAAA,IACpC;AAEA,QAAI,QAAQ,WAAW,IAAI,SAAS;AACpC,QAAI,UAAU,QAAW;AACvB,cAAQ,gBAAgB,eAAe,MAAM,WAAW,OAAO,CAAC;AAChE,iBAAW,IAAI,WAAW,KAAK;AAAA,IACjC;AAEA,eAAW,SAAS,UAAU,SAAS;AACrC,UAAI,MAAM,aAAa,CAAC,UAAU;AAChC;AAAA,MACF;AAEA,UAAI,SAAS,oBAAoB,IAAI,MAAM,IAAI,GAAG;AAChD;AAAA,MACF;AAEA,UAAI,aAAa;AACjB,UAAI,MAAM,WAAW;AACnB,qBAAa;AAAA,MACf,WAAW,UAAU;AACnB,qBAAa;AAAA,MACf;AAEA,YAAM,WAAW,WAAW,IAAI,MAAM,IAAI;AAC1C,YAAM,SACJ,CAAC,YACD,aAAa,SAAS,cACrB,eAAe,SAAS,cAAc,UAAU,QAAQ,SAAS;AACpE,UAAI,QAAQ;AACV,mBAAW,IAAI,MAAM,MAAM;AAAA,UACzB,MAAM,MAAM;AAAA,UACZ;AAAA,UACA,OAAO,MAAM;AAAA,UACb,aAAa,MAAM;AAAA,UACnB,UAAU,UAAU;AAAA,QACtB,CAAC;AAAA,MACH;AAAA,IACF;AAAA,EACF;AAEA,QAAM,UAAU,MAAM,KAAK,WAAW,OAAO,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,WAAW,EAAE,QAAQ;AACtF,QAAM,aAAa,KAAK,IAAI,IAAI;AAEhC,SAAO;AAAA,IACL,mBAAmB,6BAA6B,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI,CAAC;AAAA,IAC1E;AAAA,IACA,OAAO;AAAA,MACL,cAAc,QAAQ;AAAA,MACtB;AAAA,MACA,eAAe,YAAY,EAAE;AAAA,IAC/B;AAAA,EACF;AACF;AAKO,SAAS,gBAAgB,QAAgC;AAC9D,SAAO,OAAO,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI;AACzC;","names":[]}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@precisa-saude/fhir-ocr-utils",
3
- "version": "0.20.3",
3
+ "version": "0.21.0",
4
4
  "description": "Utilitários de ancoragem OCR para extração de biomarcadores de PDFs de resultados laboratoriais",
5
5
  "keywords": [
6
6
  "fhir",
@@ -41,7 +41,7 @@
41
41
  "dist"
42
42
  ],
43
43
  "dependencies": {
44
- "@precisa-saude/fhir": "^0.20.3"
44
+ "@precisa-saude/fhir": "^0.21.0"
45
45
  },
46
46
  "devDependencies": {
47
47
  "tsup": "^8.3.5",