@precisa-saude/fhir-ocr-utils 0.21.0 → 0.21.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +2 -2
- package/dist/index.cjs +1 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
package/dist/cli.js
CHANGED
|
@@ -18,7 +18,7 @@ var CONFIDENCE_NAME_ONLY = 0.7;
|
|
|
18
18
|
var CONFIDENCE_AMBIGUOUS = 0.4;
|
|
19
19
|
var MAX_OCCURRENCES_PER_NAME = 5;
|
|
20
20
|
function normalize(text) {
|
|
21
|
-
return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/[^\S\n]+/g, " ");
|
|
21
|
+
return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/(?<=\p{L})-(?=[\p{L}\p{N}])/gu, " ").replace(/[^\S\n]+/g, " ");
|
|
22
22
|
}
|
|
23
23
|
var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
|
|
24
24
|
"hdl",
|
|
@@ -449,7 +449,7 @@ async function main() {
|
|
|
449
449
|
strict: false
|
|
450
450
|
});
|
|
451
451
|
if (values.version) {
|
|
452
|
-
process.stdout.write(`${"0.21.
|
|
452
|
+
process.stdout.write(`${"0.21.1"}
|
|
453
453
|
`);
|
|
454
454
|
return;
|
|
455
455
|
}
|
package/dist/index.cjs
CHANGED
|
@@ -9,7 +9,7 @@ var CONFIDENCE_NAME_ONLY = 0.7;
|
|
|
9
9
|
var CONFIDENCE_AMBIGUOUS = 0.4;
|
|
10
10
|
var MAX_OCCURRENCES_PER_NAME = 5;
|
|
11
11
|
function normalize(text) {
|
|
12
|
-
return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/[^\S\n]+/g, " ");
|
|
12
|
+
return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/(?<=\p{L})-(?=[\p{L}\p{N}])/gu, " ").replace(/[^\S\n]+/g, " ");
|
|
13
13
|
}
|
|
14
14
|
var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
|
|
15
15
|
"hdl",
|
package/dist/index.cjs.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts"],"names":[],"mappings":"AAAA;ACaA;AAEE;AACA;AACA;AAAA,2CACK;AAwBA,IAAM,0BAAA,EAA4B,CAAA;AAMlC,IAAM,qBAAA,EAAuB,GAAA;AAM7B,IAAM,qBAAA,EAAuB,GAAA;AAGpC,IAAM,yBAAA,EAA2B,CAAA;AASjC,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,WAAA,EAAa,GAAG,CAAA;AAC7B;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAQD,IAAM,uBAAA,kBAAyB,IAAI,GAAA,CAAI;AAAA,EACrC,UAAA;AAAA;AAAA,EACA,WAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,YAAA;AAAA;AAAA,EACA,KAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,UAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,YAAA;AAAA,EACA,OAAA;AAAA,EACA,SAAA;AAAA,EACA;AACF,CAAC,CAAA;AAOD,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,QAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,OAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,cAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,OAAA;AAAA,EACA,UAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,QAAA;AAAA,EACA,OAAA;AAAA,EACA,cAAA;AAAA,EACA;AACF,CAAC,CAAA;AASD,IAAM,yBAAA,EAAqC;AAAA,EACzC,qBAAA;AAAA;AAAA,EACA,kBAAA;AAAA;AAAA,EACA,iBAAA;AAAA;AAAA,EACA,uBAAA;AAAA;AAAA,EACA,cAAA;AAAA;AAAA,EACA,YAAA;AAAA,EACA,iBAAA;AAAA,EACA,YAAA;AAAA,EACA,cAAA;AAAA,EACA,gBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,gBAAA;AAAA,EACA,mBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,aAAA;AAAA,EACA,WAAA;AAAA,EACA,eAAA;AAAA,EACA,aAAA;AAAA,EACA;AACF,CAAA;AASA,IAAM,oBAAA,kBAAsB,IAAI,GAAA,CAAI;AAAA,EAClC,mBAAA;AAAA,EACA,eAAA;AAAA,EACA,qBAAA;AAAA,EACA,qBAAA;AAAA,EACA,oBAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAC,CAAA;AAGD,IAAM,uBAAA,EAAmC;AAAA,EACvC,mBAAA;AAAA,EACA,sBAAA;AAAA,EACA,iBAAA;AAAA,EACA;AACF,CAAA;AAOA,IAAM,0BAAA,EAAsC,CAAC,aAAA,EAAe,kBAAA,EAAoB,aAAa,CAAA;AAa7F,IAAM,iBAAA,EAAmB,uBAAA;AAEzB,SAAS,eAAA,CAAgB,IAAA,EAAuB;AAC9C,EAAA,GAAA,CAAI,yBAAA,CAA0B,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,CAAA,EAAG;AACzD,IAAA,OAAO,KAAA;AAAA,EACT;AACA,EAAA,OAAO,sBAAA,CAAuB,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,EAAA,GAAK,gBAAA,CAAiB,IAAA,CAAK,IAAI,CAAA;AACzF;AAEA,IAAM,cAAA,EAAgB,IAAA;AAGtB,IAAI,iBAAA,EAAuC,IAAA;AAE3C,SAAS,aAAA,CAAA,EAA6B;AACpC,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,iBAAA,EAAmB,IAAI,GAAA;AAAA,MACrB,MAAA,CAAO,IAAA,CAAK,kBAAY,CAAA,CACrB,GAAA,CAAI,CAAC,IAAA,EAAA,GAAS,SAAA,CAAU,IAAI,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,CACpC,MAAA,CAAO,OAAO;AAAA,IACnB,CAAA;AAAA,EACF;AACA,EAAA,OAAO,gBAAA;AACT;AAqBA,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,mBAAA,EAAsD,IAAA;AAE1D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,YAAA,CAAa,IAAA,EAAsB;AAC1C,EAAA,OAAO,IAAA,CAAK,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AACnD;AAkBA,SAAS,gBAAA,CAAiB,cAAA,EAAgC;AACxD,EAAA,MAAM,KAAA,EAAO,cAAA,CAAe,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,YAAY,CAAA,CAAE,IAAA,CAAK,YAAY,CAAA;AAC1E,EAAA,MAAM,OAAA,EAAS,SAAA,CAAU,IAAA,CAAK,cAAc,EAAA,EAAI,KAAA,EAAO,EAAA;AACvD,EAAA,OAAO,IAAI,MAAA,CAAO,CAAA,mBAAA,EAAsB,IAAI,CAAA,EAAA;AAC9C;AAEsE;AAC3B,EAAA;AACD,EAAA;AAC1C;AAOiD;AACb,EAAA;AACzB,IAAA;AACT,EAAA;AACkC,EAAA;AACpC;AAEqD;AAC1B,EAAA;AACkB,IAAA;AACJ,IAAA;AACD,MAAA;AACG,QAAA;AAClB,QAAA;AACf,UAAA;AACF,QAAA;AAC8B,QAAA;AAC5B,UAAA;AACF,QAAA;AAC6B,QAAA;AAClB,QAAA;AACyB,UAAA;AACV,UAAA;AAC1B,QAAA;AACkB,QAAA;AACW,UAAA;AACb,UAAA;AACwB,UAAA;AAC5B,UAAA;AACX,QAAA;AACH,MAAA;AACF,IAAA;AACqB,IAAA;AACvB,EAAA;AACO,EAAA;AACT;AAEuF;AACxC,EAAA;AACA,EAAA;AACP,EAAA;AACxC;AAEkD;AACV,EAAA;AACxC;AAMiD;AACjB,EAAA;AACrB,IAAA;AACT,EAAA;AACiC,EAAA;AACF,EAAA;AACU,IAAA;AAC9B,MAAA;AACT,IAAA;AACF,EAAA;AACO,EAAA;AACT;AASgE;AAC7B,EAAA;AACU,EAAA;AACL,IAAA;AAClC,MAAA;AACF,IAAA;AACoB,IAAA;AACU,IAAA;AACZ,IAAA;AACA,IAAA;AACmB,IAAA;AACE,IAAA;AACA,MAAA;AACtB,MAAA;AACkB,MAAA;AACnC,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAa+D;AAC9B,EAAA;AACU,IAAA;AACzC,EAAA;AAC+B,EAAA;AACC,EAAA;AACW,IAAA;AACd,IAAA;AAEE,MAAA;AAG7B,IAAA;AACgB,IAAA;AACS,MAAA;AACzB,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAUoE;AACvC,EAAA;AACa,EAAA;AACQ,EAAA;AACF,EAAA;AACF,EAAA;AACA,EAAA;AAEJ,EAAA;AACK,IAAA;AAEH,IAAA;AACb,IAAA;AACG,MAAA;AACO,MAAA;AACrC,IAAA;AACa,IAAA;AACX,MAAA;AACF,IAAA;AAEuC,IAAA;AACX,IAAA;AACE,MAAA;AACM,MAAA;AACpC,IAAA;AAEoC,IAAA;AACX,IAAA;AACgB,MAAA;AACR,MAAA;AACjC,IAAA;AAEuC,IAAA;AACH,MAAA;AAChC,QAAA;AACF,MAAA;AAEqC,MAAA;AACnC,QAAA;AACF,MAAA;AAEiB,MAAA;AACI,MAAA;AACN,QAAA;AACM,MAAA;AACN,QAAA;AACf,MAAA;AAEsC,MAAA;AAGvB,MAAA;AAEH,MAAA;AACiB,QAAA;AACb,UAAA;AACZ,UAAA;AACa,UAAA;AACM,UAAA;AACC,UAAA;AACrB,QAAA;AACH,MAAA;AACF,IAAA;AACF,EAAA;AAE6C,EAAA;AACb,EAAA;AAEzB,EAAA;AACc,IAAA;AACnB,IAAA;AACO,IAAA;AACiB,MAAA;AACtB,MAAA;AAC6B,MAAA;AAC/B,IAAA;AACF,EAAA;AACF;AAKgE;AACvB,EAAA;AACzC;AD9M+C;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"]}
|
|
1
|
+
{"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts"],"names":[],"mappings":"AAAA;ACaA;AAEE;AACA;AACA;AAAA,2CACK;AAwBA,IAAM,0BAAA,EAA4B,CAAA;AAMlC,IAAM,qBAAA,EAAuB,GAAA;AAM7B,IAAM,qBAAA,EAAuB,GAAA;AAGpC,IAAM,yBAAA,EAA2B,CAAA;AA8BjC,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,+BAAA,EAAiC,GAAG,CAAA,CAC5C,OAAA,CAAQ,WAAA,EAAa,GAAG,CAAA;AAC7B;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAQD,IAAM,uBAAA,kBAAyB,IAAI,GAAA,CAAI;AAAA,EACrC,UAAA;AAAA;AAAA,EACA,WAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,YAAA;AAAA;AAAA,EACA,KAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,UAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,YAAA;AAAA,EACA,OAAA;AAAA,EACA,SAAA;AAAA,EACA;AACF,CAAC,CAAA;AAOD,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,QAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,OAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,cAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,OAAA;AAAA,EACA,UAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,QAAA;AAAA,EACA,OAAA;AAAA,EACA,cAAA;AAAA,EACA;AACF,CAAC,CAAA;AASD,IAAM,yBAAA,EAAqC;AAAA,EACzC,qBAAA;AAAA;AAAA,EACA,kBAAA;AAAA;AAAA,EACA,iBAAA;AAAA;AAAA,EACA,uBAAA;AAAA;AAAA,EACA,cAAA;AAAA;AAAA,EACA,YAAA;AAAA,EACA,iBAAA;AAAA,EACA,YAAA;AAAA,EACA,cAAA;AAAA,EACA,gBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,gBAAA;AAAA,EACA,mBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,aAAA;AAAA,EACA,WAAA;AAAA,EACA,eAAA;AAAA,EACA,aAAA;AAAA,EACA;AACF,CAAA;AASA,IAAM,oBAAA,kBAAsB,IAAI,GAAA,CAAI;AAAA,EAClC,mBAAA;AAAA,EACA,eAAA;AAAA,EACA,qBAAA;AAAA,EACA,qBAAA;AAAA,EACA,oBAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAC,CAAA;AAGD,IAAM,uBAAA,EAAmC;AAAA,EACvC,mBAAA;AAAA,EACA,sBAAA;AAAA,EACA,iBAAA;AAAA,EACA;AACF,CAAA;AAOA,IAAM,0BAAA,EAAsC,CAAC,aAAA,EAAe,kBAAA,EAAoB,aAAa,CAAA;AAa7F,IAAM,iBAAA,EAAmB,uBAAA;AAEzB,SAAS,eAAA,CAAgB,IAAA,EAAuB;AAC9C,EAAA,GAAA,CAAI,yBAAA,CAA0B,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,CAAA,EAAG;AACzD,IAAA,OAAO,KAAA;AAAA,EACT;AACA,EAAA,OAAO,sBAAA,CAAuB,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,EAAA,GAAK,gBAAA,CAAiB,IAAA,CAAK,IAAI,CAAA;AACzF;AAEA,IAAM,cAAA,EAAgB,IAAA;AAGtB,IAAI,iBAAA,EAAuC,IAAA;AAE3C,SAAS,aAAA,CAAA,EAA6B;AACpC,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,iBAAA,EAAmB,IAAI,GAAA;AAAA,MACrB,MAAA,CAAO,IAAA,CAAK,kBAAY,CAAA,CACrB,GAAA,CAAI,CAAC,IAAA,EAAA,GAAS,SAAA,CAAU,IAAI,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,CACpC,MAAA,CAAO,OAAO;AAAA,IACnB,CAAA;AAAA,EACF;AACA,EAAA,OAAO,gBAAA;AACT;AAqBA,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,mBAAA,EAAsD,IAAA;AAE1D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,YAAA,CAAa,IAAA,EAAsB;AAC1C,EAAA,OAAO,IAAA,CAAK,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AACnD;AAkBA,SAAS,gBAAA,CAAiB,cAAA,EAAgC;AACxD,EAAA,MAAM,KAAA,EAAO,cAAA,CAAe,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,YAAY,CAAA,CAAE,IAAA,CAAK,YAAY,CAAA;AAC1E,EAAA,MAAM,OAAA,EAAS,SAAA,CAAU,IAAA,CAAK,cAAc,EAAA,EAAI,KAAA,EAAO,EAAA;AACvD,EAAA,OAAO,IAAI,MAAA,CAAO,CAAA,mBAAA,EAAsB,IAAI,CAAA,EAAA;AAC9C;AAEsE;AAC3B,EAAA;AACD,EAAA;AAC1C;AAOiD;AACb,EAAA;AACzB,IAAA;AACT,EAAA;AACkC,EAAA;AACpC;AAEqD;AAC1B,EAAA;AACkB,IAAA;AACJ,IAAA;AACD,MAAA;AACG,QAAA;AAClB,QAAA;AACf,UAAA;AACF,QAAA;AAC8B,QAAA;AAC5B,UAAA;AACF,QAAA;AAC6B,QAAA;AAClB,QAAA;AACyB,UAAA;AACV,UAAA;AAC1B,QAAA;AACkB,QAAA;AACW,UAAA;AACb,UAAA;AACwB,UAAA;AAC5B,UAAA;AACX,QAAA;AACH,MAAA;AACF,IAAA;AACqB,IAAA;AACvB,EAAA;AACO,EAAA;AACT;AAEuF;AACxC,EAAA;AACA,EAAA;AACP,EAAA;AACxC;AAEkD;AACV,EAAA;AACxC;AAMiD;AACjB,EAAA;AACrB,IAAA;AACT,EAAA;AACiC,EAAA;AACF,EAAA;AACU,IAAA;AAC9B,MAAA;AACT,IAAA;AACF,EAAA;AACO,EAAA;AACT;AASgE;AAC7B,EAAA;AACU,EAAA;AACL,IAAA;AAClC,MAAA;AACF,IAAA;AACoB,IAAA;AACU,IAAA;AACZ,IAAA;AACA,IAAA;AACmB,IAAA;AACE,IAAA;AACA,MAAA;AACtB,MAAA;AACkB,MAAA;AACnC,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAa+D;AAC9B,EAAA;AACU,IAAA;AACzC,EAAA;AAC+B,EAAA;AACC,EAAA;AACW,IAAA;AACd,IAAA;AAEE,MAAA;AAG7B,IAAA;AACgB,IAAA;AACS,MAAA;AACzB,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAUoE;AACvC,EAAA;AACa,EAAA;AACQ,EAAA;AACF,EAAA;AACF,EAAA;AACA,EAAA;AAEJ,EAAA;AACK,IAAA;AAEH,IAAA;AACb,IAAA;AACG,MAAA;AACO,MAAA;AACrC,IAAA;AACa,IAAA;AACX,MAAA;AACF,IAAA;AAEuC,IAAA;AACX,IAAA;AACE,MAAA;AACM,MAAA;AACpC,IAAA;AAEoC,IAAA;AACX,IAAA;AACgB,MAAA;AACR,MAAA;AACjC,IAAA;AAEuC,IAAA;AACH,MAAA;AAChC,QAAA;AACF,MAAA;AAEqC,MAAA;AACnC,QAAA;AACF,MAAA;AAEiB,MAAA;AACI,MAAA;AACN,QAAA;AACM,MAAA;AACN,QAAA;AACf,MAAA;AAEsC,MAAA;AAGvB,MAAA;AAEH,MAAA;AACiB,QAAA;AACb,UAAA;AACZ,UAAA;AACa,UAAA;AACM,UAAA;AACC,UAAA;AACrB,QAAA;AACH,MAAA;AACF,IAAA;AACF,EAAA;AAE6C,EAAA;AACb,EAAA;AAEzB,EAAA;AACc,IAAA;AACnB,IAAA;AACO,IAAA;AACiB,MAAA;AACtB,MAAA;AAC6B,MAAA;AAC/B,IAAA;AACF,EAAA;AACF;AAKgE;AACvB,EAAA;AACzC;ADpO+C;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Treats a hyphen that joins words as a space\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n *\n * O hífen entre palavras vira espaço porque o catálogo e o laboratório\n * discordam sobre ele o tempo todo: o catálogo escreve \"Proteína C-Reativa\" e\n * \"High-Density Lipoprotein\", e os laudos imprimem \"Proteína C Reativa\" e\n * \"High Density Lipoprotein\". Sem essa equivalência, 82 dos 159 nomes com\n * hífen deixam de ancorar na grafia que o documento usa.\n *\n * Não era teórico: um GGT de verdade foi descartado como alucinação em 27\n * laudos porque o documento escrevia \"Gama glutamil transferase\" e o catálogo\n * \"Gama-Glutamil Transferase\". Um caractere derrubava o valor antes de\n * qualquer validação.\n *\n * A troca exige **letra antes** do hífen, e por isso não toca em número:\n * o `-2.5` de um T-score e o `0-5` de uma faixa de urina seguem intactos.\n * Trocar sem essa guarda apagaria o sinal de um valor negativo, que é bem\n * pior que o problema original.\n *\n * O pré-filtro de substring do `collectCandidates` compara a chave com o texto\n * já normalizado, então a equivalência precisa nascer aqui: aplicada só na\n * regex, o `includes` descartaria o nome antes de ela rodar.\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/(?<=\\p{L})-(?=[\\p{L}\\p{N}])/gu, ' ')\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"]}
|
package/dist/index.js
CHANGED
|
@@ -9,7 +9,7 @@ var CONFIDENCE_NAME_ONLY = 0.7;
|
|
|
9
9
|
var CONFIDENCE_AMBIGUOUS = 0.4;
|
|
10
10
|
var MAX_OCCURRENCES_PER_NAME = 5;
|
|
11
11
|
function normalize(text) {
|
|
12
|
-
return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/[^\S\n]+/g, " ");
|
|
12
|
+
return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/(?<=\p{L})-(?=[\p{L}\p{N}])/gu, " ").replace(/[^\S\n]+/g, " ");
|
|
13
13
|
}
|
|
14
14
|
var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
|
|
15
15
|
"hdl",
|
package/dist/index.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/anchor.ts"],"sourcesContent":["/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"],"mappings":";AAaA;AAAA,EAEE;AAAA,EACA;AAAA,EACA;AAAA,OACK;AAwBA,IAAM,4BAA4B;AAMlC,IAAM,uBAAuB;AAM7B,IAAM,uBAAuB;AAGpC,IAAM,2BAA2B;AASjC,SAAS,UAAU,MAAsB;AACvC,SAAO,KACJ,UAAU,KAAK,EACf,QAAQ,oBAAoB,EAAE,EAC9B,YAAY,EACZ,QAAQ,aAAa,GAAG;AAC7B;AAEA,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAQD,IAAM,yBAAyB,oBAAI,IAAI;AAAA,EACrC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAOD,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AASD,IAAM,2BAAqC;AAAA,EACzC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AASA,IAAM,sBAAsB,oBAAI,IAAI;AAAA,EAClC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAGD,IAAM,yBAAmC;AAAA,EACvC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAOA,IAAM,4BAAsC,CAAC,eAAe,oBAAoB,aAAa;AAa7F,IAAM,mBAAmB;AAEzB,SAAS,gBAAgB,MAAuB;AAC9C,MAAI,0BAA0B,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC,GAAG;AACzD,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC,KAAK,iBAAiB,KAAK,IAAI;AACzF;AAEA,IAAM,gBAAgB;AAGtB,IAAI,mBAAuC;AAE3C,SAAS,gBAA6B;AACpC,MAAI,CAAC,kBAAkB;AACrB,uBAAmB,IAAI;AAAA,MACrB,OAAO,KAAK,YAAY,EACrB,IAAI,CAAC,SAAS,UAAU,IAAI,EAAE,KAAK,CAAC,EACpC,OAAO,OAAO;AAAA,IACnB;AAAA,EACF;AACA,SAAO;AACT;AAqBA,IAAI,iBAAkD;AACtD,IAAI,qBAAsD;AAE1D,SAAS,cAAwC;AAC/C,MAAI,CAAC,gBAAgB;AACnB,qBAAiB,qBAAqB;AAAA,EACxC;AACA,SAAO;AACT;AAEA,SAAS,aAAa,MAAsB;AAC1C,SAAO,KAAK,QAAQ,uBAAuB,MAAM;AACnD;AAkBA,SAAS,iBAAiB,gBAAgC;AACxD,QAAM,OAAO,eAAe,MAAM,GAAG,EAAE,IAAI,YAAY,EAAE,KAAK,YAAY;AAC1E,QAAM,SAAS,UAAU,KAAK,cAAc,IAAI,OAAO;AACvD,SAAO,IAAI,OAAO,sBAAsB,IAAI,GAAG,MAAM,sBAAsB,IAAI;AACjF;AAEA,SAAS,mBAAmB,SAA0C;AACpE,QAAM,aAAa,MAAM,QAAQ,QAAQ,QAAQ,IAAI,QAAQ,WAAW,CAAC,QAAQ,QAAQ;AACzF,SAAO,WAAW,SAAS,OAAO,KAAK,CAAC,QAAQ;AAClD;AAOA,SAAS,gBAAgB,gBAAwB,SAA0C;AACzF,MAAI,eAAe,SAAS,GAAG,GAAG;AAChC,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,IAAI,cAAc,KAAK,mBAAmB,OAAO;AACjF;AAEA,SAAS,kBAA4C;AACnD,MAAI,CAAC,oBAAoB;AACvB,UAAM,MAAM,oBAAI,IAAyB;AACzC,eAAW,WAAW,YAAY,GAAG;AACnC,iBAAW,QAAQ,QAAQ,OAAO;AAChC,cAAM,aAAa,UAAU,IAAI,EAAE,KAAK;AACxC,YAAI,CAAC,YAAY;AACf;AAAA,QACF;AACA,YAAI,WAAW,SAAS,KAAK,CAAC,wBAAwB,IAAI,UAAU,GAAG;AACrE;AAAA,QACF;AACA,YAAI,OAAO,IAAI,IAAI,UAAU;AAC7B,YAAI,CAAC,MAAM;AACT,iBAAO,EAAE,SAAS,CAAC,GAAG,OAAO,KAAK;AAClC,cAAI,IAAI,YAAY,IAAI;AAAA,QAC1B;AACA,aAAK,QAAQ,KAAK;AAAA,UAChB,WAAW,gBAAgB,YAAY,OAAO;AAAA,UAC9C,MAAM,QAAQ;AAAA,UACd,GAAI,QAAQ,SAAS,EAAE,OAAO,QAAQ,MAAM;AAAA,UAC5C,UAAU;AAAA,QACZ,CAAC;AAAA,MACH;AAAA,IACF;AACA,yBAAqB;AAAA,EACvB;AACA,SAAO;AACT;AAEA,SAAS,cAAc,MAAc,UAAkD;AACrF,QAAM,QAAQ,KAAK,YAAY,MAAM,QAAQ,IAAI;AACjD,QAAM,YAAY,KAAK,QAAQ,MAAM,QAAQ;AAC7C,SAAO,EAAE,KAAK,cAAc,KAAK,KAAK,SAAS,WAAW,MAAM;AAClE;AAEA,SAAS,kBAAkB,MAAuB;AAChD,SAAO,yBAAyB,KAAK,CAAC,YAAY,QAAQ,KAAK,IAAI,CAAC;AACtE;AAMA,SAAS,iBAAiB,MAAuB;AAC/C,MAAI,cAAc,KAAK,IAAI,GAAG;AAC5B,WAAO;AAAA,EACT;AACA,QAAM,aAAa,cAAc;AACjC,aAAW,SAAS,KAAK,MAAM,mBAAmB,GAAG;AACnD,QAAI,UAAU,WAAW,IAAI,KAAK,KAAK,wBAAwB,IAAI,KAAK,IAAI;AAC1E,aAAO;AAAA,IACT;AAAA,EACF;AACA,SAAO;AACT;AASA,SAAS,kBAAkB,gBAAqC;AAC9D,QAAM,aAA0B,CAAC;AACjC,aAAW,CAAC,MAAM,IAAI,KAAK,gBAAgB,GAAG;AAC5C,QAAI,CAAC,eAAe,SAAS,IAAI,GAAG;AAClC;AAAA,IACF;AACA,UAAM,EAAE,QAAQ,IAAI;AACpB,UAAM,QAAS,KAAK,UAAU,iBAAiB,IAAI;AACnD,UAAM,YAAY;AAClB,QAAI,cAAc;AAClB,QAAI,QAAQ,MAAM,KAAK,cAAc;AACrC,WAAO,UAAU,QAAQ,cAAc,0BAA0B;AAC/D,iBAAW,KAAK,EAAE,KAAK,MAAM,QAAQ,MAAM,CAAC,EAAE,QAAQ,SAAS,OAAO,MAAM,MAAM,CAAC;AACnF,qBAAe;AACf,cAAQ,MAAM,KAAK,cAAc;AAAA,IACnC;AAAA,EACF;AACA,SAAO;AACT;AAaA,SAAS,gBAAgB,YAAsC;AAC7D,QAAM,SAAS,CAAC,GAAG,UAAU,EAAE;AAAA,IAC7B,CAAC,GAAG,MAAM,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,UAAU,EAAE,QAAQ,EAAE;AAAA,EAC/D;AACA,QAAM,WAAwB,CAAC;AAC/B,aAAW,aAAa,QAAQ;AAC9B,UAAM,SAAS,UAAU,MAAM,UAAU;AACzC,UAAM,YAAY,SAAS;AAAA,MACzB,CAAC,UACC,MAAM,SAAS,UAAU,SACzB,UAAU,OAAO,MAAM,OACvB,MAAM,MAAM,MAAM,QAAQ;AAAA,IAC9B;AACA,QAAI,CAAC,WAAW;AACd,eAAS,KAAK,SAAS;AAAA,IACzB;AAAA,EACF;AACA,SAAO;AACT;AAUO,SAAS,qBAAqB,SAA+B;AAClE,QAAM,YAAY,KAAK,IAAI;AAC3B,QAAM,iBAAiB,UAAU,OAAO;AACxC,QAAM,aAAa,oBAAI,IAAyB;AAChD,QAAM,eAAe,oBAAI,IAAqB;AAC9C,QAAM,aAAa,oBAAI,IAAqB;AAC5C,QAAM,aAAa,oBAAI,IAAqB;AAE5C,aAAW,aAAa,gBAAgB,kBAAkB,cAAc,CAAC,GAAG;AAC1E,UAAM,EAAE,KAAK,SAAS,OAAO,UAAU,IAAI,cAAc,gBAAgB,UAAU,KAAK;AAExF,QAAI,UAAU,aAAa,IAAI,SAAS;AACxC,QAAI,YAAY,QAAW;AACzB,gBAAU,kBAAkB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,mBAAa,IAAI,WAAW,OAAO;AAAA,IACrC;AACA,QAAI,SAAS;AACX;AAAA,IACF;AAEA,QAAI,WAAW,WAAW,IAAI,SAAS;AACvC,QAAI,aAAa,QAAW;AAC1B,iBAAW,iBAAiB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,iBAAW,IAAI,WAAW,QAAQ;AAAA,IACpC;AAEA,QAAI,QAAQ,WAAW,IAAI,SAAS;AACpC,QAAI,UAAU,QAAW;AACvB,cAAQ,gBAAgB,eAAe,MAAM,WAAW,OAAO,CAAC;AAChE,iBAAW,IAAI,WAAW,KAAK;AAAA,IACjC;AAEA,eAAW,SAAS,UAAU,SAAS;AACrC,UAAI,MAAM,aAAa,CAAC,UAAU;AAChC;AAAA,MACF;AAEA,UAAI,SAAS,oBAAoB,IAAI,MAAM,IAAI,GAAG;AAChD;AAAA,MACF;AAEA,UAAI,aAAa;AACjB,UAAI,MAAM,WAAW;AACnB,qBAAa;AAAA,MACf,WAAW,UAAU;AACnB,qBAAa;AAAA,MACf;AAEA,YAAM,WAAW,WAAW,IAAI,MAAM,IAAI;AAC1C,YAAM,SACJ,CAAC,YACD,aAAa,SAAS,cACrB,eAAe,SAAS,cAAc,UAAU,QAAQ,SAAS;AACpE,UAAI,QAAQ;AACV,mBAAW,IAAI,MAAM,MAAM;AAAA,UACzB,MAAM,MAAM;AAAA,UACZ;AAAA,UACA,OAAO,MAAM;AAAA,UACb,aAAa,MAAM;AAAA,UACnB,UAAU,UAAU;AAAA,QACtB,CAAC;AAAA,MACH;AAAA,IACF;AAAA,EACF;AAEA,QAAM,UAAU,MAAM,KAAK,WAAW,OAAO,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,WAAW,EAAE,QAAQ;AACtF,QAAM,aAAa,KAAK,IAAI,IAAI;AAEhC,SAAO;AAAA,IACL,mBAAmB,6BAA6B,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI,CAAC;AAAA,IAC1E;AAAA,IACA,OAAO;AAAA,MACL,cAAc,QAAQ;AAAA,MACtB;AAAA,MACA,eAAe,YAAY,EAAE;AAAA,IAC/B;AAAA,EACF;AACF;AAKO,SAAS,gBAAgB,QAAgC;AAC9D,SAAO,OAAO,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI;AACzC;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../src/anchor.ts"],"sourcesContent":["/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Treats a hyphen that joins words as a space\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n *\n * O hífen entre palavras vira espaço porque o catálogo e o laboratório\n * discordam sobre ele o tempo todo: o catálogo escreve \"Proteína C-Reativa\" e\n * \"High-Density Lipoprotein\", e os laudos imprimem \"Proteína C Reativa\" e\n * \"High Density Lipoprotein\". Sem essa equivalência, 82 dos 159 nomes com\n * hífen deixam de ancorar na grafia que o documento usa.\n *\n * Não era teórico: um GGT de verdade foi descartado como alucinação em 27\n * laudos porque o documento escrevia \"Gama glutamil transferase\" e o catálogo\n * \"Gama-Glutamil Transferase\". Um caractere derrubava o valor antes de\n * qualquer validação.\n *\n * A troca exige **letra antes** do hífen, e por isso não toca em número:\n * o `-2.5` de um T-score e o `0-5` de uma faixa de urina seguem intactos.\n * Trocar sem essa guarda apagaria o sinal de um valor negativo, que é bem\n * pior que o problema original.\n *\n * O pré-filtro de substring do `collectCandidates` compara a chave com o texto\n * já normalizado, então a equivalência precisa nascer aqui: aplicada só na\n * regex, o `includes` descartaria o nome antes de ela rodar.\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/(?<=\\p{L})-(?=[\\p{L}\\p{N}])/gu, ' ')\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"],"mappings":";AAaA;AAAA,EAEE;AAAA,EACA;AAAA,EACA;AAAA,OACK;AAwBA,IAAM,4BAA4B;AAMlC,IAAM,uBAAuB;AAM7B,IAAM,uBAAuB;AAGpC,IAAM,2BAA2B;AA8BjC,SAAS,UAAU,MAAsB;AACvC,SAAO,KACJ,UAAU,KAAK,EACf,QAAQ,oBAAoB,EAAE,EAC9B,YAAY,EACZ,QAAQ,iCAAiC,GAAG,EAC5C,QAAQ,aAAa,GAAG;AAC7B;AAEA,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAQD,IAAM,yBAAyB,oBAAI,IAAI;AAAA,EACrC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAOD,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AASD,IAAM,2BAAqC;AAAA,EACzC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AASA,IAAM,sBAAsB,oBAAI,IAAI;AAAA,EAClC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAGD,IAAM,yBAAmC;AAAA,EACvC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAOA,IAAM,4BAAsC,CAAC,eAAe,oBAAoB,aAAa;AAa7F,IAAM,mBAAmB;AAEzB,SAAS,gBAAgB,MAAuB;AAC9C,MAAI,0BAA0B,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC,GAAG;AACzD,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,KAAK,CAAC,OAAO,GAAG,KAAK,IAAI,CAAC,KAAK,iBAAiB,KAAK,IAAI;AACzF;AAEA,IAAM,gBAAgB;AAGtB,IAAI,mBAAuC;AAE3C,SAAS,gBAA6B;AACpC,MAAI,CAAC,kBAAkB;AACrB,uBAAmB,IAAI;AAAA,MACrB,OAAO,KAAK,YAAY,EACrB,IAAI,CAAC,SAAS,UAAU,IAAI,EAAE,KAAK,CAAC,EACpC,OAAO,OAAO;AAAA,IACnB;AAAA,EACF;AACA,SAAO;AACT;AAqBA,IAAI,iBAAkD;AACtD,IAAI,qBAAsD;AAE1D,SAAS,cAAwC;AAC/C,MAAI,CAAC,gBAAgB;AACnB,qBAAiB,qBAAqB;AAAA,EACxC;AACA,SAAO;AACT;AAEA,SAAS,aAAa,MAAsB;AAC1C,SAAO,KAAK,QAAQ,uBAAuB,MAAM;AACnD;AAkBA,SAAS,iBAAiB,gBAAgC;AACxD,QAAM,OAAO,eAAe,MAAM,GAAG,EAAE,IAAI,YAAY,EAAE,KAAK,YAAY;AAC1E,QAAM,SAAS,UAAU,KAAK,cAAc,IAAI,OAAO;AACvD,SAAO,IAAI,OAAO,sBAAsB,IAAI,GAAG,MAAM,sBAAsB,IAAI;AACjF;AAEA,SAAS,mBAAmB,SAA0C;AACpE,QAAM,aAAa,MAAM,QAAQ,QAAQ,QAAQ,IAAI,QAAQ,WAAW,CAAC,QAAQ,QAAQ;AACzF,SAAO,WAAW,SAAS,OAAO,KAAK,CAAC,QAAQ;AAClD;AAOA,SAAS,gBAAgB,gBAAwB,SAA0C;AACzF,MAAI,eAAe,SAAS,GAAG,GAAG;AAChC,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,IAAI,cAAc,KAAK,mBAAmB,OAAO;AACjF;AAEA,SAAS,kBAA4C;AACnD,MAAI,CAAC,oBAAoB;AACvB,UAAM,MAAM,oBAAI,IAAyB;AACzC,eAAW,WAAW,YAAY,GAAG;AACnC,iBAAW,QAAQ,QAAQ,OAAO;AAChC,cAAM,aAAa,UAAU,IAAI,EAAE,KAAK;AACxC,YAAI,CAAC,YAAY;AACf;AAAA,QACF;AACA,YAAI,WAAW,SAAS,KAAK,CAAC,wBAAwB,IAAI,UAAU,GAAG;AACrE;AAAA,QACF;AACA,YAAI,OAAO,IAAI,IAAI,UAAU;AAC7B,YAAI,CAAC,MAAM;AACT,iBAAO,EAAE,SAAS,CAAC,GAAG,OAAO,KAAK;AAClC,cAAI,IAAI,YAAY,IAAI;AAAA,QAC1B;AACA,aAAK,QAAQ,KAAK;AAAA,UAChB,WAAW,gBAAgB,YAAY,OAAO;AAAA,UAC9C,MAAM,QAAQ;AAAA,UACd,GAAI,QAAQ,SAAS,EAAE,OAAO,QAAQ,MAAM;AAAA,UAC5C,UAAU;AAAA,QACZ,CAAC;AAAA,MACH;AAAA,IACF;AACA,yBAAqB;AAAA,EACvB;AACA,SAAO;AACT;AAEA,SAAS,cAAc,MAAc,UAAkD;AACrF,QAAM,QAAQ,KAAK,YAAY,MAAM,QAAQ,IAAI;AACjD,QAAM,YAAY,KAAK,QAAQ,MAAM,QAAQ;AAC7C,SAAO,EAAE,KAAK,cAAc,KAAK,KAAK,SAAS,WAAW,MAAM;AAClE;AAEA,SAAS,kBAAkB,MAAuB;AAChD,SAAO,yBAAyB,KAAK,CAAC,YAAY,QAAQ,KAAK,IAAI,CAAC;AACtE;AAMA,SAAS,iBAAiB,MAAuB;AAC/C,MAAI,cAAc,KAAK,IAAI,GAAG;AAC5B,WAAO;AAAA,EACT;AACA,QAAM,aAAa,cAAc;AACjC,aAAW,SAAS,KAAK,MAAM,mBAAmB,GAAG;AACnD,QAAI,UAAU,WAAW,IAAI,KAAK,KAAK,wBAAwB,IAAI,KAAK,IAAI;AAC1E,aAAO;AAAA,IACT;AAAA,EACF;AACA,SAAO;AACT;AASA,SAAS,kBAAkB,gBAAqC;AAC9D,QAAM,aAA0B,CAAC;AACjC,aAAW,CAAC,MAAM,IAAI,KAAK,gBAAgB,GAAG;AAC5C,QAAI,CAAC,eAAe,SAAS,IAAI,GAAG;AAClC;AAAA,IACF;AACA,UAAM,EAAE,QAAQ,IAAI;AACpB,UAAM,QAAS,KAAK,UAAU,iBAAiB,IAAI;AACnD,UAAM,YAAY;AAClB,QAAI,cAAc;AAClB,QAAI,QAAQ,MAAM,KAAK,cAAc;AACrC,WAAO,UAAU,QAAQ,cAAc,0BAA0B;AAC/D,iBAAW,KAAK,EAAE,KAAK,MAAM,QAAQ,MAAM,CAAC,EAAE,QAAQ,SAAS,OAAO,MAAM,MAAM,CAAC;AACnF,qBAAe;AACf,cAAQ,MAAM,KAAK,cAAc;AAAA,IACnC;AAAA,EACF;AACA,SAAO;AACT;AAaA,SAAS,gBAAgB,YAAsC;AAC7D,QAAM,SAAS,CAAC,GAAG,UAAU,EAAE;AAAA,IAC7B,CAAC,GAAG,MAAM,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,UAAU,EAAE,QAAQ,EAAE;AAAA,EAC/D;AACA,QAAM,WAAwB,CAAC;AAC/B,aAAW,aAAa,QAAQ;AAC9B,UAAM,SAAS,UAAU,MAAM,UAAU;AACzC,UAAM,YAAY,SAAS;AAAA,MACzB,CAAC,UACC,MAAM,SAAS,UAAU,SACzB,UAAU,OAAO,MAAM,OACvB,MAAM,MAAM,MAAM,QAAQ;AAAA,IAC9B;AACA,QAAI,CAAC,WAAW;AACd,eAAS,KAAK,SAAS;AAAA,IACzB;AAAA,EACF;AACA,SAAO;AACT;AAUO,SAAS,qBAAqB,SAA+B;AAClE,QAAM,YAAY,KAAK,IAAI;AAC3B,QAAM,iBAAiB,UAAU,OAAO;AACxC,QAAM,aAAa,oBAAI,IAAyB;AAChD,QAAM,eAAe,oBAAI,IAAqB;AAC9C,QAAM,aAAa,oBAAI,IAAqB;AAC5C,QAAM,aAAa,oBAAI,IAAqB;AAE5C,aAAW,aAAa,gBAAgB,kBAAkB,cAAc,CAAC,GAAG;AAC1E,UAAM,EAAE,KAAK,SAAS,OAAO,UAAU,IAAI,cAAc,gBAAgB,UAAU,KAAK;AAExF,QAAI,UAAU,aAAa,IAAI,SAAS;AACxC,QAAI,YAAY,QAAW;AACzB,gBAAU,kBAAkB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,mBAAa,IAAI,WAAW,OAAO;AAAA,IACrC;AACA,QAAI,SAAS;AACX;AAAA,IACF;AAEA,QAAI,WAAW,WAAW,IAAI,SAAS;AACvC,QAAI,aAAa,QAAW;AAC1B,iBAAW,iBAAiB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,iBAAW,IAAI,WAAW,QAAQ;AAAA,IACpC;AAEA,QAAI,QAAQ,WAAW,IAAI,SAAS;AACpC,QAAI,UAAU,QAAW;AACvB,cAAQ,gBAAgB,eAAe,MAAM,WAAW,OAAO,CAAC;AAChE,iBAAW,IAAI,WAAW,KAAK;AAAA,IACjC;AAEA,eAAW,SAAS,UAAU,SAAS;AACrC,UAAI,MAAM,aAAa,CAAC,UAAU;AAChC;AAAA,MACF;AAEA,UAAI,SAAS,oBAAoB,IAAI,MAAM,IAAI,GAAG;AAChD;AAAA,MACF;AAEA,UAAI,aAAa;AACjB,UAAI,MAAM,WAAW;AACnB,qBAAa;AAAA,MACf,WAAW,UAAU;AACnB,qBAAa;AAAA,MACf;AAEA,YAAM,WAAW,WAAW,IAAI,MAAM,IAAI;AAC1C,YAAM,SACJ,CAAC,YACD,aAAa,SAAS,cACrB,eAAe,SAAS,cAAc,UAAU,QAAQ,SAAS;AACpE,UAAI,QAAQ;AACV,mBAAW,IAAI,MAAM,MAAM;AAAA,UACzB,MAAM,MAAM;AAAA,UACZ;AAAA,UACA,OAAO,MAAM;AAAA,UACb,aAAa,MAAM;AAAA,UACnB,UAAU,UAAU;AAAA,QACtB,CAAC;AAAA,MACH;AAAA,IACF;AAAA,EACF;AAEA,QAAM,UAAU,MAAM,KAAK,WAAW,OAAO,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,WAAW,EAAE,QAAQ;AACtF,QAAM,aAAa,KAAK,IAAI,IAAI;AAEhC,SAAO;AAAA,IACL,mBAAmB,6BAA6B,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI,CAAC;AAAA,IAC1E;AAAA,IACA,OAAO;AAAA,MACL,cAAc,QAAQ;AAAA,MACtB;AAAA,MACA,eAAe,YAAY,EAAE;AAAA,IAC/B;AAAA,EACF;AACF;AAKO,SAAS,gBAAgB,QAAgC;AAC9D,SAAO,OAAO,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI;AACzC;","names":[]}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@precisa-saude/fhir-ocr-utils",
|
|
3
|
-
"version": "0.21.
|
|
3
|
+
"version": "0.21.1",
|
|
4
4
|
"description": "Utilitários de ancoragem OCR para extração de biomarcadores de PDFs de resultados laboratoriais",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"fhir",
|
|
@@ -41,7 +41,7 @@
|
|
|
41
41
|
"dist"
|
|
42
42
|
],
|
|
43
43
|
"dependencies": {
|
|
44
|
-
"@precisa-saude/fhir": "^0.21.
|
|
44
|
+
"@precisa-saude/fhir": "^0.21.1"
|
|
45
45
|
},
|
|
46
46
|
"devDependencies": {
|
|
47
47
|
"tsup": "^8.3.5",
|