@precisa-saude/fhir-ocr-utils 0.16.3 → 0.16.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -47,7 +47,7 @@ console.log(result.filteredReference);
47
47
 
48
48
  Este pacote implementa um padrão de **ancoragem antes do LLM** para prevenir alucinações na extração de dados laboratoriais:
49
49
 
50
- 1. **Ancoragem (este pacote):** Escaneia o texto OCR bruto procurando nomes de biomarcadores conhecidos usando correspondência exata de strings contra as 180+ definições de `@precisa-saude/fhir`.
50
+ 1. **Ancoragem (este pacote):** Escaneia o texto OCR bruto procurando nomes de biomarcadores conhecidos contra as 180+ definições de `@precisa-saude/fhir`.
51
51
 
52
52
  2. **Filtragem:** Gera uma referência LLM filtrada (`filteredReference`) contendo apenas os biomarcadores que foram realmente encontrados no texto. O LLM só pode extrair valores para biomarcadores presentes nesta lista.
53
53
 
@@ -62,16 +62,58 @@ PDF → OCR → findBiomarkersInText() → filteredReference → LLM → valores
62
62
  o LLM pode extrair
63
63
  ```
64
64
 
65
+ ### Regras de correspondência
66
+
67
+ Uma âncora falsa é cara: ela envia ao LLM uma referência de biomarcadores que
68
+ não estão no documento, convidando o modelo a preencher valores inexistentes.
69
+ Por isso a correspondência é conservadora:
70
+
71
+ - **Fronteira de token** — o nome precisa aparecer inteiro. `Color` não casa
72
+ dentro de "Colorectal", nem `Cholesterol` dentro de "Hypercholesterolemia".
73
+ Plurais impressos pelos laboratórios ("Proteínas", "Cetonas") continuam
74
+ ancorando no nome singular do catálogo.
75
+ - **Nome mais longo vence** — em sobreposição, o casamento mais específico
76
+ prevalece: "HDL Cholesterol" ancora `HDL`, não `Cholesterol`; "Blood Glucose"
77
+ ancora `Glucose`, não `Blood_Urine`.
78
+ - **Nomes de uma linha só** — nos layouts em coluna dos laudos, linhas
79
+ consecutivas são biomarcadores distintos, então um nome composto não
80
+ atravessa quebra de linha ("Colesterol\nHDL" não vira "Colesterol HDL").
81
+ - **Co-ocorrência de valor para nomes genéricos** — nomes que também são
82
+ palavras comuns (`Color`, `Protein`, `Blood`, `Volume`, `Peso`) só ancoram se
83
+ a linha carregar um valor: um número, uma unidade conhecida, ou um termo
84
+ qualitativo esperado ("Negativo", "Ausente", "Amarelo Citrino"). "Specimen
85
+ type: Blood" não ancora; "Sangue Oculto: Negativo" ancora.
86
+ - **Contexto genético é descartado** — símbolos de gene colidem com nomes de
87
+ biomarcador (o gene `APOB` vs. a lipoproteína `ApoB`). Linhas com acesso
88
+ RefSeq (`NM_000384.2`), notação HGVS (`p.Trp448*`, `c.1234A>G`), `rs` do dbSNP
89
+ ou vocabulário de laudo genético não ancoram. É o contexto que decide, não uma
90
+ lista de símbolos proibidos — bani-los quebraria laudos lipídicos reais.
91
+
65
92
  ## API
66
93
 
67
94
  ### `findBiomarkersInText(ocrText: string): AnchorResult`
68
95
 
69
- Escaneia texto OCR e retorna todos os biomarcadores encontrados.
96
+ Escaneia texto OCR e retorna todos os biomarcadores encontrados — um casamento
97
+ por código, o de maior confiança.
70
98
 
71
- - `result.matches` — Lista de biomarcadores encontrados com código, LOINC, nome e posição
99
+ - `result.matches` — Lista de biomarcadores encontrados com código, LOINC, nome, confiança e posição
72
100
  - `result.filteredReference` — Referência formatada para enviar ao LLM
73
101
  - `result.stats` — Estatísticas de execução (total de padrões, encontrados, tempo)
74
102
 
103
+ `position` é o índice no texto normalizado (sem acentos, minúsculas, espaços
104
+ horizontais colapsados), não no texto OCR original.
105
+
106
+ #### Confiança
107
+
108
+ `match.confidence` reflete a qualidade do casamento — nome completo ao lado de
109
+ um valor não é a mesma coisa que uma menção solta em prosa:
110
+
111
+ | Valor | Constante | Significado |
112
+ | ----- | --------------------------- | ------------------------------------------------------------------------ |
113
+ | `1.0` | `CONFIDENCE_VALUE_ADJACENT` | Nome específico com valor na mesma linha (`Glicose: 95 mg/dL`) |
114
+ | `0.7` | `CONFIDENCE_NAME_ONLY` | Nome específico sem valor por perto (cabeçalho de seção, menção solta) |
115
+ | `0.4` | `CONFIDENCE_AMBIGUOUS` | Nome genérico que só ancorou por causa do valor ao lado (`Cor: Amarelo`) |
116
+
75
117
  ### `getMatchedCodes(result: AnchorResult): string[]`
76
118
 
77
119
  Extrai a lista de códigos de biomarcadores de um resultado de ancoragem.
package/dist/cli.js CHANGED
@@ -10,10 +10,15 @@ import { getInput, outputJson, outputText } from "@precisa-saude/fhir/cli-utils"
10
10
  // src/anchor.ts
11
11
  import {
12
12
  generateFilteredLLMReference,
13
- getAllSearchPatterns
13
+ getAllSearchPatterns,
14
+ UNIT_TO_UCUM
14
15
  } from "@precisa-saude/fhir";
16
+ var CONFIDENCE_VALUE_ADJACENT = 1;
17
+ var CONFIDENCE_NAME_ONLY = 0.7;
18
+ var CONFIDENCE_AMBIGUOUS = 0.4;
19
+ var MAX_OCCURRENCES_PER_NAME = 5;
15
20
  function normalize(text) {
16
- return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/\s+/g, " ");
21
+ return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/[^\S\n]+/g, " ");
17
22
  }
18
23
  var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
19
24
  "hdl",
@@ -50,73 +55,266 @@ var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
50
55
  "mlg",
51
56
  "tav"
52
57
  ]);
58
+ var CONTEXT_REQUIRED_NAMES = /* @__PURE__ */ new Set([
59
+ "bacteria",
60
+ // Bacteria_Urine — tem unidade, escapa da regra automática
61
+ "bacterias",
62
+ // Bacteria_Urine
63
+ "lead",
64
+ // Lead — verbo/substantivo comuníssimo em inglês
65
+ "peso",
66
+ // TotalMass
67
+ "saturation",
68
+ // TransferrinSaturation — "oxygen saturation", "saturation index"
69
+ "tap",
70
+ // ProthrombinTime — "tap" em inglês
71
+ "volume",
72
+ // VATVolume
73
+ "weight"
74
+ // TotalMass
75
+ ]);
76
+ var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
77
+ "absent",
78
+ "alguns",
79
+ "amarela",
80
+ "amarelo",
81
+ "anormal",
82
+ "ausencia",
83
+ "ausente",
84
+ "ausentes",
85
+ "citrino",
86
+ "claro",
87
+ "clear",
88
+ "cloudy",
89
+ "colorless",
90
+ "detectado",
91
+ "detected",
92
+ "escuro",
93
+ "incolor",
94
+ "indetectavel",
95
+ "limpido",
96
+ "moderada",
97
+ "moderado",
98
+ "negativa",
99
+ "negative",
100
+ "negativo",
101
+ "normais",
102
+ "normal",
103
+ "numerosos",
104
+ "ocasional",
105
+ "positiva",
106
+ "positive",
107
+ "positivo",
108
+ "present",
109
+ "presente",
110
+ "presentes",
111
+ "raras",
112
+ "raro",
113
+ "raros",
114
+ "reagente",
115
+ "trace",
116
+ "traces",
117
+ "tracos",
118
+ "turvo",
119
+ "undetectable",
120
+ "yellow"
121
+ ]);
122
+ var GENETIC_CONTEXT_PATTERNS = [
123
+ /\b[nx][mrpc]_\d{6,}/,
124
+ // RefSeq: NM_000384.2, NP_, NR_, XM_
125
+ /\bens[gtp]\d{6,}/,
126
+ // Ensembl: ENSG00000084674
127
+ /\bp\.[a-z]{3}\d/,
128
+ // HGVS proteína: p.Trp448*
129
+ /\bc\.\d+[acgt]?[>_+-]/,
130
+ // HGVS codificante: c.1234A>G, c.76_78del
131
+ /\brs\d{4,}\b/,
132
+ // dbSNP
133
+ /\bgenes?\b/,
134
+ /\bvariante?s?\b/,
135
+ /\bexons?\b/,
136
+ /\bzygosity\b/,
137
+ /\bzigosidade\b/,
138
+ /\balleles?\b/,
139
+ /\balelos?\b/,
140
+ /\bmutations?\b/,
141
+ /\bmutac(ao|oes)\b/,
142
+ /\bpathogenic/,
143
+ /\bpatogenic/,
144
+ /\bheterozyg/,
145
+ /\bhomozyg/,
146
+ /\bheterozigot/,
147
+ /\bhomozigot/,
148
+ /\bsequence change\b/
149
+ ];
150
+ var DIGIT_PATTERN = /\d/;
151
+ var cachedUnitTokens = null;
152
+ function getUnitTokens() {
153
+ if (!cachedUnitTokens) {
154
+ cachedUnitTokens = new Set(
155
+ Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter(Boolean)
156
+ );
157
+ }
158
+ return cachedUnitTokens;
159
+ }
53
160
  var cachedPatterns = null;
54
- var cachedNormalized = null;
161
+ var cachedNamePatterns = null;
55
162
  function getPatterns() {
56
163
  if (!cachedPatterns) {
57
164
  cachedPatterns = getAllSearchPatterns();
58
165
  }
59
166
  return cachedPatterns;
60
167
  }
61
- function getNormalizedPatterns() {
62
- if (!cachedNormalized) {
63
- const patterns = getPatterns();
168
+ function escapeRegExp(text) {
169
+ return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
170
+ }
171
+ function buildNamePattern(normalizedName) {
172
+ const body = normalizedName.split(" ").map(escapeRegExp).join("[^\\S\\n]+");
173
+ const plural = /\p{L}$/u.test(normalizedName) ? "s?" : "";
174
+ return new RegExp(`(?<![\\p{L}\\p{N}])${body}${plural}(?![\\p{L}\\p{N}])`, "gu");
175
+ }
176
+ function isQualitativeUrine(pattern) {
177
+ const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];
178
+ return categories.includes("urina") && !pattern.unit;
179
+ }
180
+ function isAmbiguousName(normalizedName, pattern) {
181
+ if (normalizedName.includes(" ")) {
182
+ return false;
183
+ }
184
+ return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);
185
+ }
186
+ function getNamePatterns() {
187
+ if (!cachedNamePatterns) {
64
188
  const map = /* @__PURE__ */ new Map();
65
- for (const pattern of patterns) {
189
+ for (const pattern of getPatterns()) {
66
190
  for (const name of pattern.names) {
67
- const normalized = normalize(name);
68
- const existing = map.get(normalized) || [];
69
- existing.push({
191
+ const normalized = normalize(name).trim();
192
+ if (!normalized) {
193
+ continue;
194
+ }
195
+ if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {
196
+ continue;
197
+ }
198
+ let slot = map.get(normalized);
199
+ if (!slot) {
200
+ slot = { entries: [], regex: null };
201
+ map.set(normalized, slot);
202
+ }
203
+ slot.entries.push({
204
+ ambiguous: isAmbiguousName(normalized, pattern),
70
205
  code: pattern.code,
71
206
  ...pattern.loinc && { loinc: pattern.loinc },
72
207
  original: name
73
208
  });
74
- map.set(normalized, existing);
75
209
  }
76
210
  }
77
- cachedNormalized = map;
211
+ cachedNamePatterns = map;
78
212
  }
79
- return cachedNormalized;
213
+ return cachedNamePatterns;
214
+ }
215
+ function getLineBounds(text, position) {
216
+ const start = text.lastIndexOf("\n", position) + 1;
217
+ const nextBreak = text.indexOf("\n", position);
218
+ return { end: nextBreak === -1 ? text.length : nextBreak, start };
219
+ }
220
+ function hasGeneticContext(line) {
221
+ return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));
222
+ }
223
+ function hasValueEvidence(line) {
224
+ if (DIGIT_PATTERN.test(line)) {
225
+ return true;
226
+ }
227
+ const unitTokens = getUnitTokens();
228
+ for (const token of line.split(/[^\p{L}\p{N}%/]+/u)) {
229
+ if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {
230
+ return true;
231
+ }
232
+ }
233
+ return false;
234
+ }
235
+ function collectCandidates(normalizedText) {
236
+ const candidates = [];
237
+ for (const [name, slot] of getNamePatterns()) {
238
+ if (!normalizedText.includes(name)) {
239
+ continue;
240
+ }
241
+ const { entries } = slot;
242
+ const regex = slot.regex ??= buildNamePattern(name);
243
+ regex.lastIndex = 0;
244
+ let occurrences = 0;
245
+ let match = regex.exec(normalizedText);
246
+ while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {
247
+ candidates.push({ end: match.index + match[0].length, entries, start: match.index });
248
+ occurrences += 1;
249
+ match = regex.exec(normalizedText);
250
+ }
251
+ }
252
+ return candidates;
253
+ }
254
+ function resolveOverlaps(candidates) {
255
+ const sorted = [...candidates].sort(
256
+ (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start
257
+ );
258
+ const accepted = [];
259
+ for (const candidate of sorted) {
260
+ const length = candidate.end - candidate.start;
261
+ const swallowed = accepted.some(
262
+ (other) => other.start <= candidate.start && candidate.end <= other.end && other.end - other.start > length
263
+ );
264
+ if (!swallowed) {
265
+ accepted.push(candidate);
266
+ }
267
+ }
268
+ return accepted;
80
269
  }
81
270
  function findBiomarkersInText(ocrText) {
82
271
  const startTime = Date.now();
83
272
  const normalizedText = normalize(ocrText);
84
- const matchedCodes = /* @__PURE__ */ new Set();
85
- const matches = [];
86
- const normalizedPatterns = getNormalizedPatterns();
87
- for (const [normalizedName, entries] of normalizedPatterns) {
88
- if (normalizedName.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalizedName)) {
273
+ const bestByCode = /* @__PURE__ */ new Map();
274
+ const geneticLines = /* @__PURE__ */ new Map();
275
+ const valueLines = /* @__PURE__ */ new Map();
276
+ for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {
277
+ const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);
278
+ let genetic = geneticLines.get(lineStart);
279
+ if (genetic === void 0) {
280
+ genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));
281
+ geneticLines.set(lineStart, genetic);
282
+ }
283
+ if (genetic) {
89
284
  continue;
90
285
  }
91
- let position = -1;
92
- if (normalizedName.length <= 4) {
93
- const regex = new RegExp(`\\b${normalizedName}\\b`);
94
- const match = regex.exec(normalizedText);
95
- if (match) {
96
- position = match.index;
97
- }
98
- } else {
99
- position = normalizedText.indexOf(normalizedName);
286
+ let hasValue = valueLines.get(lineStart);
287
+ if (hasValue === void 0) {
288
+ hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));
289
+ valueLines.set(lineStart, hasValue);
100
290
  }
101
- if (position !== -1) {
102
- for (const entry of entries) {
103
- if (!matchedCodes.has(entry.code)) {
104
- matchedCodes.add(entry.code);
105
- matches.push({
106
- code: entry.code,
107
- confidence: 1,
108
- loinc: entry.loinc,
109
- matchedName: entry.original,
110
- position
111
- });
112
- }
291
+ for (const entry of candidate.entries) {
292
+ if (entry.ambiguous && !hasValue) {
293
+ continue;
294
+ }
295
+ let confidence = CONFIDENCE_NAME_ONLY;
296
+ if (entry.ambiguous) {
297
+ confidence = CONFIDENCE_AMBIGUOUS;
298
+ } else if (hasValue) {
299
+ confidence = CONFIDENCE_VALUE_ADJACENT;
300
+ }
301
+ const existing = bestByCode.get(entry.code);
302
+ const better = !existing || confidence > existing.confidence || confidence === existing.confidence && candidate.start < existing.position;
303
+ if (better) {
304
+ bestByCode.set(entry.code, {
305
+ code: entry.code,
306
+ confidence,
307
+ loinc: entry.loinc,
308
+ matchedName: entry.original,
309
+ position: candidate.start
310
+ });
113
311
  }
114
312
  }
115
313
  }
314
+ const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);
116
315
  const scanTimeMs = Date.now() - startTime;
117
- const matchedCodesArray = Array.from(matchedCodes);
118
316
  return {
119
- filteredReference: generateFilteredLLMReference(matchedCodesArray),
317
+ filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),
120
318
  matches,
121
319
  stats: {
122
320
  matchedCount: matches.length,
@@ -205,7 +403,7 @@ async function main() {
205
403
  strict: false
206
404
  });
207
405
  if (values.version) {
208
- process.stdout.write(`${"0.16.3"}
406
+ process.stdout.write(`${"0.16.4"}
209
407
  `);
210
408
  return;
211
409
  }
package/dist/index.cjs CHANGED
@@ -2,9 +2,14 @@
2
2
 
3
3
 
4
4
 
5
+
5
6
  var _fhir = require('@precisa-saude/fhir');
7
+ var CONFIDENCE_VALUE_ADJACENT = 1;
8
+ var CONFIDENCE_NAME_ONLY = 0.7;
9
+ var CONFIDENCE_AMBIGUOUS = 0.4;
10
+ var MAX_OCCURRENCES_PER_NAME = 5;
6
11
  function normalize(text) {
7
- return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/\s+/g, " ");
12
+ return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/[^\S\n]+/g, " ");
8
13
  }
9
14
  var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
10
15
  "hdl",
@@ -41,73 +46,266 @@ var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
41
46
  "mlg",
42
47
  "tav"
43
48
  ]);
49
+ var CONTEXT_REQUIRED_NAMES = /* @__PURE__ */ new Set([
50
+ "bacteria",
51
+ // Bacteria_Urine — tem unidade, escapa da regra automática
52
+ "bacterias",
53
+ // Bacteria_Urine
54
+ "lead",
55
+ // Lead — verbo/substantivo comuníssimo em inglês
56
+ "peso",
57
+ // TotalMass
58
+ "saturation",
59
+ // TransferrinSaturation — "oxygen saturation", "saturation index"
60
+ "tap",
61
+ // ProthrombinTime — "tap" em inglês
62
+ "volume",
63
+ // VATVolume
64
+ "weight"
65
+ // TotalMass
66
+ ]);
67
+ var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
68
+ "absent",
69
+ "alguns",
70
+ "amarela",
71
+ "amarelo",
72
+ "anormal",
73
+ "ausencia",
74
+ "ausente",
75
+ "ausentes",
76
+ "citrino",
77
+ "claro",
78
+ "clear",
79
+ "cloudy",
80
+ "colorless",
81
+ "detectado",
82
+ "detected",
83
+ "escuro",
84
+ "incolor",
85
+ "indetectavel",
86
+ "limpido",
87
+ "moderada",
88
+ "moderado",
89
+ "negativa",
90
+ "negative",
91
+ "negativo",
92
+ "normais",
93
+ "normal",
94
+ "numerosos",
95
+ "ocasional",
96
+ "positiva",
97
+ "positive",
98
+ "positivo",
99
+ "present",
100
+ "presente",
101
+ "presentes",
102
+ "raras",
103
+ "raro",
104
+ "raros",
105
+ "reagente",
106
+ "trace",
107
+ "traces",
108
+ "tracos",
109
+ "turvo",
110
+ "undetectable",
111
+ "yellow"
112
+ ]);
113
+ var GENETIC_CONTEXT_PATTERNS = [
114
+ /\b[nx][mrpc]_\d{6,}/,
115
+ // RefSeq: NM_000384.2, NP_, NR_, XM_
116
+ /\bens[gtp]\d{6,}/,
117
+ // Ensembl: ENSG00000084674
118
+ /\bp\.[a-z]{3}\d/,
119
+ // HGVS proteína: p.Trp448*
120
+ /\bc\.\d+[acgt]?[>_+-]/,
121
+ // HGVS codificante: c.1234A>G, c.76_78del
122
+ /\brs\d{4,}\b/,
123
+ // dbSNP
124
+ /\bgenes?\b/,
125
+ /\bvariante?s?\b/,
126
+ /\bexons?\b/,
127
+ /\bzygosity\b/,
128
+ /\bzigosidade\b/,
129
+ /\balleles?\b/,
130
+ /\balelos?\b/,
131
+ /\bmutations?\b/,
132
+ /\bmutac(ao|oes)\b/,
133
+ /\bpathogenic/,
134
+ /\bpatogenic/,
135
+ /\bheterozyg/,
136
+ /\bhomozyg/,
137
+ /\bheterozigot/,
138
+ /\bhomozigot/,
139
+ /\bsequence change\b/
140
+ ];
141
+ var DIGIT_PATTERN = /\d/;
142
+ var cachedUnitTokens = null;
143
+ function getUnitTokens() {
144
+ if (!cachedUnitTokens) {
145
+ cachedUnitTokens = new Set(
146
+ Object.keys(_fhir.UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter(Boolean)
147
+ );
148
+ }
149
+ return cachedUnitTokens;
150
+ }
44
151
  var cachedPatterns = null;
45
- var cachedNormalized = null;
152
+ var cachedNamePatterns = null;
46
153
  function getPatterns() {
47
154
  if (!cachedPatterns) {
48
155
  cachedPatterns = _fhir.getAllSearchPatterns.call(void 0, );
49
156
  }
50
157
  return cachedPatterns;
51
158
  }
52
- function getNormalizedPatterns() {
53
- if (!cachedNormalized) {
54
- const patterns = getPatterns();
159
+ function escapeRegExp(text) {
160
+ return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
161
+ }
162
+ function buildNamePattern(normalizedName) {
163
+ const body = normalizedName.split(" ").map(escapeRegExp).join("[^\\S\\n]+");
164
+ const plural = /\p{L}$/u.test(normalizedName) ? "s?" : "";
165
+ return new RegExp(`(?<![\\p{L}\\p{N}])${body}${plural}(?![\\p{L}\\p{N}])`, "gu");
166
+ }
167
+ function isQualitativeUrine(pattern) {
168
+ const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];
169
+ return categories.includes("urina") && !pattern.unit;
170
+ }
171
+ function isAmbiguousName(normalizedName, pattern) {
172
+ if (normalizedName.includes(" ")) {
173
+ return false;
174
+ }
175
+ return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);
176
+ }
177
+ function getNamePatterns() {
178
+ if (!cachedNamePatterns) {
55
179
  const map = /* @__PURE__ */ new Map();
56
- for (const pattern of patterns) {
180
+ for (const pattern of getPatterns()) {
57
181
  for (const name of pattern.names) {
58
- const normalized = normalize(name);
59
- const existing = map.get(normalized) || [];
60
- existing.push({
182
+ const normalized = normalize(name).trim();
183
+ if (!normalized) {
184
+ continue;
185
+ }
186
+ if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {
187
+ continue;
188
+ }
189
+ let slot = map.get(normalized);
190
+ if (!slot) {
191
+ slot = { entries: [], regex: null };
192
+ map.set(normalized, slot);
193
+ }
194
+ slot.entries.push({
195
+ ambiguous: isAmbiguousName(normalized, pattern),
61
196
  code: pattern.code,
62
197
  ...pattern.loinc && { loinc: pattern.loinc },
63
198
  original: name
64
199
  });
65
- map.set(normalized, existing);
66
200
  }
67
201
  }
68
- cachedNormalized = map;
202
+ cachedNamePatterns = map;
203
+ }
204
+ return cachedNamePatterns;
205
+ }
206
+ function getLineBounds(text, position) {
207
+ const start = text.lastIndexOf("\n", position) + 1;
208
+ const nextBreak = text.indexOf("\n", position);
209
+ return { end: nextBreak === -1 ? text.length : nextBreak, start };
210
+ }
211
+ function hasGeneticContext(line) {
212
+ return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));
213
+ }
214
+ function hasValueEvidence(line) {
215
+ if (DIGIT_PATTERN.test(line)) {
216
+ return true;
217
+ }
218
+ const unitTokens = getUnitTokens();
219
+ for (const token of line.split(/[^\p{L}\p{N}%/]+/u)) {
220
+ if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {
221
+ return true;
222
+ }
223
+ }
224
+ return false;
225
+ }
226
+ function collectCandidates(normalizedText) {
227
+ const candidates = [];
228
+ for (const [name, slot] of getNamePatterns()) {
229
+ if (!normalizedText.includes(name)) {
230
+ continue;
231
+ }
232
+ const { entries } = slot;
233
+ const regex = slot.regex ??= buildNamePattern(name);
234
+ regex.lastIndex = 0;
235
+ let occurrences = 0;
236
+ let match = regex.exec(normalizedText);
237
+ while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {
238
+ candidates.push({ end: match.index + match[0].length, entries, start: match.index });
239
+ occurrences += 1;
240
+ match = regex.exec(normalizedText);
241
+ }
242
+ }
243
+ return candidates;
244
+ }
245
+ function resolveOverlaps(candidates) {
246
+ const sorted = [...candidates].sort(
247
+ (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start
248
+ );
249
+ const accepted = [];
250
+ for (const candidate of sorted) {
251
+ const length = candidate.end - candidate.start;
252
+ const swallowed = accepted.some(
253
+ (other) => other.start <= candidate.start && candidate.end <= other.end && other.end - other.start > length
254
+ );
255
+ if (!swallowed) {
256
+ accepted.push(candidate);
257
+ }
69
258
  }
70
- return cachedNormalized;
259
+ return accepted;
71
260
  }
72
261
  function findBiomarkersInText(ocrText) {
73
262
  const startTime = Date.now();
74
263
  const normalizedText = normalize(ocrText);
75
- const matchedCodes = /* @__PURE__ */ new Set();
76
- const matches = [];
77
- const normalizedPatterns = getNormalizedPatterns();
78
- for (const [normalizedName, entries] of normalizedPatterns) {
79
- if (normalizedName.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalizedName)) {
264
+ const bestByCode = /* @__PURE__ */ new Map();
265
+ const geneticLines = /* @__PURE__ */ new Map();
266
+ const valueLines = /* @__PURE__ */ new Map();
267
+ for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {
268
+ const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);
269
+ let genetic = geneticLines.get(lineStart);
270
+ if (genetic === void 0) {
271
+ genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));
272
+ geneticLines.set(lineStart, genetic);
273
+ }
274
+ if (genetic) {
80
275
  continue;
81
276
  }
82
- let position = -1;
83
- if (normalizedName.length <= 4) {
84
- const regex = new RegExp(`\\b${normalizedName}\\b`);
85
- const match = regex.exec(normalizedText);
86
- if (match) {
87
- position = match.index;
88
- }
89
- } else {
90
- position = normalizedText.indexOf(normalizedName);
277
+ let hasValue = valueLines.get(lineStart);
278
+ if (hasValue === void 0) {
279
+ hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));
280
+ valueLines.set(lineStart, hasValue);
91
281
  }
92
- if (position !== -1) {
93
- for (const entry of entries) {
94
- if (!matchedCodes.has(entry.code)) {
95
- matchedCodes.add(entry.code);
96
- matches.push({
97
- code: entry.code,
98
- confidence: 1,
99
- loinc: entry.loinc,
100
- matchedName: entry.original,
101
- position
102
- });
103
- }
282
+ for (const entry of candidate.entries) {
283
+ if (entry.ambiguous && !hasValue) {
284
+ continue;
285
+ }
286
+ let confidence = CONFIDENCE_NAME_ONLY;
287
+ if (entry.ambiguous) {
288
+ confidence = CONFIDENCE_AMBIGUOUS;
289
+ } else if (hasValue) {
290
+ confidence = CONFIDENCE_VALUE_ADJACENT;
291
+ }
292
+ const existing = bestByCode.get(entry.code);
293
+ const better = !existing || confidence > existing.confidence || confidence === existing.confidence && candidate.start < existing.position;
294
+ if (better) {
295
+ bestByCode.set(entry.code, {
296
+ code: entry.code,
297
+ confidence,
298
+ loinc: entry.loinc,
299
+ matchedName: entry.original,
300
+ position: candidate.start
301
+ });
104
302
  }
105
303
  }
106
304
  }
305
+ const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);
107
306
  const scanTimeMs = Date.now() - startTime;
108
- const matchedCodesArray = Array.from(matchedCodes);
109
307
  return {
110
- filteredReference: _fhir.generateFilteredLLMReference.call(void 0, matchedCodesArray),
308
+ filteredReference: _fhir.generateFilteredLLMReference.call(void 0, matches.map((m) => m.code)),
111
309
  matches,
112
310
  stats: {
113
311
  matchedCount: matches.length,
@@ -122,5 +320,8 @@ function getMatchedCodes(result) {
122
320
 
123
321
 
124
322
 
125
- exports.findBiomarkersInText = findBiomarkersInText; exports.getMatchedCodes = getMatchedCodes;
323
+
324
+
325
+
326
+ exports.CONFIDENCE_AMBIGUOUS = CONFIDENCE_AMBIGUOUS; exports.CONFIDENCE_NAME_ONLY = CONFIDENCE_NAME_ONLY; exports.CONFIDENCE_VALUE_ADJACENT = CONFIDENCE_VALUE_ADJACENT; exports.findBiomarkersInText = findBiomarkersInText; exports.getMatchedCodes = getMatchedCodes;
126
327
  //# sourceMappingURL=index.cjs.map
@@ -1 +1 @@
1
- {"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts"],"names":[],"mappings":"AAAA;ACQA;AAEE;AACA;AAAA,2CACK;AA0BP,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,MAAA,EAAQ,GAAG,CAAA;AACxB;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAID,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,iBAAA,EAAuD,IAAA;AAE3D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,qBAAA,CAAA,EAAqD;AAC5D,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,MAAM,SAAA,EAAW,WAAA,CAAY,CAAA;AAC7B,IAAA,MAAM,IAAA,kBAAM,IAAI,GAAA,CAA4B,CAAA;AAC5C,IAAA,IAAA,CAAA,MAAW,QAAA,GAAW,QAAA,EAAU;AAC9B,MAAA,IAAA,CAAA,MAAW,KAAA,GAAQ,OAAA,CAAQ,KAAA,EAAO;AAChC,QAAA,MAAM,WAAA,EAAa,SAAA,CAAU,IAAI,CAAA;AACjC,QAAA,MAAM,SAAA,EAAW,GAAA,CAAI,GAAA,CAAI,UAAU,EAAA,GAAK,CAAC,CAAA;AACzC,QAAA,QAAA,CAAS,IAAA,CAAK;AAAA,UACZ,IAAA,EAAM,OAAA,CAAQ,IAAA;AAAA,UACd,GAAI,OAAA,CAAQ,MAAA,GAAS,EAAE,KAAA,EAAO,OAAA,CAAQ,MAAM,CAAA;AAAA,UAC5C,QAAA,EAAU;AAAA,QACZ,CAAC,CAAA;AACD,QAAA,GAAA,CAAI,GAAA,CAAI,UAAA,EAAY,QAAQ,CAAA;AAAA,MAC9B;AAAA,IACF;AACA,IAAA,iBAAA,EAAmB,GAAA;AAAA,EACrB;AACA,EAAA,OAAO,gBAAA;AACT;AAOO,SAAS,oBAAA,CAAqB,OAAA,EAA+B;AAClE,EAAA,MAAM,UAAA,EAAY,IAAA,CAAK,GAAA,CAAI,CAAA;AAC3B,EAAA,MAAM,eAAA,EAAiB,SAAA,CAAU,OAAO,CAAA;AACxC,EAAA,MAAM,aAAA,kBAAe,IAAI,GAAA,CAAY,CAAA;AACrC,EAAA,MAAM,QAAA,EAAyB,CAAC,CAAA;AAChC,EAAA,MAAM,mBAAA,EAAqB,qBAAA,CAAsB,CAAA;AAEjD,EAAA,IAAA,CAAA,MAAW,CAAC,cAAA,EAAgB,OAAO,EAAA,GAAK,kBAAA,EAAoB;AAC1D,IAAA,GAAA,CAAI,cAAA,CAAe,OAAA,EAAS,EAAA,GAAK,CAAC,uBAAA,CAAwB,GAAA,CAAI,cAAc,CAAA,EAAG;AAC7E,MAAA,QAAA;AAAA,IACF;AAEA,IAAA,IAAI,SAAA,EAAW,CAAA,CAAA;AACf,IAAA,GAAA,CAAI,cAAA,CAAe,OAAA,GAAU,CAAA,EAAG;AAC9B,MAAA,MAAM,MAAA,EAAQ,IAAI,MAAA,CAAO,CAAA,GAAA,EAAM,cAAc,CAAA,GAAA,CAAK,CAAA;AAClD,MAAA,MAAM,MAAA,EAAQ,KAAA,CAAM,IAAA,CAAK,cAAc,CAAA;AACvC,MAAA,GAAA,CAAI,KAAA,EAAO;AACT,QAAA,SAAA,EAAW,KAAA,CAAM,KAAA;AAAA,MACnB;AAAA,IACF,EAAA,KAAO;AACL,MAAA,SAAA,EAAW,cAAA,CAAe,OAAA,CAAQ,cAAc,CAAA;AAAA,IAClD;AAEA,IAAA,GAAA,CAAI,SAAA,IAAa,CAAA,CAAA,EAAI;AACnB,MAAA,IAAA,CAAA,MAAW,MAAA,GAAS,OAAA,EAAS;AAC3B,QAAA,GAAA,CAAI,CAAC,YAAA,CAAa,GAAA,CAAI,KAAA,CAAM,IAAI,CAAA,EAAG;AACjC,UAAA,YAAA,CAAa,GAAA,CAAI,KAAA,CAAM,IAAI,CAAA;AAC3B,UAAA,OAAA,CAAQ,IAAA,CAAK;AAAA,YACX,IAAA,EAAM,KAAA,CAAM,IAAA;AAAA,YACZ,UAAA,EAAY,CAAA;AAAA,YACZ,KAAA,EAAO,KAAA,CAAM,KAAA;AAAA,YACb,WAAA,EAAa,KAAA,CAAM,QAAA;AAAA,YACnB;AAAA,UACF,CAAC,CAAA;AAAA,QACH;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAEA,EAAA,MAAM,WAAA,EAAa,IAAA,CAAK,GAAA,CAAI,EAAA,EAAI,SAAA;AAChC,EAAA,MAAM,kBAAA,EAAoB,KAAA,CAAM,IAAA,CAAK,YAAY,CAAA;AAEjD,EAAA,OAAO;AAAA,IACL,iBAAA,EAAmB,gDAAA,iBAA8C,CAAA;AAAA,IACjE,OAAA;AAAA,IACA,KAAA,EAAO;AAAA,MACL,YAAA,EAAc,OAAA,CAAQ,MAAA;AAAA,MACtB,UAAA;AAAA,MACA,aAAA,EAAe,WAAA,CAAY,CAAA,CAAE;AAAA,IAC/B;AAAA,EACF,CAAA;AACF;AAKO,SAAS,eAAA,CAAgB,MAAA,EAAgC;AAC9D,EAAA,OAAO,MAAA,CAAO,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAA,EAAA,GAAM,CAAA,CAAE,IAAI,CAAA;AACzC;ADzDA;AACE;AACA;AACF,+FAAC","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Normalizes whitespace\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/\\s+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\ntype PatternEntry = { original: string; code: string; loinc?: string };\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNormalized: Map<string, PatternEntry[]> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction getNormalizedPatterns(): Map<string, PatternEntry[]> {\n if (!cachedNormalized) {\n const patterns = getPatterns();\n const map = new Map<string, PatternEntry[]>();\n for (const pattern of patterns) {\n for (const name of pattern.names) {\n const normalized = normalize(name);\n const existing = map.get(normalized) || [];\n existing.push({\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n map.set(normalized, existing);\n }\n }\n cachedNormalized = map;\n }\n return cachedNormalized;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n * Uses exact string matching on normalized text.\n * Returns unique matches (same biomarker won't be matched twice).\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const matchedCodes = new Set<string>();\n const matches: AnchorMatch[] = [];\n const normalizedPatterns = getNormalizedPatterns();\n\n for (const [normalizedName, entries] of normalizedPatterns) {\n if (normalizedName.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalizedName)) {\n continue;\n }\n\n let position = -1;\n if (normalizedName.length <= 4) {\n const regex = new RegExp(`\\\\b${normalizedName}\\\\b`);\n const match = regex.exec(normalizedText);\n if (match) {\n position = match.index;\n }\n } else {\n position = normalizedText.indexOf(normalizedName);\n }\n\n if (position !== -1) {\n for (const entry of entries) {\n if (!matchedCodes.has(entry.code)) {\n matchedCodes.add(entry.code);\n matches.push({\n code: entry.code,\n confidence: 1.0,\n loinc: entry.loinc,\n matchedName: entry.original,\n position,\n });\n }\n }\n }\n }\n\n const scanTimeMs = Date.now() - startTime;\n const matchedCodesArray = Array.from(matchedCodes);\n\n return {\n filteredReference: generateFilteredLLMReference(matchedCodesArray),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"]}
1
+ {"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts"],"names":[],"mappings":"AAAA;ACaA;AAEE;AACA;AACA;AAAA,2CACK;AAwBA,IAAM,0BAAA,EAA4B,CAAA;AAMlC,IAAM,qBAAA,EAAuB,GAAA;AAM7B,IAAM,qBAAA,EAAuB,GAAA;AAGpC,IAAM,yBAAA,EAA2B,CAAA;AASjC,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,WAAA,EAAa,GAAG,CAAA;AAC7B;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAQD,IAAM,uBAAA,kBAAyB,IAAI,GAAA,CAAI;AAAA,EACrC,UAAA;AAAA;AAAA,EACA,WAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,YAAA;AAAA;AAAA,EACA,KAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA,EACA;AAAA;AACF,CAAC,CAAA;AAOD,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,QAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,OAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,cAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,OAAA;AAAA,EACA,UAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,QAAA;AAAA,EACA,OAAA;AAAA,EACA,cAAA;AAAA,EACA;AACF,CAAC,CAAA;AASD,IAAM,yBAAA,EAAqC;AAAA,EACzC,qBAAA;AAAA;AAAA,EACA,kBAAA;AAAA;AAAA,EACA,iBAAA;AAAA;AAAA,EACA,uBAAA;AAAA;AAAA,EACA,cAAA;AAAA;AAAA,EACA,YAAA;AAAA,EACA,iBAAA;AAAA,EACA,YAAA;AAAA,EACA,cAAA;AAAA,EACA,gBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,gBAAA;AAAA,EACA,mBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,aAAA;AAAA,EACA,WAAA;AAAA,EACA,eAAA;AAAA,EACA,aAAA;AAAA,EACA;AACF,CAAA;AAEA,IAAM,cAAA,EAAgB,IAAA;AAGtB,IAAI,iBAAA,EAAuC,IAAA;AAE3C,SAAS,aAAA,CAAA,EAA6B;AACpC,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,iBAAA,EAAmB,IAAI,GAAA;AAAA,MACrB,MAAA,CAAO,IAAA,CAAK,kBAAY,CAAA,CACrB,GAAA,CAAI,CAAC,IAAA,EAAA,GAAS,SAAA,CAAU,IAAI,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,CACpC,MAAA,CAAO,OAAO;AAAA,IACnB,CAAA;AAAA,EACF;AACA,EAAA,OAAO,gBAAA;AACT;AAqBA,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,mBAAA,EAAsD,IAAA;AAE1D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,YAAA,CAAa,IAAA,EAAsB;AAC1C,EAAA,OAAO,IAAA,CAAK,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AACnD;AAkBA,SAAS,gBAAA,CAAiB,cAAA,EAAgC;AACxD,EAAA,MAAM,KAAA,EAAO,cAAA,CAAe,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,YAAY,CAAA,CAAE,IAAA,CAAK,YAAY,CAAA;AAC1E,EAAA,MAAM,OAAA,EAAS,SAAA,CAAU,IAAA,CAAK,cAAc,EAAA,EAAI,KAAA,EAAO,EAAA;AACvD,EAAA,OAAO,IAAI,MAAA,CAAO,CAAA,mBAAA,EAAsB,IAAI,CAAA,EAAA;AAC9C;AAEsE;AAC3B,EAAA;AACD,EAAA;AAC1C;AAOiD;AACb,EAAA;AACzB,IAAA;AACT,EAAA;AACkC,EAAA;AACpC;AAEqD;AAC1B,EAAA;AACkB,IAAA;AACJ,IAAA;AACD,MAAA;AACG,QAAA;AAClB,QAAA;AACf,UAAA;AACF,QAAA;AAC8B,QAAA;AAC5B,UAAA;AACF,QAAA;AAC6B,QAAA;AAClB,QAAA;AACyB,UAAA;AACV,UAAA;AAC1B,QAAA;AACkB,QAAA;AACW,UAAA;AACb,UAAA;AACwB,UAAA;AAC5B,UAAA;AACX,QAAA;AACH,MAAA;AACF,IAAA;AACqB,IAAA;AACvB,EAAA;AACO,EAAA;AACT;AAEuF;AACxC,EAAA;AACA,EAAA;AACP,EAAA;AACxC;AAEkD;AACV,EAAA;AACxC;AAMiD;AACjB,EAAA;AACrB,IAAA;AACT,EAAA;AACiC,EAAA;AACF,EAAA;AACU,IAAA;AAC9B,MAAA;AACT,IAAA;AACF,EAAA;AACO,EAAA;AACT;AASgE;AAC7B,EAAA;AACU,EAAA;AACL,IAAA;AAClC,MAAA;AACF,IAAA;AACoB,IAAA;AACU,IAAA;AACZ,IAAA;AACA,IAAA;AACmB,IAAA;AACE,IAAA;AACA,MAAA;AACtB,MAAA;AACkB,MAAA;AACnC,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAa+D;AAC9B,EAAA;AACU,IAAA;AACzC,EAAA;AAC+B,EAAA;AACC,EAAA;AACW,IAAA;AACd,IAAA;AAEE,MAAA;AAG7B,IAAA;AACgB,IAAA;AACS,MAAA;AACzB,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAUoE;AACvC,EAAA;AACa,EAAA;AACQ,EAAA;AACF,EAAA;AACF,EAAA;AAEJ,EAAA;AACK,IAAA;AAEH,IAAA;AACb,IAAA;AACG,MAAA;AACO,MAAA;AACrC,IAAA;AACa,IAAA;AACX,MAAA;AACF,IAAA;AAEuC,IAAA;AACX,IAAA;AACE,MAAA;AACM,MAAA;AACpC,IAAA;AAEuC,IAAA;AACH,MAAA;AAChC,QAAA;AACF,MAAA;AAEiB,MAAA;AACI,MAAA;AACN,QAAA;AACM,MAAA;AACN,QAAA;AACf,MAAA;AAEsC,MAAA;AAGvB,MAAA;AAEH,MAAA;AACiB,QAAA;AACb,UAAA;AACZ,UAAA;AACa,UAAA;AACM,UAAA;AACC,UAAA;AACrB,QAAA;AACH,MAAA;AACF,IAAA;AACF,EAAA;AAE6C,EAAA;AACb,EAAA;AAEzB,EAAA;AACc,IAAA;AACnB,IAAA;AACO,IAAA;AACiB,MAAA;AACtB,MAAA;AAC6B,MAAA;AAC/B,IAAA;AACF,EAAA;AACF;AAKgE;AACvB,EAAA;AACzC;AD/K+C;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"]}
package/dist/index.d.cts CHANGED
@@ -4,6 +4,11 @@
4
4
  * Scans OCR text for biomarker names BEFORE sending to LLM.
5
5
  * This prevents hallucination by constraining what biomarkers
6
6
  * the LLM is allowed to extract.
7
+ *
8
+ * Matching is deliberately conservative: a name only anchors when it appears
9
+ * as a whole token, is not swallowed by a longer biomarker name, is not inside
10
+ * a genetic report line, and — for generic single-word names — sits on a line
11
+ * that actually carries a value.
7
12
  */
8
13
  interface AnchorMatch {
9
14
  code: string;
@@ -21,10 +26,28 @@ interface AnchorResult {
21
26
  scanTimeMs: number;
22
27
  };
23
28
  }
29
+ /**
30
+ * Confidence assigned to a specific biomarker name found on a line that also
31
+ * carries a value (a number, a unit, or an expected qualitative term).
32
+ */
33
+ declare const CONFIDENCE_VALUE_ADJACENT = 1;
34
+ /**
35
+ * Confidence assigned to a specific biomarker name with no value evidence
36
+ * nearby — a section heading, or a mention in prose.
37
+ */
38
+ declare const CONFIDENCE_NAME_ONLY = 0.7;
39
+ /**
40
+ * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,
41
+ * `Blood`, …) that only anchored because a value was found next to it.
42
+ */
43
+ declare const CONFIDENCE_AMBIGUOUS = 0.4;
24
44
  /**
25
45
  * Find all biomarker names present in OCR text.
26
- * Uses exact string matching on normalized text.
27
- * Returns unique matches (same biomarker won't be matched twice).
46
+ *
47
+ * Matching is whole-token, longest-match-wins, and context-aware: matches
48
+ * inside genetic report lines are discarded, and generic names only anchor
49
+ * when a value sits on the same line. Returns one match per biomarker code —
50
+ * the highest-confidence occurrence.
28
51
  */
29
52
  declare function findBiomarkersInText(ocrText: string): AnchorResult;
30
53
  /**
@@ -32,4 +55,4 @@ declare function findBiomarkersInText(ocrText: string): AnchorResult;
32
55
  */
33
56
  declare function getMatchedCodes(result: AnchorResult): string[];
34
57
 
35
- export { type AnchorMatch, type AnchorResult, findBiomarkersInText, getMatchedCodes };
58
+ export { type AnchorMatch, type AnchorResult, CONFIDENCE_AMBIGUOUS, CONFIDENCE_NAME_ONLY, CONFIDENCE_VALUE_ADJACENT, findBiomarkersInText, getMatchedCodes };
package/dist/index.d.ts CHANGED
@@ -4,6 +4,11 @@
4
4
  * Scans OCR text for biomarker names BEFORE sending to LLM.
5
5
  * This prevents hallucination by constraining what biomarkers
6
6
  * the LLM is allowed to extract.
7
+ *
8
+ * Matching is deliberately conservative: a name only anchors when it appears
9
+ * as a whole token, is not swallowed by a longer biomarker name, is not inside
10
+ * a genetic report line, and — for generic single-word names — sits on a line
11
+ * that actually carries a value.
7
12
  */
8
13
  interface AnchorMatch {
9
14
  code: string;
@@ -21,10 +26,28 @@ interface AnchorResult {
21
26
  scanTimeMs: number;
22
27
  };
23
28
  }
29
+ /**
30
+ * Confidence assigned to a specific biomarker name found on a line that also
31
+ * carries a value (a number, a unit, or an expected qualitative term).
32
+ */
33
+ declare const CONFIDENCE_VALUE_ADJACENT = 1;
34
+ /**
35
+ * Confidence assigned to a specific biomarker name with no value evidence
36
+ * nearby — a section heading, or a mention in prose.
37
+ */
38
+ declare const CONFIDENCE_NAME_ONLY = 0.7;
39
+ /**
40
+ * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,
41
+ * `Blood`, …) that only anchored because a value was found next to it.
42
+ */
43
+ declare const CONFIDENCE_AMBIGUOUS = 0.4;
24
44
  /**
25
45
  * Find all biomarker names present in OCR text.
26
- * Uses exact string matching on normalized text.
27
- * Returns unique matches (same biomarker won't be matched twice).
46
+ *
47
+ * Matching is whole-token, longest-match-wins, and context-aware: matches
48
+ * inside genetic report lines are discarded, and generic names only anchor
49
+ * when a value sits on the same line. Returns one match per biomarker code —
50
+ * the highest-confidence occurrence.
28
51
  */
29
52
  declare function findBiomarkersInText(ocrText: string): AnchorResult;
30
53
  /**
@@ -32,4 +55,4 @@ declare function findBiomarkersInText(ocrText: string): AnchorResult;
32
55
  */
33
56
  declare function getMatchedCodes(result: AnchorResult): string[];
34
57
 
35
- export { type AnchorMatch, type AnchorResult, findBiomarkersInText, getMatchedCodes };
58
+ export { type AnchorMatch, type AnchorResult, CONFIDENCE_AMBIGUOUS, CONFIDENCE_NAME_ONLY, CONFIDENCE_VALUE_ADJACENT, findBiomarkersInText, getMatchedCodes };
package/dist/index.js CHANGED
@@ -1,10 +1,15 @@
1
1
  // src/anchor.ts
2
2
  import {
3
3
  generateFilteredLLMReference,
4
- getAllSearchPatterns
4
+ getAllSearchPatterns,
5
+ UNIT_TO_UCUM
5
6
  } from "@precisa-saude/fhir";
7
+ var CONFIDENCE_VALUE_ADJACENT = 1;
8
+ var CONFIDENCE_NAME_ONLY = 0.7;
9
+ var CONFIDENCE_AMBIGUOUS = 0.4;
10
+ var MAX_OCCURRENCES_PER_NAME = 5;
6
11
  function normalize(text) {
7
- return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/\s+/g, " ");
12
+ return text.normalize("NFD").replace(/[\u0300-\u036f]/g, "").toLowerCase().replace(/[^\S\n]+/g, " ");
8
13
  }
9
14
  var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
10
15
  "hdl",
@@ -41,73 +46,266 @@ var UNAMBIGUOUS_SHORT_NAMES = /* @__PURE__ */ new Set([
41
46
  "mlg",
42
47
  "tav"
43
48
  ]);
49
+ var CONTEXT_REQUIRED_NAMES = /* @__PURE__ */ new Set([
50
+ "bacteria",
51
+ // Bacteria_Urine — tem unidade, escapa da regra automática
52
+ "bacterias",
53
+ // Bacteria_Urine
54
+ "lead",
55
+ // Lead — verbo/substantivo comuníssimo em inglês
56
+ "peso",
57
+ // TotalMass
58
+ "saturation",
59
+ // TransferrinSaturation — "oxygen saturation", "saturation index"
60
+ "tap",
61
+ // ProthrombinTime — "tap" em inglês
62
+ "volume",
63
+ // VATVolume
64
+ "weight"
65
+ // TotalMass
66
+ ]);
67
+ var QUALITATIVE_VALUE_TERMS = /* @__PURE__ */ new Set([
68
+ "absent",
69
+ "alguns",
70
+ "amarela",
71
+ "amarelo",
72
+ "anormal",
73
+ "ausencia",
74
+ "ausente",
75
+ "ausentes",
76
+ "citrino",
77
+ "claro",
78
+ "clear",
79
+ "cloudy",
80
+ "colorless",
81
+ "detectado",
82
+ "detected",
83
+ "escuro",
84
+ "incolor",
85
+ "indetectavel",
86
+ "limpido",
87
+ "moderada",
88
+ "moderado",
89
+ "negativa",
90
+ "negative",
91
+ "negativo",
92
+ "normais",
93
+ "normal",
94
+ "numerosos",
95
+ "ocasional",
96
+ "positiva",
97
+ "positive",
98
+ "positivo",
99
+ "present",
100
+ "presente",
101
+ "presentes",
102
+ "raras",
103
+ "raro",
104
+ "raros",
105
+ "reagente",
106
+ "trace",
107
+ "traces",
108
+ "tracos",
109
+ "turvo",
110
+ "undetectable",
111
+ "yellow"
112
+ ]);
113
+ var GENETIC_CONTEXT_PATTERNS = [
114
+ /\b[nx][mrpc]_\d{6,}/,
115
+ // RefSeq: NM_000384.2, NP_, NR_, XM_
116
+ /\bens[gtp]\d{6,}/,
117
+ // Ensembl: ENSG00000084674
118
+ /\bp\.[a-z]{3}\d/,
119
+ // HGVS proteína: p.Trp448*
120
+ /\bc\.\d+[acgt]?[>_+-]/,
121
+ // HGVS codificante: c.1234A>G, c.76_78del
122
+ /\brs\d{4,}\b/,
123
+ // dbSNP
124
+ /\bgenes?\b/,
125
+ /\bvariante?s?\b/,
126
+ /\bexons?\b/,
127
+ /\bzygosity\b/,
128
+ /\bzigosidade\b/,
129
+ /\balleles?\b/,
130
+ /\balelos?\b/,
131
+ /\bmutations?\b/,
132
+ /\bmutac(ao|oes)\b/,
133
+ /\bpathogenic/,
134
+ /\bpatogenic/,
135
+ /\bheterozyg/,
136
+ /\bhomozyg/,
137
+ /\bheterozigot/,
138
+ /\bhomozigot/,
139
+ /\bsequence change\b/
140
+ ];
141
+ var DIGIT_PATTERN = /\d/;
142
+ var cachedUnitTokens = null;
143
+ function getUnitTokens() {
144
+ if (!cachedUnitTokens) {
145
+ cachedUnitTokens = new Set(
146
+ Object.keys(UNIT_TO_UCUM).map((unit) => normalize(unit).trim()).filter(Boolean)
147
+ );
148
+ }
149
+ return cachedUnitTokens;
150
+ }
44
151
  var cachedPatterns = null;
45
- var cachedNormalized = null;
152
+ var cachedNamePatterns = null;
46
153
  function getPatterns() {
47
154
  if (!cachedPatterns) {
48
155
  cachedPatterns = getAllSearchPatterns();
49
156
  }
50
157
  return cachedPatterns;
51
158
  }
52
- function getNormalizedPatterns() {
53
- if (!cachedNormalized) {
54
- const patterns = getPatterns();
159
+ function escapeRegExp(text) {
160
+ return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
161
+ }
162
+ function buildNamePattern(normalizedName) {
163
+ const body = normalizedName.split(" ").map(escapeRegExp).join("[^\\S\\n]+");
164
+ const plural = /\p{L}$/u.test(normalizedName) ? "s?" : "";
165
+ return new RegExp(`(?<![\\p{L}\\p{N}])${body}${plural}(?![\\p{L}\\p{N}])`, "gu");
166
+ }
167
+ function isQualitativeUrine(pattern) {
168
+ const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];
169
+ return categories.includes("urina") && !pattern.unit;
170
+ }
171
+ function isAmbiguousName(normalizedName, pattern) {
172
+ if (normalizedName.includes(" ")) {
173
+ return false;
174
+ }
175
+ return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);
176
+ }
177
+ function getNamePatterns() {
178
+ if (!cachedNamePatterns) {
55
179
  const map = /* @__PURE__ */ new Map();
56
- for (const pattern of patterns) {
180
+ for (const pattern of getPatterns()) {
57
181
  for (const name of pattern.names) {
58
- const normalized = normalize(name);
59
- const existing = map.get(normalized) || [];
60
- existing.push({
182
+ const normalized = normalize(name).trim();
183
+ if (!normalized) {
184
+ continue;
185
+ }
186
+ if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {
187
+ continue;
188
+ }
189
+ let slot = map.get(normalized);
190
+ if (!slot) {
191
+ slot = { entries: [], regex: null };
192
+ map.set(normalized, slot);
193
+ }
194
+ slot.entries.push({
195
+ ambiguous: isAmbiguousName(normalized, pattern),
61
196
  code: pattern.code,
62
197
  ...pattern.loinc && { loinc: pattern.loinc },
63
198
  original: name
64
199
  });
65
- map.set(normalized, existing);
66
200
  }
67
201
  }
68
- cachedNormalized = map;
202
+ cachedNamePatterns = map;
203
+ }
204
+ return cachedNamePatterns;
205
+ }
206
+ function getLineBounds(text, position) {
207
+ const start = text.lastIndexOf("\n", position) + 1;
208
+ const nextBreak = text.indexOf("\n", position);
209
+ return { end: nextBreak === -1 ? text.length : nextBreak, start };
210
+ }
211
+ function hasGeneticContext(line) {
212
+ return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));
213
+ }
214
+ function hasValueEvidence(line) {
215
+ if (DIGIT_PATTERN.test(line)) {
216
+ return true;
217
+ }
218
+ const unitTokens = getUnitTokens();
219
+ for (const token of line.split(/[^\p{L}\p{N}%/]+/u)) {
220
+ if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {
221
+ return true;
222
+ }
223
+ }
224
+ return false;
225
+ }
226
+ function collectCandidates(normalizedText) {
227
+ const candidates = [];
228
+ for (const [name, slot] of getNamePatterns()) {
229
+ if (!normalizedText.includes(name)) {
230
+ continue;
231
+ }
232
+ const { entries } = slot;
233
+ const regex = slot.regex ??= buildNamePattern(name);
234
+ regex.lastIndex = 0;
235
+ let occurrences = 0;
236
+ let match = regex.exec(normalizedText);
237
+ while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {
238
+ candidates.push({ end: match.index + match[0].length, entries, start: match.index });
239
+ occurrences += 1;
240
+ match = regex.exec(normalizedText);
241
+ }
242
+ }
243
+ return candidates;
244
+ }
245
+ function resolveOverlaps(candidates) {
246
+ const sorted = [...candidates].sort(
247
+ (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start
248
+ );
249
+ const accepted = [];
250
+ for (const candidate of sorted) {
251
+ const length = candidate.end - candidate.start;
252
+ const swallowed = accepted.some(
253
+ (other) => other.start <= candidate.start && candidate.end <= other.end && other.end - other.start > length
254
+ );
255
+ if (!swallowed) {
256
+ accepted.push(candidate);
257
+ }
69
258
  }
70
- return cachedNormalized;
259
+ return accepted;
71
260
  }
72
261
  function findBiomarkersInText(ocrText) {
73
262
  const startTime = Date.now();
74
263
  const normalizedText = normalize(ocrText);
75
- const matchedCodes = /* @__PURE__ */ new Set();
76
- const matches = [];
77
- const normalizedPatterns = getNormalizedPatterns();
78
- for (const [normalizedName, entries] of normalizedPatterns) {
79
- if (normalizedName.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalizedName)) {
264
+ const bestByCode = /* @__PURE__ */ new Map();
265
+ const geneticLines = /* @__PURE__ */ new Map();
266
+ const valueLines = /* @__PURE__ */ new Map();
267
+ for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {
268
+ const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);
269
+ let genetic = geneticLines.get(lineStart);
270
+ if (genetic === void 0) {
271
+ genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));
272
+ geneticLines.set(lineStart, genetic);
273
+ }
274
+ if (genetic) {
80
275
  continue;
81
276
  }
82
- let position = -1;
83
- if (normalizedName.length <= 4) {
84
- const regex = new RegExp(`\\b${normalizedName}\\b`);
85
- const match = regex.exec(normalizedText);
86
- if (match) {
87
- position = match.index;
88
- }
89
- } else {
90
- position = normalizedText.indexOf(normalizedName);
277
+ let hasValue = valueLines.get(lineStart);
278
+ if (hasValue === void 0) {
279
+ hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));
280
+ valueLines.set(lineStart, hasValue);
91
281
  }
92
- if (position !== -1) {
93
- for (const entry of entries) {
94
- if (!matchedCodes.has(entry.code)) {
95
- matchedCodes.add(entry.code);
96
- matches.push({
97
- code: entry.code,
98
- confidence: 1,
99
- loinc: entry.loinc,
100
- matchedName: entry.original,
101
- position
102
- });
103
- }
282
+ for (const entry of candidate.entries) {
283
+ if (entry.ambiguous && !hasValue) {
284
+ continue;
285
+ }
286
+ let confidence = CONFIDENCE_NAME_ONLY;
287
+ if (entry.ambiguous) {
288
+ confidence = CONFIDENCE_AMBIGUOUS;
289
+ } else if (hasValue) {
290
+ confidence = CONFIDENCE_VALUE_ADJACENT;
291
+ }
292
+ const existing = bestByCode.get(entry.code);
293
+ const better = !existing || confidence > existing.confidence || confidence === existing.confidence && candidate.start < existing.position;
294
+ if (better) {
295
+ bestByCode.set(entry.code, {
296
+ code: entry.code,
297
+ confidence,
298
+ loinc: entry.loinc,
299
+ matchedName: entry.original,
300
+ position: candidate.start
301
+ });
104
302
  }
105
303
  }
106
304
  }
305
+ const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);
107
306
  const scanTimeMs = Date.now() - startTime;
108
- const matchedCodesArray = Array.from(matchedCodes);
109
307
  return {
110
- filteredReference: generateFilteredLLMReference(matchedCodesArray),
308
+ filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),
111
309
  matches,
112
310
  stats: {
113
311
  matchedCount: matches.length,
@@ -120,6 +318,9 @@ function getMatchedCodes(result) {
120
318
  return result.matches.map((m) => m.code);
121
319
  }
122
320
  export {
321
+ CONFIDENCE_AMBIGUOUS,
322
+ CONFIDENCE_NAME_ONLY,
323
+ CONFIDENCE_VALUE_ADJACENT,
123
324
  findBiomarkersInText,
124
325
  getMatchedCodes
125
326
  };
package/dist/index.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/anchor.ts"],"sourcesContent":["/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Normalizes whitespace\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/\\s+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\ntype PatternEntry = { original: string; code: string; loinc?: string };\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNormalized: Map<string, PatternEntry[]> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction getNormalizedPatterns(): Map<string, PatternEntry[]> {\n if (!cachedNormalized) {\n const patterns = getPatterns();\n const map = new Map<string, PatternEntry[]>();\n for (const pattern of patterns) {\n for (const name of pattern.names) {\n const normalized = normalize(name);\n const existing = map.get(normalized) || [];\n existing.push({\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n map.set(normalized, existing);\n }\n }\n cachedNormalized = map;\n }\n return cachedNormalized;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n * Uses exact string matching on normalized text.\n * Returns unique matches (same biomarker won't be matched twice).\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const matchedCodes = new Set<string>();\n const matches: AnchorMatch[] = [];\n const normalizedPatterns = getNormalizedPatterns();\n\n for (const [normalizedName, entries] of normalizedPatterns) {\n if (normalizedName.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalizedName)) {\n continue;\n }\n\n let position = -1;\n if (normalizedName.length <= 4) {\n const regex = new RegExp(`\\\\b${normalizedName}\\\\b`);\n const match = regex.exec(normalizedText);\n if (match) {\n position = match.index;\n }\n } else {\n position = normalizedText.indexOf(normalizedName);\n }\n\n if (position !== -1) {\n for (const entry of entries) {\n if (!matchedCodes.has(entry.code)) {\n matchedCodes.add(entry.code);\n matches.push({\n code: entry.code,\n confidence: 1.0,\n loinc: entry.loinc,\n matchedName: entry.original,\n position,\n });\n }\n }\n }\n }\n\n const scanTimeMs = Date.now() - startTime;\n const matchedCodesArray = Array.from(matchedCodes);\n\n return {\n filteredReference: generateFilteredLLMReference(matchedCodesArray),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"],"mappings":";AAQA;AAAA,EAEE;AAAA,EACA;AAAA,OACK;AA0BP,SAAS,UAAU,MAAsB;AACvC,SAAO,KACJ,UAAU,KAAK,EACf,QAAQ,oBAAoB,EAAE,EAC9B,YAAY,EACZ,QAAQ,QAAQ,GAAG;AACxB;AAEA,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAID,IAAI,iBAAkD;AACtD,IAAI,mBAAuD;AAE3D,SAAS,cAAwC;AAC/C,MAAI,CAAC,gBAAgB;AACnB,qBAAiB,qBAAqB;AAAA,EACxC;AACA,SAAO;AACT;AAEA,SAAS,wBAAqD;AAC5D,MAAI,CAAC,kBAAkB;AACrB,UAAM,WAAW,YAAY;AAC7B,UAAM,MAAM,oBAAI,IAA4B;AAC5C,eAAW,WAAW,UAAU;AAC9B,iBAAW,QAAQ,QAAQ,OAAO;AAChC,cAAM,aAAa,UAAU,IAAI;AACjC,cAAM,WAAW,IAAI,IAAI,UAAU,KAAK,CAAC;AACzC,iBAAS,KAAK;AAAA,UACZ,MAAM,QAAQ;AAAA,UACd,GAAI,QAAQ,SAAS,EAAE,OAAO,QAAQ,MAAM;AAAA,UAC5C,UAAU;AAAA,QACZ,CAAC;AACD,YAAI,IAAI,YAAY,QAAQ;AAAA,MAC9B;AAAA,IACF;AACA,uBAAmB;AAAA,EACrB;AACA,SAAO;AACT;AAOO,SAAS,qBAAqB,SAA+B;AAClE,QAAM,YAAY,KAAK,IAAI;AAC3B,QAAM,iBAAiB,UAAU,OAAO;AACxC,QAAM,eAAe,oBAAI,IAAY;AACrC,QAAM,UAAyB,CAAC;AAChC,QAAM,qBAAqB,sBAAsB;AAEjD,aAAW,CAAC,gBAAgB,OAAO,KAAK,oBAAoB;AAC1D,QAAI,eAAe,SAAS,KAAK,CAAC,wBAAwB,IAAI,cAAc,GAAG;AAC7E;AAAA,IACF;AAEA,QAAI,WAAW;AACf,QAAI,eAAe,UAAU,GAAG;AAC9B,YAAM,QAAQ,IAAI,OAAO,MAAM,cAAc,KAAK;AAClD,YAAM,QAAQ,MAAM,KAAK,cAAc;AACvC,UAAI,OAAO;AACT,mBAAW,MAAM;AAAA,MACnB;AAAA,IACF,OAAO;AACL,iBAAW,eAAe,QAAQ,cAAc;AAAA,IAClD;AAEA,QAAI,aAAa,IAAI;AACnB,iBAAW,SAAS,SAAS;AAC3B,YAAI,CAAC,aAAa,IAAI,MAAM,IAAI,GAAG;AACjC,uBAAa,IAAI,MAAM,IAAI;AAC3B,kBAAQ,KAAK;AAAA,YACX,MAAM,MAAM;AAAA,YACZ,YAAY;AAAA,YACZ,OAAO,MAAM;AAAA,YACb,aAAa,MAAM;AAAA,YACnB;AAAA,UACF,CAAC;AAAA,QACH;AAAA,MACF;AAAA,IACF;AAAA,EACF;AAEA,QAAM,aAAa,KAAK,IAAI,IAAI;AAChC,QAAM,oBAAoB,MAAM,KAAK,YAAY;AAEjD,SAAO;AAAA,IACL,mBAAmB,6BAA6B,iBAAiB;AAAA,IACjE;AAAA,IACA,OAAO;AAAA,MACL,cAAc,QAAQ;AAAA,MACtB;AAAA,MACA,eAAe,YAAY,EAAE;AAAA,IAC/B;AAAA,EACF;AACF;AAKO,SAAS,gBAAgB,QAAgC;AAC9D,SAAO,OAAO,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI;AACzC;","names":[]}
1
+ {"version":3,"sources":["../src/anchor.ts"],"sourcesContent":["/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"],"mappings":";AAaA;AAAA,EAEE;AAAA,EACA;AAAA,EACA;AAAA,OACK;AAwBA,IAAM,4BAA4B;AAMlC,IAAM,uBAAuB;AAM7B,IAAM,uBAAuB;AAGpC,IAAM,2BAA2B;AASjC,SAAS,UAAU,MAAsB;AACvC,SAAO,KACJ,UAAU,KAAK,EACf,QAAQ,oBAAoB,EAAE,EAC9B,YAAY,EACZ,QAAQ,aAAa,GAAG;AAC7B;AAEA,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AAQD,IAAM,yBAAyB,oBAAI,IAAI;AAAA,EACrC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AACF,CAAC;AAOD,IAAM,0BAA0B,oBAAI,IAAI;AAAA,EACtC;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF,CAAC;AASD,IAAM,2BAAqC;AAAA,EACzC;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AAAA,EACA;AACF;AAEA,IAAM,gBAAgB;AAGtB,IAAI,mBAAuC;AAE3C,SAAS,gBAA6B;AACpC,MAAI,CAAC,kBAAkB;AACrB,uBAAmB,IAAI;AAAA,MACrB,OAAO,KAAK,YAAY,EACrB,IAAI,CAAC,SAAS,UAAU,IAAI,EAAE,KAAK,CAAC,EACpC,OAAO,OAAO;AAAA,IACnB;AAAA,EACF;AACA,SAAO;AACT;AAqBA,IAAI,iBAAkD;AACtD,IAAI,qBAAsD;AAE1D,SAAS,cAAwC;AAC/C,MAAI,CAAC,gBAAgB;AACnB,qBAAiB,qBAAqB;AAAA,EACxC;AACA,SAAO;AACT;AAEA,SAAS,aAAa,MAAsB;AAC1C,SAAO,KAAK,QAAQ,uBAAuB,MAAM;AACnD;AAkBA,SAAS,iBAAiB,gBAAgC;AACxD,QAAM,OAAO,eAAe,MAAM,GAAG,EAAE,IAAI,YAAY,EAAE,KAAK,YAAY;AAC1E,QAAM,SAAS,UAAU,KAAK,cAAc,IAAI,OAAO;AACvD,SAAO,IAAI,OAAO,sBAAsB,IAAI,GAAG,MAAM,sBAAsB,IAAI;AACjF;AAEA,SAAS,mBAAmB,SAA0C;AACpE,QAAM,aAAa,MAAM,QAAQ,QAAQ,QAAQ,IAAI,QAAQ,WAAW,CAAC,QAAQ,QAAQ;AACzF,SAAO,WAAW,SAAS,OAAO,KAAK,CAAC,QAAQ;AAClD;AAOA,SAAS,gBAAgB,gBAAwB,SAA0C;AACzF,MAAI,eAAe,SAAS,GAAG,GAAG;AAChC,WAAO;AAAA,EACT;AACA,SAAO,uBAAuB,IAAI,cAAc,KAAK,mBAAmB,OAAO;AACjF;AAEA,SAAS,kBAA4C;AACnD,MAAI,CAAC,oBAAoB;AACvB,UAAM,MAAM,oBAAI,IAAyB;AACzC,eAAW,WAAW,YAAY,GAAG;AACnC,iBAAW,QAAQ,QAAQ,OAAO;AAChC,cAAM,aAAa,UAAU,IAAI,EAAE,KAAK;AACxC,YAAI,CAAC,YAAY;AACf;AAAA,QACF;AACA,YAAI,WAAW,SAAS,KAAK,CAAC,wBAAwB,IAAI,UAAU,GAAG;AACrE;AAAA,QACF;AACA,YAAI,OAAO,IAAI,IAAI,UAAU;AAC7B,YAAI,CAAC,MAAM;AACT,iBAAO,EAAE,SAAS,CAAC,GAAG,OAAO,KAAK;AAClC,cAAI,IAAI,YAAY,IAAI;AAAA,QAC1B;AACA,aAAK,QAAQ,KAAK;AAAA,UAChB,WAAW,gBAAgB,YAAY,OAAO;AAAA,UAC9C,MAAM,QAAQ;AAAA,UACd,GAAI,QAAQ,SAAS,EAAE,OAAO,QAAQ,MAAM;AAAA,UAC5C,UAAU;AAAA,QACZ,CAAC;AAAA,MACH;AAAA,IACF;AACA,yBAAqB;AAAA,EACvB;AACA,SAAO;AACT;AAEA,SAAS,cAAc,MAAc,UAAkD;AACrF,QAAM,QAAQ,KAAK,YAAY,MAAM,QAAQ,IAAI;AACjD,QAAM,YAAY,KAAK,QAAQ,MAAM,QAAQ;AAC7C,SAAO,EAAE,KAAK,cAAc,KAAK,KAAK,SAAS,WAAW,MAAM;AAClE;AAEA,SAAS,kBAAkB,MAAuB;AAChD,SAAO,yBAAyB,KAAK,CAAC,YAAY,QAAQ,KAAK,IAAI,CAAC;AACtE;AAMA,SAAS,iBAAiB,MAAuB;AAC/C,MAAI,cAAc,KAAK,IAAI,GAAG;AAC5B,WAAO;AAAA,EACT;AACA,QAAM,aAAa,cAAc;AACjC,aAAW,SAAS,KAAK,MAAM,mBAAmB,GAAG;AACnD,QAAI,UAAU,WAAW,IAAI,KAAK,KAAK,wBAAwB,IAAI,KAAK,IAAI;AAC1E,aAAO;AAAA,IACT;AAAA,EACF;AACA,SAAO;AACT;AASA,SAAS,kBAAkB,gBAAqC;AAC9D,QAAM,aAA0B,CAAC;AACjC,aAAW,CAAC,MAAM,IAAI,KAAK,gBAAgB,GAAG;AAC5C,QAAI,CAAC,eAAe,SAAS,IAAI,GAAG;AAClC;AAAA,IACF;AACA,UAAM,EAAE,QAAQ,IAAI;AACpB,UAAM,QAAS,KAAK,UAAU,iBAAiB,IAAI;AACnD,UAAM,YAAY;AAClB,QAAI,cAAc;AAClB,QAAI,QAAQ,MAAM,KAAK,cAAc;AACrC,WAAO,UAAU,QAAQ,cAAc,0BAA0B;AAC/D,iBAAW,KAAK,EAAE,KAAK,MAAM,QAAQ,MAAM,CAAC,EAAE,QAAQ,SAAS,OAAO,MAAM,MAAM,CAAC;AACnF,qBAAe;AACf,cAAQ,MAAM,KAAK,cAAc;AAAA,IACnC;AAAA,EACF;AACA,SAAO;AACT;AAaA,SAAS,gBAAgB,YAAsC;AAC7D,QAAM,SAAS,CAAC,GAAG,UAAU,EAAE;AAAA,IAC7B,CAAC,GAAG,MAAM,EAAE,MAAM,EAAE,SAAS,EAAE,MAAM,EAAE,UAAU,EAAE,QAAQ,EAAE;AAAA,EAC/D;AACA,QAAM,WAAwB,CAAC;AAC/B,aAAW,aAAa,QAAQ;AAC9B,UAAM,SAAS,UAAU,MAAM,UAAU;AACzC,UAAM,YAAY,SAAS;AAAA,MACzB,CAAC,UACC,MAAM,SAAS,UAAU,SACzB,UAAU,OAAO,MAAM,OACvB,MAAM,MAAM,MAAM,QAAQ;AAAA,IAC9B;AACA,QAAI,CAAC,WAAW;AACd,eAAS,KAAK,SAAS;AAAA,IACzB;AAAA,EACF;AACA,SAAO;AACT;AAUO,SAAS,qBAAqB,SAA+B;AAClE,QAAM,YAAY,KAAK,IAAI;AAC3B,QAAM,iBAAiB,UAAU,OAAO;AACxC,QAAM,aAAa,oBAAI,IAAyB;AAChD,QAAM,eAAe,oBAAI,IAAqB;AAC9C,QAAM,aAAa,oBAAI,IAAqB;AAE5C,aAAW,aAAa,gBAAgB,kBAAkB,cAAc,CAAC,GAAG;AAC1E,UAAM,EAAE,KAAK,SAAS,OAAO,UAAU,IAAI,cAAc,gBAAgB,UAAU,KAAK;AAExF,QAAI,UAAU,aAAa,IAAI,SAAS;AACxC,QAAI,YAAY,QAAW;AACzB,gBAAU,kBAAkB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,mBAAa,IAAI,WAAW,OAAO;AAAA,IACrC;AACA,QAAI,SAAS;AACX;AAAA,IACF;AAEA,QAAI,WAAW,WAAW,IAAI,SAAS;AACvC,QAAI,aAAa,QAAW;AAC1B,iBAAW,iBAAiB,eAAe,MAAM,WAAW,OAAO,CAAC;AACpE,iBAAW,IAAI,WAAW,QAAQ;AAAA,IACpC;AAEA,eAAW,SAAS,UAAU,SAAS;AACrC,UAAI,MAAM,aAAa,CAAC,UAAU;AAChC;AAAA,MACF;AAEA,UAAI,aAAa;AACjB,UAAI,MAAM,WAAW;AACnB,qBAAa;AAAA,MACf,WAAW,UAAU;AACnB,qBAAa;AAAA,MACf;AAEA,YAAM,WAAW,WAAW,IAAI,MAAM,IAAI;AAC1C,YAAM,SACJ,CAAC,YACD,aAAa,SAAS,cACrB,eAAe,SAAS,cAAc,UAAU,QAAQ,SAAS;AACpE,UAAI,QAAQ;AACV,mBAAW,IAAI,MAAM,MAAM;AAAA,UACzB,MAAM,MAAM;AAAA,UACZ;AAAA,UACA,OAAO,MAAM;AAAA,UACb,aAAa,MAAM;AAAA,UACnB,UAAU,UAAU;AAAA,QACtB,CAAC;AAAA,MACH;AAAA,IACF;AAAA,EACF;AAEA,QAAM,UAAU,MAAM,KAAK,WAAW,OAAO,CAAC,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,WAAW,EAAE,QAAQ;AACtF,QAAM,aAAa,KAAK,IAAI,IAAI;AAEhC,SAAO;AAAA,IACL,mBAAmB,6BAA6B,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI,CAAC;AAAA,IAC1E;AAAA,IACA,OAAO;AAAA,MACL,cAAc,QAAQ;AAAA,MACtB;AAAA,MACA,eAAe,YAAY,EAAE;AAAA,IAC/B;AAAA,EACF;AACF;AAKO,SAAS,gBAAgB,QAAgC;AAC9D,SAAO,OAAO,QAAQ,IAAI,CAAC,MAAM,EAAE,IAAI;AACzC;","names":[]}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@precisa-saude/fhir-ocr-utils",
3
- "version": "0.16.3",
3
+ "version": "0.16.4",
4
4
  "description": "Utilitários de ancoragem OCR para extração de biomarcadores de PDFs de resultados laboratoriais",
5
5
  "keywords": [
6
6
  "fhir",
@@ -41,7 +41,7 @@
41
41
  "dist"
42
42
  ],
43
43
  "dependencies": {
44
- "@precisa-saude/fhir": "^0.16.3"
44
+ "@precisa-saude/fhir": "^0.16.4"
45
45
  },
46
46
  "devDependencies": {
47
47
  "tsup": "^8.3.5",