@precisa-saude/fhir-ocr-utils 0.25.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +387 -19
- package/dist/index.cjs +254 -2
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +260 -1
- package/dist/index.d.ts +260 -1
- package/dist/index.js +253 -1
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
package/dist/index.cjs
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"use strict";Object.defineProperty(exports, "__esModule", {value: true})
|
|
1
|
+
"use strict";Object.defineProperty(exports, "__esModule", {value: true}); function _nullishCoalesce(lhs, rhsFn) { if (lhs != null) { return lhs; } else { return rhsFn(); } } function _optionalChain(ops) { let lastAccessLHS = undefined; let value = ops[0]; let i = 1; while (i < ops.length) { const op = ops[i]; const fn = ops[i + 1]; i += 2; if ((op === 'optionalAccess' || op === 'optionalCall') && value == null) { return undefined; } if (op === 'access' || op === 'optionalAccess') { lastAccessLHS = value; value = fn(value); } else if (op === 'call' || op === 'optionalCall') { value = fn((...args) => value.call(lastAccessLHS, ...args)); lastAccessLHS = undefined; } } return value; }// src/anchor.ts
|
|
2
2
|
|
|
3
3
|
|
|
4
4
|
|
|
@@ -364,10 +364,262 @@ function getMatchedCodes(result) {
|
|
|
364
364
|
return result.matches.map((m) => m.code);
|
|
365
365
|
}
|
|
366
366
|
|
|
367
|
+
// src/extraction-schema.ts
|
|
368
|
+
var LAB_EXTRACTION_SCHEMA = {
|
|
369
|
+
$id: "https://fhir-brasil.dev.br/schemas/lab-extraction.json",
|
|
370
|
+
$schema: "https://json-schema.org/draft/2020-12/schema",
|
|
371
|
+
additionalProperties: false,
|
|
372
|
+
properties: {
|
|
373
|
+
biomarkers: {
|
|
374
|
+
description: "The measurements read from the report.",
|
|
375
|
+
items: {
|
|
376
|
+
additionalProperties: false,
|
|
377
|
+
properties: {
|
|
378
|
+
confidence: {
|
|
379
|
+
description: "Confidence in this reading, from 0 to 1.",
|
|
380
|
+
maximum: 1,
|
|
381
|
+
minimum: 0,
|
|
382
|
+
type: "number"
|
|
383
|
+
},
|
|
384
|
+
loinc: {
|
|
385
|
+
anyOf: [{ type: "string" }, { type: "null" }],
|
|
386
|
+
description: "A LOINC code from the allowed list, or null when none of them applies."
|
|
387
|
+
},
|
|
388
|
+
name: {
|
|
389
|
+
description: "The measurement name as the report prints it.",
|
|
390
|
+
type: "string"
|
|
391
|
+
},
|
|
392
|
+
referenceMax: {
|
|
393
|
+
anyOf: [{ type: "number" }, { type: "null" }],
|
|
394
|
+
description: "Upper bound of the range printed on the report, or null."
|
|
395
|
+
},
|
|
396
|
+
referenceMin: {
|
|
397
|
+
anyOf: [{ type: "number" }, { type: "null" }],
|
|
398
|
+
description: "Lower bound of the range printed on the report, or null."
|
|
399
|
+
},
|
|
400
|
+
sourceText: {
|
|
401
|
+
description: "The snippet of the report carrying this measurement and its value.",
|
|
402
|
+
type: "string"
|
|
403
|
+
},
|
|
404
|
+
unit: {
|
|
405
|
+
description: "Unit as the report prints it. Empty string when there is none.",
|
|
406
|
+
type: "string"
|
|
407
|
+
},
|
|
408
|
+
value: {
|
|
409
|
+
anyOf: [{ type: "number" }, { type: "string" }],
|
|
410
|
+
description: "Numeric value, or text for a qualitative result."
|
|
411
|
+
}
|
|
412
|
+
},
|
|
413
|
+
required: ["name", "value", "unit", "sourceText", "confidence"],
|
|
414
|
+
type: "object"
|
|
415
|
+
},
|
|
416
|
+
type: "array"
|
|
417
|
+
}
|
|
418
|
+
},
|
|
419
|
+
required: ["biomarkers"],
|
|
420
|
+
title: "Laboratory report extraction",
|
|
421
|
+
type: "object"
|
|
422
|
+
};
|
|
423
|
+
|
|
424
|
+
// src/extraction-to-lab-result.ts
|
|
425
|
+
|
|
426
|
+
function flagFor(b) {
|
|
427
|
+
if (typeof b.value !== "number") return "";
|
|
428
|
+
if (typeof b.referenceMax === "number" && b.value > b.referenceMax) return "H";
|
|
429
|
+
if (typeof b.referenceMin === "number" && b.value < b.referenceMin) return "L";
|
|
430
|
+
return "";
|
|
431
|
+
}
|
|
432
|
+
function extractionToLabResult(biomarkers, options = {}) {
|
|
433
|
+
const reportId = _nullishCoalesce(options.reportId, () => ( "laudo-demo"));
|
|
434
|
+
const userId = _nullishCoalesce(options.userId, () => ( "paciente-demo"));
|
|
435
|
+
const collectionDate = _nullishCoalesce(options.collectionDate, () => ( (/* @__PURE__ */ new Date()).toISOString().slice(0, 10)));
|
|
436
|
+
const observations = biomarkers.flatMap((b) => {
|
|
437
|
+
const code = b.loinc ? _fhir.loincToCode.call(void 0, b.loinc) : void 0;
|
|
438
|
+
if (!code) return [];
|
|
439
|
+
return [
|
|
440
|
+
{
|
|
441
|
+
biomarkerCode: code,
|
|
442
|
+
biomarkerName: b.name,
|
|
443
|
+
flag: flagFor(b),
|
|
444
|
+
...typeof b.referenceMax === "number" ? { referenceMax: b.referenceMax } : {},
|
|
445
|
+
...typeof b.referenceMin === "number" ? { referenceMin: b.referenceMin } : {},
|
|
446
|
+
reportId,
|
|
447
|
+
unit: b.unit,
|
|
448
|
+
value: b.value
|
|
449
|
+
}
|
|
450
|
+
];
|
|
451
|
+
});
|
|
452
|
+
return {
|
|
453
|
+
observations,
|
|
454
|
+
profile: { name: "Paciente de Demonstra\xE7\xE3o", userId },
|
|
455
|
+
report: {
|
|
456
|
+
collectionDate,
|
|
457
|
+
createdAt: `${collectionDate}T00:00:00Z`,
|
|
458
|
+
overallStatus: observations.some((o) => o.flag !== "") ? "ANORMAL" : "NORMAL",
|
|
459
|
+
processingStatus: "complete",
|
|
460
|
+
reportId,
|
|
461
|
+
userId
|
|
462
|
+
}
|
|
463
|
+
};
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
// src/extraction-validator.ts
|
|
467
|
+
|
|
468
|
+
var isRecord = (v) => typeof v === "object" && v !== null && !Array.isArray(v);
|
|
469
|
+
function schemaErrors(raw) {
|
|
470
|
+
if (!isRecord(raw)) return ["n\xE3o \xE9 um objeto"];
|
|
471
|
+
const errors = [];
|
|
472
|
+
const { confidence, loinc, name, referenceMax, referenceMin, sourceText, unit, value } = raw;
|
|
473
|
+
if (typeof name !== "string" || name.length === 0) errors.push("`name` ausente ou vazio");
|
|
474
|
+
if (typeof sourceText !== "string" || sourceText.length === 0)
|
|
475
|
+
errors.push("`sourceText` ausente ou vazio");
|
|
476
|
+
if (typeof unit !== "string") errors.push("`unit` ausente");
|
|
477
|
+
if (typeof value !== "number" && typeof value !== "string") errors.push("`value` ausente");
|
|
478
|
+
if (typeof confidence !== "number" || confidence < 0 || confidence > 1)
|
|
479
|
+
errors.push("`confidence` fora de 0..1");
|
|
480
|
+
if (loinc !== void 0 && loinc !== null && typeof loinc !== "string")
|
|
481
|
+
errors.push("`loinc` n\xE3o \xE9 string nem null");
|
|
482
|
+
for (const [key, v] of [
|
|
483
|
+
["referenceMax", referenceMax],
|
|
484
|
+
["referenceMin", referenceMin]
|
|
485
|
+
]) {
|
|
486
|
+
if (v !== void 0 && v !== null && typeof v !== "number")
|
|
487
|
+
errors.push(`\`${key}\` n\xE3o \xE9 n\xFAmero nem null`);
|
|
488
|
+
}
|
|
489
|
+
return errors;
|
|
490
|
+
}
|
|
491
|
+
function allowedKeys(anchors) {
|
|
492
|
+
const allowed = /* @__PURE__ */ new Set();
|
|
493
|
+
for (const match of anchors.matches) {
|
|
494
|
+
allowed.add(match.code);
|
|
495
|
+
if (match.loinc) allowed.add(match.loinc);
|
|
496
|
+
}
|
|
497
|
+
return allowed;
|
|
498
|
+
}
|
|
499
|
+
function validateExtraction(raw, options = {}) {
|
|
500
|
+
const { anchors } = options;
|
|
501
|
+
if (!isRecord(raw)) {
|
|
502
|
+
return { accepted: [], errors: ["a sa\xEDda n\xE3o \xE9 um objeto JSON"], rejected: [], valid: false };
|
|
503
|
+
}
|
|
504
|
+
if (!Array.isArray(raw.biomarkers)) {
|
|
505
|
+
return {
|
|
506
|
+
accepted: [],
|
|
507
|
+
errors: ["`biomarkers` ausente ou n\xE3o \xE9 lista"],
|
|
508
|
+
rejected: [],
|
|
509
|
+
valid: false
|
|
510
|
+
};
|
|
511
|
+
}
|
|
512
|
+
const allowed = anchors ? allowedKeys(anchors) : void 0;
|
|
513
|
+
const accepted = [];
|
|
514
|
+
const rejected = [];
|
|
515
|
+
for (const entry of raw.biomarkers) {
|
|
516
|
+
const problems = schemaErrors(entry);
|
|
517
|
+
if (problems.length > 0) {
|
|
518
|
+
rejected.push({ detail: problems.join("; "), raw: entry, reason: "schema" });
|
|
519
|
+
continue;
|
|
520
|
+
}
|
|
521
|
+
const biomarker = entry;
|
|
522
|
+
if (allowed) {
|
|
523
|
+
const loinc = _nullishCoalesce(biomarker.loinc, () => ( void 0));
|
|
524
|
+
if (!loinc) {
|
|
525
|
+
rejected.push({
|
|
526
|
+
detail: `"${biomarker.name}" veio sem c\xF3digo LOINC`,
|
|
527
|
+
raw: entry,
|
|
528
|
+
reason: "not-anchored"
|
|
529
|
+
});
|
|
530
|
+
continue;
|
|
531
|
+
}
|
|
532
|
+
const internalCode = _fhir.loincToCode.call(void 0, loinc);
|
|
533
|
+
const isAnchored = allowed.has(loinc) || internalCode !== void 0 && allowed.has(internalCode);
|
|
534
|
+
if (!isAnchored) {
|
|
535
|
+
rejected.push({
|
|
536
|
+
detail: `${loinc} n\xE3o foi ancorado no texto de origem`,
|
|
537
|
+
raw: entry,
|
|
538
|
+
reason: "not-anchored"
|
|
539
|
+
});
|
|
540
|
+
continue;
|
|
541
|
+
}
|
|
542
|
+
}
|
|
543
|
+
accepted.push(biomarker);
|
|
544
|
+
}
|
|
545
|
+
return { accepted, errors: [], rejected, valid: true };
|
|
546
|
+
}
|
|
547
|
+
function acceptedBiomarkers(raw, options = {}) {
|
|
548
|
+
return validateExtraction(raw, options).accepted;
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
// src/llm-client.ts
|
|
552
|
+
function buildPrompt(text, allowed) {
|
|
553
|
+
return [
|
|
554
|
+
"Extract the laboratory results from the report below.",
|
|
555
|
+
"",
|
|
556
|
+
"Return JSON matching this schema, and nothing else:",
|
|
557
|
+
JSON.stringify(LAB_EXTRACTION_SCHEMA),
|
|
558
|
+
"",
|
|
559
|
+
"Use only LOINC codes from this list:",
|
|
560
|
+
allowed,
|
|
561
|
+
"",
|
|
562
|
+
"REPORT:",
|
|
563
|
+
text
|
|
564
|
+
].join("\n");
|
|
565
|
+
}
|
|
566
|
+
function stripFence(raw) {
|
|
567
|
+
const fenced = /```(?:json)?\s*([\s\S]*?)```/.exec(raw);
|
|
568
|
+
return (_nullishCoalesce(_optionalChain([fenced, 'optionalAccess', _ => _[1]]), () => ( raw))).trim();
|
|
569
|
+
}
|
|
570
|
+
async function extractWithModel(text, options) {
|
|
571
|
+
const { apiKey, baseUrl, model, responseFormat, timeoutMs = 3e5 } = options;
|
|
572
|
+
const allowed = findBiomarkersInText(text).filteredReference;
|
|
573
|
+
const startedAt = Date.now();
|
|
574
|
+
const formatBody = (mode2) => JSON.stringify({
|
|
575
|
+
messages: [{ content: buildPrompt(text, allowed), role: "user" }],
|
|
576
|
+
model,
|
|
577
|
+
...mode2 === "json_object" ? { response_format: { type: "json_object" } } : {},
|
|
578
|
+
...mode2 === "json_schema" ? {
|
|
579
|
+
response_format: {
|
|
580
|
+
json_schema: { name: "lab_extraction", schema: LAB_EXTRACTION_SCHEMA, strict: true },
|
|
581
|
+
type: "json_schema"
|
|
582
|
+
}
|
|
583
|
+
} : {},
|
|
584
|
+
temperature: 0
|
|
585
|
+
});
|
|
586
|
+
const post = async (mode2) => fetch(`${baseUrl.replace(/\/$/, "")}/chat/completions`, {
|
|
587
|
+
body: formatBody(mode2),
|
|
588
|
+
headers: {
|
|
589
|
+
"content-type": "application/json",
|
|
590
|
+
...apiKey ? { authorization: `Bearer ${apiKey}` } : {}
|
|
591
|
+
},
|
|
592
|
+
method: "POST",
|
|
593
|
+
signal: AbortSignal.timeout(timeoutMs)
|
|
594
|
+
});
|
|
595
|
+
const mode = _nullishCoalesce(responseFormat, () => ( "auto"));
|
|
596
|
+
let response = await post(mode === "auto" ? "json_schema" : mode);
|
|
597
|
+
if (!response.ok && mode === "auto" && response.status >= 400 && response.status < 500) {
|
|
598
|
+
response = await post("none");
|
|
599
|
+
}
|
|
600
|
+
if (!response.ok) {
|
|
601
|
+
throw new Error(`${String(response.status)} de ${baseUrl}: ${await response.text()}`);
|
|
602
|
+
}
|
|
603
|
+
const body = await response.json();
|
|
604
|
+
const raw = _nullishCoalesce(_optionalChain([body, 'access', _2 => _2.choices, 'optionalAccess', _3 => _3[0], 'optionalAccess', _4 => _4.message, 'optionalAccess', _5 => _5.content]), () => ( ""));
|
|
605
|
+
const tookMs = Date.now() - startedAt;
|
|
606
|
+
try {
|
|
607
|
+
return { payload: JSON.parse(stripFence(raw)), raw, tookMs };
|
|
608
|
+
} catch (e) {
|
|
609
|
+
throw new Error(`O modelo n\xE3o devolveu JSON analis\xE1vel. Resposta crua:
|
|
610
|
+
${raw.slice(0, 500)}`);
|
|
611
|
+
}
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
|
|
617
|
+
|
|
618
|
+
|
|
367
619
|
|
|
368
620
|
|
|
369
621
|
|
|
370
622
|
|
|
371
623
|
|
|
372
|
-
exports.CONFIDENCE_AMBIGUOUS = CONFIDENCE_AMBIGUOUS; exports.CONFIDENCE_NAME_ONLY = CONFIDENCE_NAME_ONLY; exports.CONFIDENCE_VALUE_ADJACENT = CONFIDENCE_VALUE_ADJACENT; exports.findBiomarkersInText = findBiomarkersInText; exports.getMatchedCodes = getMatchedCodes;
|
|
624
|
+
exports.CONFIDENCE_AMBIGUOUS = CONFIDENCE_AMBIGUOUS; exports.CONFIDENCE_NAME_ONLY = CONFIDENCE_NAME_ONLY; exports.CONFIDENCE_VALUE_ADJACENT = CONFIDENCE_VALUE_ADJACENT; exports.LAB_EXTRACTION_SCHEMA = LAB_EXTRACTION_SCHEMA; exports.acceptedBiomarkers = acceptedBiomarkers; exports.extractWithModel = extractWithModel; exports.extractionToLabResult = extractionToLabResult; exports.findBiomarkersInText = findBiomarkersInText; exports.getMatchedCodes = getMatchedCodes; exports.validateExtraction = validateExtraction;
|
|
373
625
|
//# sourceMappingURL=index.cjs.map
|
package/dist/index.cjs.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts"],"names":[],"mappings":"AAAA;ACaA;AAEE;AACA;AACA;AAAA,2CACK;AAwBA,IAAM,0BAAA,EAA4B,CAAA;AAMlC,IAAM,qBAAA,EAAuB,GAAA;AAM7B,IAAM,qBAAA,EAAuB,GAAA;AAGpC,IAAM,yBAAA,EAA2B,CAAA;AA8BjC,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,+BAAA,EAAiC,GAAG,CAAA,CAC5C,OAAA,CAAQ,WAAA,EAAa,GAAG,CAAA;AAC7B;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAQD,IAAM,uBAAA,kBAAyB,IAAI,GAAA,CAAI;AAAA,EACrC,UAAA;AAAA;AAAA,EACA,WAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,YAAA;AAAA;AAAA,EACA,KAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,UAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,YAAA;AAAA,EACA,OAAA;AAAA,EACA,SAAA;AAAA,EACA;AACF,CAAC,CAAA;AAOD,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,QAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,OAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,cAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,OAAA;AAAA,EACA,UAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,QAAA;AAAA,EACA,OAAA;AAAA,EACA,cAAA;AAAA,EACA;AACF,CAAC,CAAA;AASD,IAAM,yBAAA,EAAqC;AAAA,EACzC,qBAAA;AAAA;AAAA,EACA,kBAAA;AAAA;AAAA,EACA,iBAAA;AAAA;AAAA,EACA,uBAAA;AAAA;AAAA,EACA,cAAA;AAAA;AAAA,EACA,YAAA;AAAA,EACA,iBAAA;AAAA,EACA,YAAA;AAAA,EACA,cAAA;AAAA,EACA,gBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,gBAAA;AAAA,EACA,mBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,aAAA;AAAA,EACA,WAAA;AAAA,EACA,eAAA;AAAA,EACA,aAAA;AAAA,EACA;AACF,CAAA;AASA,IAAM,oBAAA,kBAAsB,IAAI,GAAA,CAAI;AAAA,EAClC,mBAAA;AAAA,EACA,eAAA;AAAA,EACA,qBAAA;AAAA,EACA,qBAAA;AAAA,EACA,oBAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAC,CAAA;AAGD,IAAM,uBAAA,EAAmC;AAAA,EACvC,mBAAA;AAAA,EACA,sBAAA;AAAA,EACA,iBAAA;AAAA,EACA;AACF,CAAA;AAOA,IAAM,0BAAA,EAAsC,CAAC,aAAA,EAAe,kBAAA,EAAoB,aAAa,CAAA;AAa7F,IAAM,iBAAA,EAAmB,uBAAA;AAEzB,SAAS,eAAA,CAAgB,IAAA,EAAuB;AAC9C,EAAA,GAAA,CAAI,yBAAA,CAA0B,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,CAAA,EAAG;AACzD,IAAA,OAAO,KAAA;AAAA,EACT;AACA,EAAA,OAAO,sBAAA,CAAuB,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,EAAA,GAAK,gBAAA,CAAiB,IAAA,CAAK,IAAI,CAAA;AACzF;AAEA,IAAM,cAAA,EAAgB,IAAA;AAGtB,IAAI,iBAAA,EAAuC,IAAA;AAE3C,SAAS,aAAA,CAAA,EAA6B;AACpC,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,iBAAA,EAAmB,IAAI,GAAA;AAAA,MACrB,MAAA,CAAO,IAAA,CAAK,kBAAY,CAAA,CACrB,GAAA,CAAI,CAAC,IAAA,EAAA,GAAS,SAAA,CAAU,IAAI,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,CACpC,MAAA,CAAO,OAAO;AAAA,IACnB,CAAA;AAAA,EACF;AACA,EAAA,OAAO,gBAAA;AACT;AAqBA,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,mBAAA,EAAsD,IAAA;AAE1D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,YAAA,CAAa,IAAA,EAAsB;AAC1C,EAAA,OAAO,IAAA,CAAK,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AACnD;AAkBA,SAAS,gBAAA,CAAiB,cAAA,EAAgC;AACxD,EAAA,MAAM,KAAA,EAAO,cAAA,CAAe,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,YAAY,CAAA,CAAE,IAAA,CAAK,YAAY,CAAA;AAC1E,EAAA,MAAM,OAAA,EAAS,SAAA,CAAU,IAAA,CAAK,cAAc,EAAA,EAAI,KAAA,EAAO,EAAA;AACvD,EAAA,OAAO,IAAI,MAAA,CAAO,CAAA,mBAAA,EAAsB,IAAI,CAAA,EAAA;AAC9C;AAEsE;AAC3B,EAAA;AACD,EAAA;AAC1C;AAOiD;AACb,EAAA;AACzB,IAAA;AACT,EAAA;AACkC,EAAA;AACpC;AAEqD;AAC1B,EAAA;AACkB,IAAA;AACJ,IAAA;AACD,MAAA;AACG,QAAA;AAClB,QAAA;AACf,UAAA;AACF,QAAA;AAC8B,QAAA;AAC5B,UAAA;AACF,QAAA;AAC6B,QAAA;AAClB,QAAA;AACyB,UAAA;AACV,UAAA;AAC1B,QAAA;AACkB,QAAA;AACW,UAAA;AACb,UAAA;AACwB,UAAA;AAC5B,UAAA;AACX,QAAA;AACH,MAAA;AACF,IAAA;AACqB,IAAA;AACvB,EAAA;AACO,EAAA;AACT;AAEuF;AACxC,EAAA;AACA,EAAA;AACP,EAAA;AACxC;AAEkD;AACV,EAAA;AACxC;AAMiD;AACjB,EAAA;AACrB,IAAA;AACT,EAAA;AACiC,EAAA;AACF,EAAA;AACU,IAAA;AAC9B,MAAA;AACT,IAAA;AACF,EAAA;AACO,EAAA;AACT;AASgE;AAC7B,EAAA;AACU,EAAA;AACL,IAAA;AAClC,MAAA;AACF,IAAA;AACoB,IAAA;AACU,IAAA;AACZ,IAAA;AACA,IAAA;AACmB,IAAA;AACE,IAAA;AACA,MAAA;AACtB,MAAA;AACkB,MAAA;AACnC,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAa+D;AAC9B,EAAA;AACU,IAAA;AACzC,EAAA;AAC+B,EAAA;AACC,EAAA;AACW,IAAA;AACd,IAAA;AAEE,MAAA;AAG7B,IAAA;AACgB,IAAA;AACS,MAAA;AACzB,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAUoE;AACvC,EAAA;AACa,EAAA;AACQ,EAAA;AACF,EAAA;AACF,EAAA;AACA,EAAA;AAEJ,EAAA;AACK,IAAA;AAEH,IAAA;AACb,IAAA;AACG,MAAA;AACO,MAAA;AACrC,IAAA;AACa,IAAA;AACX,MAAA;AACF,IAAA;AAEuC,IAAA;AACX,IAAA;AACE,MAAA;AACM,MAAA;AACpC,IAAA;AAEoC,IAAA;AACX,IAAA;AACgB,MAAA;AACR,MAAA;AACjC,IAAA;AAEuC,IAAA;AACH,MAAA;AAChC,QAAA;AACF,MAAA;AAEqC,MAAA;AACnC,QAAA;AACF,MAAA;AAEiB,MAAA;AACI,MAAA;AACN,QAAA;AACM,MAAA;AACN,QAAA;AACf,MAAA;AAEsC,MAAA;AAGvB,MAAA;AAEH,MAAA;AACiB,QAAA;AACb,UAAA;AACZ,UAAA;AACa,UAAA;AACM,UAAA;AACC,UAAA;AACrB,QAAA;AACH,MAAA;AACF,IAAA;AACF,EAAA;AAE6C,EAAA;AACb,EAAA;AAEzB,EAAA;AACc,IAAA;AACnB,IAAA;AACO,IAAA;AACiB,MAAA;AACtB,MAAA;AAC6B,MAAA;AAC/B,IAAA;AACF,EAAA;AACF;AAKgE;AACvB,EAAA;AACzC;ADpO+C;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Treats a hyphen that joins words as a space\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n *\n * O hífen entre palavras vira espaço porque o catálogo e o laboratório\n * discordam sobre ele o tempo todo: o catálogo escreve \"Proteína C-Reativa\" e\n * \"High-Density Lipoprotein\", e os laudos imprimem \"Proteína C Reativa\" e\n * \"High Density Lipoprotein\". Sem essa equivalência, 82 dos 159 nomes com\n * hífen deixam de ancorar na grafia que o documento usa.\n *\n * Não era teórico: um GGT de verdade foi descartado como alucinação em 27\n * laudos porque o documento escrevia \"Gama glutamil transferase\" e o catálogo\n * \"Gama-Glutamil Transferase\". Um caractere derrubava o valor antes de\n * qualquer validação.\n *\n * A troca exige **letra antes** do hífen, e por isso não toca em número:\n * o `-2.5` de um T-score e o `0-5` de uma faixa de urina seguem intactos.\n * Trocar sem essa guarda apagaria o sinal de um valor negativo, que é bem\n * pior que o problema original.\n *\n * O pré-filtro de substring do `collectCandidates` compara a chave com o texto\n * já normalizado, então a equivalência precisa nascer aqui: aplicada só na\n * regex, o `includes` descartaria o nome antes de ela rodar.\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/(?<=\\p{L})-(?=[\\p{L}\\p{N}])/gu, ' ')\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n"]}
|
|
1
|
+
{"version":3,"sources":["/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","../src/anchor.ts","../src/extraction-schema.ts","../src/extraction-to-lab-result.ts","../src/extraction-validator.ts","../src/llm-client.ts"],"names":[],"mappings":"AAAA;ACaA;AAEE;AACA;AACA;AAAA,2CACK;AAwBA,IAAM,0BAAA,EAA4B,CAAA;AAMlC,IAAM,qBAAA,EAAuB,GAAA;AAM7B,IAAM,qBAAA,EAAuB,GAAA;AAGpC,IAAM,yBAAA,EAA2B,CAAA;AA8BjC,SAAS,SAAA,CAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,IAAA,CACJ,SAAA,CAAU,KAAK,CAAA,CACf,OAAA,CAAQ,kBAAA,EAAoB,EAAE,CAAA,CAC9B,WAAA,CAAY,CAAA,CACZ,OAAA,CAAQ,+BAAA,EAAiC,GAAG,CAAA,CAC5C,OAAA,CAAQ,WAAA,EAAa,GAAG,CAAA;AAC7B;AAEA,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,KAAA;AAAA,EACA,KAAA;AAAA,EACA,IAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA,KAAA;AAAA,EACA;AACF,CAAC,CAAA;AAQD,IAAM,uBAAA,kBAAyB,IAAI,GAAA,CAAI;AAAA,EACrC,UAAA;AAAA;AAAA,EACA,WAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,MAAA;AAAA;AAAA,EACA,YAAA;AAAA;AAAA,EACA,KAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA,EACA,QAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAKA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,UAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,YAAA;AAAA,EACA,OAAA;AAAA,EACA,SAAA;AAAA,EACA;AACF,CAAC,CAAA;AAOD,IAAM,wBAAA,kBAA0B,IAAI,GAAA,CAAI;AAAA,EACtC,QAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,OAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,cAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,QAAA;AAAA,EACA,WAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,OAAA;AAAA,EACA,MAAA;AAAA,EACA,OAAA;AAAA,EACA,UAAA;AAAA,EACA,OAAA;AAAA,EACA,QAAA;AAAA,EACA,QAAA;AAAA,EACA,OAAA;AAAA,EACA,cAAA;AAAA,EACA;AACF,CAAC,CAAA;AASD,IAAM,yBAAA,EAAqC;AAAA,EACzC,qBAAA;AAAA;AAAA,EACA,kBAAA;AAAA;AAAA,EACA,iBAAA;AAAA;AAAA,EACA,uBAAA;AAAA;AAAA,EACA,cAAA;AAAA;AAAA,EACA,YAAA;AAAA,EACA,iBAAA;AAAA,EACA,YAAA;AAAA,EACA,cAAA;AAAA,EACA,gBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,gBAAA;AAAA,EACA,mBAAA;AAAA,EACA,cAAA;AAAA,EACA,aAAA;AAAA,EACA,aAAA;AAAA,EACA,WAAA;AAAA,EACA,eAAA;AAAA,EACA,aAAA;AAAA,EACA;AACF,CAAA;AASA,IAAM,oBAAA,kBAAsB,IAAI,GAAA,CAAI;AAAA,EAClC,mBAAA;AAAA,EACA,eAAA;AAAA,EACA,qBAAA;AAAA,EACA,qBAAA;AAAA,EACA,oBAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAC,CAAA;AAGD,IAAM,uBAAA,EAAmC;AAAA,EACvC,mBAAA;AAAA,EACA,sBAAA;AAAA,EACA,iBAAA;AAAA,EACA;AACF,CAAA;AAOA,IAAM,0BAAA,EAAsC,CAAC,aAAA,EAAe,kBAAA,EAAoB,aAAa,CAAA;AAa7F,IAAM,iBAAA,EAAmB,uBAAA;AAEzB,SAAS,eAAA,CAAgB,IAAA,EAAuB;AAC9C,EAAA,GAAA,CAAI,yBAAA,CAA0B,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,CAAA,EAAG;AACzD,IAAA,OAAO,KAAA;AAAA,EACT;AACA,EAAA,OAAO,sBAAA,CAAuB,IAAA,CAAK,CAAC,EAAA,EAAA,GAAO,EAAA,CAAG,IAAA,CAAK,IAAI,CAAC,EAAA,GAAK,gBAAA,CAAiB,IAAA,CAAK,IAAI,CAAA;AACzF;AAEA,IAAM,cAAA,EAAgB,IAAA;AAGtB,IAAI,iBAAA,EAAuC,IAAA;AAE3C,SAAS,aAAA,CAAA,EAA6B;AACpC,EAAA,GAAA,CAAI,CAAC,gBAAA,EAAkB;AACrB,IAAA,iBAAA,EAAmB,IAAI,GAAA;AAAA,MACrB,MAAA,CAAO,IAAA,CAAK,kBAAY,CAAA,CACrB,GAAA,CAAI,CAAC,IAAA,EAAA,GAAS,SAAA,CAAU,IAAI,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,CACpC,MAAA,CAAO,OAAO;AAAA,IACnB,CAAA;AAAA,EACF;AACA,EAAA,OAAO,gBAAA;AACT;AAqBA,IAAI,eAAA,EAAkD,IAAA;AACtD,IAAI,mBAAA,EAAsD,IAAA;AAE1D,SAAS,WAAA,CAAA,EAAwC;AAC/C,EAAA,GAAA,CAAI,CAAC,cAAA,EAAgB;AACnB,IAAA,eAAA,EAAiB,wCAAA,CAAqB;AAAA,EACxC;AACA,EAAA,OAAO,cAAA;AACT;AAEA,SAAS,YAAA,CAAa,IAAA,EAAsB;AAC1C,EAAA,OAAO,IAAA,CAAK,OAAA,CAAQ,qBAAA,EAAuB,MAAM,CAAA;AACnD;AAkBA,SAAS,gBAAA,CAAiB,cAAA,EAAgC;AACxD,EAAA,MAAM,KAAA,EAAO,cAAA,CAAe,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,YAAY,CAAA,CAAE,IAAA,CAAK,YAAY,CAAA;AAC1E,EAAA,MAAM,OAAA,EAAS,SAAA,CAAU,IAAA,CAAK,cAAc,EAAA,EAAI,KAAA,EAAO,EAAA;AACvD,EAAA,OAAO,IAAI,MAAA,CAAO,CAAA,mBAAA,EAAsB,IAAI,CAAA,EAAA;AAC9C;AAEsE;AAC3B,EAAA;AACD,EAAA;AAC1C;AAOiD;AACb,EAAA;AACzB,IAAA;AACT,EAAA;AACkC,EAAA;AACpC;AAEqD;AAC1B,EAAA;AACkB,IAAA;AACJ,IAAA;AACD,MAAA;AACG,QAAA;AAClB,QAAA;AACf,UAAA;AACF,QAAA;AAC8B,QAAA;AAC5B,UAAA;AACF,QAAA;AAC6B,QAAA;AAClB,QAAA;AACyB,UAAA;AACV,UAAA;AAC1B,QAAA;AACkB,QAAA;AACW,UAAA;AACb,UAAA;AACwB,UAAA;AAC5B,UAAA;AACX,QAAA;AACH,MAAA;AACF,IAAA;AACqB,IAAA;AACvB,EAAA;AACO,EAAA;AACT;AAEuF;AACxC,EAAA;AACA,EAAA;AACP,EAAA;AACxC;AAEkD;AACV,EAAA;AACxC;AAMiD;AACjB,EAAA;AACrB,IAAA;AACT,EAAA;AACiC,EAAA;AACF,EAAA;AACU,IAAA;AAC9B,MAAA;AACT,IAAA;AACF,EAAA;AACO,EAAA;AACT;AASgE;AAC7B,EAAA;AACU,EAAA;AACL,IAAA;AAClC,MAAA;AACF,IAAA;AACoB,IAAA;AACU,IAAA;AACZ,IAAA;AACA,IAAA;AACmB,IAAA;AACE,IAAA;AACA,MAAA;AACtB,MAAA;AACkB,MAAA;AACnC,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAa+D;AAC9B,EAAA;AACU,IAAA;AACzC,EAAA;AAC+B,EAAA;AACC,EAAA;AACW,IAAA;AACd,IAAA;AAEE,MAAA;AAG7B,IAAA;AACgB,IAAA;AACS,MAAA;AACzB,IAAA;AACF,EAAA;AACO,EAAA;AACT;AAUoE;AACvC,EAAA;AACa,EAAA;AACQ,EAAA;AACF,EAAA;AACF,EAAA;AACA,EAAA;AAEJ,EAAA;AACK,IAAA;AAEH,IAAA;AACb,IAAA;AACG,MAAA;AACO,MAAA;AACrC,IAAA;AACa,IAAA;AACX,MAAA;AACF,IAAA;AAEuC,IAAA;AACX,IAAA;AACE,MAAA;AACM,MAAA;AACpC,IAAA;AAEoC,IAAA;AACX,IAAA;AACgB,MAAA;AACR,MAAA;AACjC,IAAA;AAEuC,IAAA;AACH,MAAA;AAChC,QAAA;AACF,MAAA;AAEqC,MAAA;AACnC,QAAA;AACF,MAAA;AAEiB,MAAA;AACI,MAAA;AACN,QAAA;AACM,MAAA;AACN,QAAA;AACf,MAAA;AAEsC,MAAA;AAGvB,MAAA;AAEH,MAAA;AACiB,QAAA;AACb,UAAA;AACZ,UAAA;AACa,UAAA;AACM,UAAA;AACC,UAAA;AACrB,QAAA;AACH,MAAA;AACF,IAAA;AACF,EAAA;AAE6C,EAAA;AACb,EAAA;AAEzB,EAAA;AACc,IAAA;AACnB,IAAA;AACO,IAAA;AACiB,MAAA;AACtB,MAAA;AAC6B,MAAA;AAC/B,IAAA;AACF,EAAA;AACF;AAKgE;AACvB,EAAA;AACzC;ADpO+C;AACA;AEnVV;AAC9B,EAAA;AACI,EAAA;AACa,EAAA;AACV,EAAA;AACE,IAAA;AACG,MAAA;AACN,MAAA;AACiB,QAAA;AACV,QAAA;AACE,UAAA;AACG,YAAA;AACJ,YAAA;AACA,YAAA;AACH,YAAA;AACR,UAAA;AACO,UAAA;AACyB,YAAA;AACjB,YAAA;AACf,UAAA;AACM,UAAA;AACS,YAAA;AACP,YAAA;AACR,UAAA;AACc,UAAA;AACkB,YAAA;AACjB,YAAA;AACf,UAAA;AACc,UAAA;AACkB,YAAA;AACjB,YAAA;AACf,UAAA;AACY,UAAA;AACG,YAAA;AACP,YAAA;AACR,UAAA;AACM,UAAA;AACS,YAAA;AACP,YAAA;AACR,UAAA;AACO,UAAA;AACyB,YAAA;AACjB,YAAA;AACf,UAAA;AACF,QAAA;AACoC,QAAA;AAC9B,QAAA;AACR,MAAA;AACM,MAAA;AACR,IAAA;AACF,EAAA;AACuB,EAAA;AAChB,EAAA;AACD,EAAA;AACR;AFqV+C;AACA;AGvanB;AAyC4B;AACd,EAAA;AACI,EAAA;AACA,EAAA;AACrC,EAAA;AACT;AAIE;AAEqC,EAAA;AACJ,EAAA;AACF,EAAA;AAEU,EAAA;AAIG,IAAA;AACvB,IAAA;AAEZ,IAAA;AACL,MAAA;AACiB,QAAA;AACE,QAAA;AACF,QAAA;AACe,QAAA;AACA,QAAA;AAC9B,QAAA;AACQ,QAAA;AACC,QAAA;AACX,MAAA;AACF,IAAA;AACD,EAAA;AAEM,EAAA;AACL,IAAA;AACiB,IAAA;AACT,IAAA;AACN,MAAA;AAC4B,MAAA;AACY,MAAA;AACtB,MAAA;AAClB,MAAA;AACA,MAAA;AACF,IAAA;AACF,EAAA;AACF;AHuX+C;AACA;AIjdnB;AAwDD;AAGmB;AAChB,EAAA;AAEF,EAAA;AACO,EAAA;AAEI,EAAA;AACC,EAAA;AACO,IAAA;AACH,EAAA;AACF,EAAA;AACF,EAAA;AACG,IAAA;AAEI,EAAA;AAC/B,IAAA;AAES,EAAA;AACQ,IAAA;AACA,IAAA;AACnB,EAAA;AACkC,IAAA;AACtB,MAAA;AACxB,EAAA;AAEO,EAAA;AACT;AAOyD;AACvB,EAAA;AACK,EAAA;AACb,IAAA;AACkB,IAAA;AAC1C,EAAA;AACO,EAAA;AACT;AAS8B;AACR,EAAA;AAEA,EAAA;AACc,IAAA;AAClC,EAAA;AACoC,EAAA;AAC3B,IAAA;AACM,MAAA;AACF,MAAA;AACE,MAAA;AACJ,MAAA;AACT,IAAA;AACF,EAAA;AAE6C,EAAA;AACL,EAAA;AACD,EAAA;AAEH,EAAA;AACC,IAAA;AACV,IAAA;AACe,MAAA;AACtC,MAAA;AACF,IAAA;AAEkB,IAAA;AAEL,IAAA;AACsB,MAAA;AAGrB,MAAA;AACI,QAAA;AACc,UAAA;AACrB,UAAA;AACG,UAAA;AACT,QAAA;AACD,QAAA;AACF,MAAA;AAIsC,MAAA;AAEb,MAAA;AACR,MAAA;AACD,QAAA;AACI,UAAA;AACX,UAAA;AACG,UAAA;AACT,QAAA;AACD,QAAA;AACF,MAAA;AACF,IAAA;AAEuB,IAAA;AACzB,EAAA;AAEyC,EAAA;AAC3C;AAMwB;AACkB,EAAA;AAC1C;AJqX+C;AACA;AK7ea;AACnD,EAAA;AACL,IAAA;AACA,IAAA;AACA,IAAA;AACoC,IAAA;AACpC,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACA,IAAA;AACS,EAAA;AACb;AAGyC;AACxB,EAAA;AACkB,EAAA;AACnC;AAY0B;AACQ,EAAA;AAIW,EAAA;AAEhB,EAAA;AAGV,EAAA;AAC2B,IAAA;AACxC,IAAA;AAC+B,IAAA;AAE3B,IAAA;AACmB,MAAA;AACwB,QAAA;AACjC,QAAA;AACR,MAAA;AAED,IAAA;AACQ,IAAA;AACd,EAAA;AAGQ,EAAA;AACc,IAAA;AACZ,IAAA;AACS,MAAA;AACwB,MAAA;AAC1C,IAAA;AACQ,IAAA;AAC6B,IAAA;AACtC,EAAA;AAE4B,EAAA;AACa,EAAA;AAIL,EAAA;AACT,IAAA;AAC9B,EAAA;AAEkB,EAAA;AAC0B,IAAA;AAC5C,EAAA;AAEkC,EAAA;AAGM,EAAA;AACZ,EAAA;AAExB,EAAA;AACyC,IAAA;AACrC,EAAA;AACU,IAAA;AAA6E;AAC/F,EAAA;AACF;ALgd+C;AACA;AACA;AACA;AACA;AACA;AACA;AACA;AACA;AACA;AACA;AACA","file":"/home/runner/work/fhir-brasil/fhir-brasil/packages/ocr-utils/dist/index.cjs","sourcesContent":[null,"/**\n * OCR Anchor — Biomarker text anchoring\n *\n * Scans OCR text for biomarker names BEFORE sending to LLM.\n * This prevents hallucination by constraining what biomarkers\n * the LLM is allowed to extract.\n *\n * Matching is deliberately conservative: a name only anchors when it appears\n * as a whole token, is not swallowed by a longer biomarker name, is not inside\n * a genetic report line, and — for generic single-word names — sits on a line\n * that actually carries a value.\n */\n\nimport {\n type BiomarkerSearchPattern,\n generateFilteredLLMReference,\n getAllSearchPatterns,\n UNIT_TO_UCUM,\n} from '@precisa-saude/fhir';\n\nexport interface AnchorMatch {\n code: string;\n confidence: number;\n loinc?: string;\n matchedName: string;\n position: number;\n}\n\nexport interface AnchorResult {\n filteredReference: string;\n matches: AnchorMatch[];\n stats: {\n totalPatterns: number;\n matchedCount: number;\n scanTimeMs: number;\n };\n}\n\n/**\n * Confidence assigned to a specific biomarker name found on a line that also\n * carries a value (a number, a unit, or an expected qualitative term).\n */\nexport const CONFIDENCE_VALUE_ADJACENT = 1.0;\n\n/**\n * Confidence assigned to a specific biomarker name with no value evidence\n * nearby — a section heading, or a mention in prose.\n */\nexport const CONFIDENCE_NAME_ONLY = 0.7;\n\n/**\n * Confidence assigned to a generic/ambiguous name (`Color`, `Protein`,\n * `Blood`, …) that only anchored because a value was found next to it.\n */\nexport const CONFIDENCE_AMBIGUOUS = 0.4;\n\n/** Cap on how many occurrences of the same name are inspected per document. */\nconst MAX_OCCURRENCES_PER_NAME = 5;\n\n/**\n * Normalize text for comparison:\n * - Removes diacritics (ã→a, ç→c, é→e)\n * - Converts to lowercase\n * - Treats a hyphen that joins words as a space\n * - Collapses horizontal whitespace, but KEEPS line breaks — the line is the\n * context window used to decide whether a match is a real biomarker mention\n *\n * O hífen entre palavras vira espaço porque o catálogo e o laboratório\n * discordam sobre ele o tempo todo: o catálogo escreve \"Proteína C-Reativa\" e\n * \"High-Density Lipoprotein\", e os laudos imprimem \"Proteína C Reativa\" e\n * \"High Density Lipoprotein\". Sem essa equivalência, 82 dos 159 nomes com\n * hífen deixam de ancorar na grafia que o documento usa.\n *\n * Não era teórico: um GGT de verdade foi descartado como alucinação em 27\n * laudos porque o documento escrevia \"Gama glutamil transferase\" e o catálogo\n * \"Gama-Glutamil Transferase\". Um caractere derrubava o valor antes de\n * qualquer validação.\n *\n * A troca exige **letra antes** do hífen, e por isso não toca em número:\n * o `-2.5` de um T-score e o `0-5` de uma faixa de urina seguem intactos.\n * Trocar sem essa guarda apagaria o sinal de um valor negativo, que é bem\n * pior que o problema original.\n *\n * O pré-filtro de substring do `collectCandidates` compara a chave com o texto\n * já normalizado, então a equivalência precisa nascer aqui: aplicada só na\n * regex, o `includes` descartaria o nome antes de ela rodar.\n */\nfunction normalize(text: string): string {\n return text\n .normalize('NFD')\n .replace(/[\\u0300-\\u036f]/g, '')\n .toLowerCase()\n .replace(/(?<=\\p{L})-(?=[\\p{L}\\p{N}])/gu, ' ')\n .replace(/[^\\S\\n]+/g, ' ');\n}\n\nconst UNAMBIGUOUS_SHORT_NAMES = new Set([\n 'hdl',\n 'ldl',\n 'lh',\n 'tsh',\n 'crp',\n 'pcr',\n 'ggt',\n 'alt',\n 'ast',\n 'bun',\n 'wbc',\n 'rbc',\n 'mcv',\n 'mch',\n 'rdw',\n 'mpv',\n 'psa',\n 'fsh',\n 'hba1c',\n 'egfr',\n 'acr',\n 'esr',\n 'vhs',\n 'bmc',\n 'bmd',\n 'vat',\n 'dxa',\n 'dmo',\n 'cmo',\n 'ffm',\n 'lbm',\n 'mlg',\n 'tav',\n]);\n\n/**\n * Single-word catalog names that are ordinary words in EN/PT, so seeing them\n * proves nothing on its own. They only anchor when the line also carries a\n * value. Qualitative urine markers (`Color`, `Protein`, `Blood`, …) are\n * detected automatically — see `isQualitativeUrine` — and don't belong here.\n */\nconst CONTEXT_REQUIRED_NAMES = new Set([\n 'bacteria', // Bacteria_Urine — tem unidade, escapa da regra automática\n 'bacterias', // Bacteria_Urine\n 'lead', // Lead — verbo/substantivo comuníssimo em inglês\n 'peso', // TotalMass\n 'saturation', // TransferrinSaturation — \"oxygen saturation\", \"saturation index\"\n 'tap', // ProthrombinTime — \"tap\" em inglês\n 'volume', // VATVolume\n 'weight', // TotalMass\n // Sítios de dobra pelo nome nu. São partes do corpo antes de serem medidas,\n // e aparecem em prosa: num laudo de DEXA real, \"hips and thighs\" e\n // \"abdominal region\" ancoravam dobra cutânea que o documento não tem.\n // Exigir valor na linha separa a tabela do parágrafo.\n 'abdominal',\n 'chest',\n 'coxa',\n 'peitoral',\n 'subescapular',\n 'subscapular',\n 'suprailiac',\n 'thigh',\n 'triceps',\n 'tricipital',\n]);\n\n/**\n * Qualitative results expected next to a non-numeric biomarker\n * (urine dipstick, sediment, appearance). Normalized, single tokens —\n * \"não reagente\" is covered by `reagente`, \"não detectado\" by `detectado`.\n */\nconst QUALITATIVE_VALUE_TERMS = new Set([\n 'absent',\n 'alguns',\n 'amarela',\n 'amarelo',\n 'anormal',\n 'ausencia',\n 'ausente',\n 'ausentes',\n 'citrino',\n 'claro',\n 'clear',\n 'cloudy',\n 'colorless',\n 'detectado',\n 'detected',\n 'escuro',\n 'incolor',\n 'indetectavel',\n 'limpido',\n 'moderada',\n 'moderado',\n 'negativa',\n 'negative',\n 'negativo',\n 'normais',\n 'normal',\n 'numerosos',\n 'ocasional',\n 'positiva',\n 'positive',\n 'positivo',\n 'present',\n 'presente',\n 'presentes',\n 'raras',\n 'raro',\n 'raros',\n 'reagente',\n 'trace',\n 'traces',\n 'tracos',\n 'turvo',\n 'undetectable',\n 'yellow',\n]);\n\n/**\n * Signals that a line comes from a genetic/molecular report rather than from a\n * panel of measured values. Gene symbols collide with biomarker names (`APOB`\n * the gene vs. `ApoB` the lipoprotein), so the context — not a static HGNC\n * blocklist — is what tells them apart. Blocking the token itself would break\n * real lipid panels.\n */\nconst GENETIC_CONTEXT_PATTERNS: RegExp[] = [\n /\\b[nx][mrpc]_\\d{6,}/, // RefSeq: NM_000384.2, NP_, NR_, XM_\n /\\bens[gtp]\\d{6,}/, // Ensembl: ENSG00000084674\n /\\bp\\.[a-z]{3}\\d/, // HGVS proteína: p.Trp448*\n /\\bc\\.\\d+[acgt]?[>_+-]/, // HGVS codificante: c.1234A>G, c.76_78del\n /\\brs\\d{4,}\\b/, // dbSNP\n /\\bgenes?\\b/,\n /\\bvariante?s?\\b/,\n /\\bexons?\\b/,\n /\\bzygosity\\b/,\n /\\bzigosidade\\b/,\n /\\balleles?\\b/,\n /\\balelos?\\b/,\n /\\bmutations?\\b/,\n /\\bmutac(ao|oes)\\b/,\n /\\bpathogenic/,\n /\\bpatogenic/,\n /\\bheterozyg/,\n /\\bhomozyg/,\n /\\bheterozigot/,\n /\\bhomozigot/,\n /\\bsequence change\\b/,\n];\n\n/**\n * Sítios de dobra cutânea cujo nome nu também nomeia uma circunferência:\n * \"Coxa\" aparece tanto em \"Dobra Cutânea Coxa\" quanto em \"Circunferência da\n * Coxa\". O termo nu precisa existir como alias, porque há laudo que imprime\n * só o sítio na coluna, então a desambiguação tem que vir do contexto da\n * linha, como já se faz com laudo genético.\n */\nconst SKINFOLD_SITE_CODES = new Set([\n 'SkinfoldAbdominal',\n 'SkinfoldChest',\n 'SkinfoldMidaxillary',\n 'SkinfoldSubscapular',\n 'SkinfoldSuprailiac',\n 'SkinfoldThigh',\n 'SkinfoldTriceps',\n]);\n\n/** Uma linha de circunferência ou perímetro não mede dobra. */\nconst GIRTH_CONTEXT_PATTERNS: RegExp[] = [\n /\\bcircumference\\b/,\n /\\bcircunferencias?\\b/,\n /\\bperimetros?\\b/,\n /\\bgirth\\b/,\n];\n\n/**\n * Só bloqueia quando a linha fala de circunferência e não fala de dobra:\n * \"Dobra Cutânea Coxa\" e \"Thigh Skinfold\" continuam ancorando normalmente,\n * e uma linha que traga as duas palavras é ambígua demais para descartar.\n */\nconst SKINFOLD_CONTEXT_PATTERNS: RegExp[] = [/\\bdobras?\\b/, /\\bskin ?folds?\\b/, /\\bpregas?\\b/];\n\n/**\n * Medida em centímetros numa linha de sítio corporal.\n *\n * Dobra cutânea é em milímetros, sempre: um valor em cm no mesmo sítio é\n * circunferência. É o desambiguador mais forte que existe aqui, porque não\n * depende de a folha escrever a palavra \"circunferência\", e num laudo de\n * antropometria a coluna costuma trazer só o sítio e o número.\n *\n * Rejeita cm em vez de exigir mm: há folha que imprime a unidade no cabeçalho\n * da coluna e não em cada linha, e exigir mm perderia essas.\n */\nconst CENTIMETRE_VALUE = /\\d\\s*(?:,\\d+\\s*)?cm\\b/;\n\nfunction hasGirthContext(line: string): boolean {\n if (SKINFOLD_CONTEXT_PATTERNS.some((re) => re.test(line))) {\n return false;\n }\n return GIRTH_CONTEXT_PATTERNS.some((re) => re.test(line)) || CENTIMETRE_VALUE.test(line);\n}\n\nconst DIGIT_PATTERN = /\\d/;\n\n/** Unit tokens reused from the core catalog instead of a parallel list. */\nlet cachedUnitTokens: Set<string> | null = null;\n\nfunction getUnitTokens(): Set<string> {\n if (!cachedUnitTokens) {\n cachedUnitTokens = new Set(\n Object.keys(UNIT_TO_UCUM)\n .map((unit) => normalize(unit).trim())\n .filter(Boolean),\n );\n }\n return cachedUnitTokens;\n}\n\ninterface PatternEntry {\n ambiguous: boolean;\n code: string;\n loinc?: string;\n original: string;\n}\n\ninterface NamePattern {\n entries: PatternEntry[];\n /** Built on first use — most names never match a given document. */\n regex: RegExp | null;\n}\n\ninterface Candidate {\n end: number;\n entries: PatternEntry[];\n start: number;\n}\n\nlet cachedPatterns: BiomarkerSearchPattern[] | null = null;\nlet cachedNamePatterns: Map<string, NamePattern> | null = null;\n\nfunction getPatterns(): BiomarkerSearchPattern[] {\n if (!cachedPatterns) {\n cachedPatterns = getAllSearchPatterns();\n }\n return cachedPatterns;\n}\n\nfunction escapeRegExp(text: string): string {\n return text.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n}\n\n/**\n * Build a whole-token matcher for a normalized name.\n *\n * Lookarounds instead of `\\b` because names may start or end with a non-word\n * character (`Lp(a)`), where `\\b` asserts the wrong thing.\n *\n * A multi-word name must sit on a single line: in the column layouts labs\n * print, consecutive lines are separate biomarkers, and allowing a line break\n * inside a name turns \"Colesterol\\nHDL\" into the name \"Colesterol HDL\".\n * A wrapped name still anchors through its head token when that token is a\n * name of its own (\"Colesterol\\nTotal\" → `Cholesterol`).\n *\n * The trailing optional `s` keeps the plurals labs actually print\n * (\"Proteínas\", \"Cetonas\") anchored to the singular catalog name — without\n * letting `proteína` match inside `proteinúria`.\n */\nfunction buildNamePattern(normalizedName: string): RegExp {\n const body = normalizedName.split(' ').map(escapeRegExp).join('[^\\\\S\\\\n]+');\n const plural = /\\p{L}$/u.test(normalizedName) ? 's?' : '';\n return new RegExp(`(?<![\\\\p{L}\\\\p{N}])${body}${plural}(?![\\\\p{L}\\\\p{N}])`, 'gu');\n}\n\nfunction isQualitativeUrine(pattern: BiomarkerSearchPattern): boolean {\n const categories = Array.isArray(pattern.category) ? pattern.category : [pattern.category];\n return categories.includes('urina') && !pattern.unit;\n}\n\n/**\n * A name is ambiguous when it is a single token that also reads as ordinary\n * text. Multi-word names (`Occult Blood`, `Urine Protein`) are specific enough\n * on their own.\n */\nfunction isAmbiguousName(normalizedName: string, pattern: BiomarkerSearchPattern): boolean {\n if (normalizedName.includes(' ')) {\n return false;\n }\n return CONTEXT_REQUIRED_NAMES.has(normalizedName) || isQualitativeUrine(pattern);\n}\n\nfunction getNamePatterns(): Map<string, NamePattern> {\n if (!cachedNamePatterns) {\n const map = new Map<string, NamePattern>();\n for (const pattern of getPatterns()) {\n for (const name of pattern.names) {\n const normalized = normalize(name).trim();\n if (!normalized) {\n continue;\n }\n if (normalized.length < 3 && !UNAMBIGUOUS_SHORT_NAMES.has(normalized)) {\n continue;\n }\n let slot = map.get(normalized);\n if (!slot) {\n slot = { entries: [], regex: null };\n map.set(normalized, slot);\n }\n slot.entries.push({\n ambiguous: isAmbiguousName(normalized, pattern),\n code: pattern.code,\n ...(pattern.loinc && { loinc: pattern.loinc }),\n original: name,\n });\n }\n }\n cachedNamePatterns = map;\n }\n return cachedNamePatterns;\n}\n\nfunction getLineBounds(text: string, position: number): { end: number; start: number } {\n const start = text.lastIndexOf('\\n', position) + 1;\n const nextBreak = text.indexOf('\\n', position);\n return { end: nextBreak === -1 ? text.length : nextBreak, start };\n}\n\nfunction hasGeneticContext(line: string): boolean {\n return GENETIC_CONTEXT_PATTERNS.some((pattern) => pattern.test(line));\n}\n\n/**\n * Does this line carry something that looks like a measured result?\n * A digit, a known unit, or an expected qualitative term.\n */\nfunction hasValueEvidence(line: string): boolean {\n if (DIGIT_PATTERN.test(line)) {\n return true;\n }\n const unitTokens = getUnitTokens();\n for (const token of line.split(/[^\\p{L}\\p{N}%/]+/u)) {\n if (token && (unitTokens.has(token) || QUALITATIVE_VALUE_TERMS.has(token))) {\n return true;\n }\n }\n return false;\n}\n\n/**\n * Cheap pre-filter before the (much costlier) boundary regex.\n *\n * Sound because `normalize` collapses horizontal whitespace to a single space\n * and a name never spans a line break: whenever the pattern can match, the\n * literal name is a substring of the text.\n */\nfunction collectCandidates(normalizedText: string): Candidate[] {\n const candidates: Candidate[] = [];\n for (const [name, slot] of getNamePatterns()) {\n if (!normalizedText.includes(name)) {\n continue;\n }\n const { entries } = slot;\n const regex = (slot.regex ??= buildNamePattern(name));\n regex.lastIndex = 0;\n let occurrences = 0;\n let match = regex.exec(normalizedText);\n while (match !== null && occurrences < MAX_OCCURRENCES_PER_NAME) {\n candidates.push({ end: match.index + match[0].length, entries, start: match.index });\n occurrences += 1;\n match = regex.exec(normalizedText);\n }\n }\n return candidates;\n}\n\n/**\n * Longest match wins: drop a match fully contained in a longer one, so\n * `Cholesterol` doesn't anchor inside `HDL Cholesterol` and `Blood` doesn't\n * anchor inside `Blood Glucose`.\n *\n * Strictly longer, not longer-or-equal: containment plus equal length means an\n * identical span, which only happens when two distinct catalog names match the\n * same text (a singular and its plural form, say). Dropping one of those by\n * catalog order would silently lose a code, and losing an anchor is worse than\n * keeping both — `findBiomarkersInText` dedups per code anyway.\n */\nfunction resolveOverlaps(candidates: Candidate[]): Candidate[] {\n const sorted = [...candidates].sort(\n (a, b) => b.end - b.start - (a.end - a.start) || a.start - b.start,\n );\n const accepted: Candidate[] = [];\n for (const candidate of sorted) {\n const length = candidate.end - candidate.start;\n const swallowed = accepted.some(\n (other) =>\n other.start <= candidate.start &&\n candidate.end <= other.end &&\n other.end - other.start > length,\n );\n if (!swallowed) {\n accepted.push(candidate);\n }\n }\n return accepted;\n}\n\n/**\n * Find all biomarker names present in OCR text.\n *\n * Matching is whole-token, longest-match-wins, and context-aware: matches\n * inside genetic report lines are discarded, and generic names only anchor\n * when a value sits on the same line. Returns one match per biomarker code —\n * the highest-confidence occurrence.\n */\nexport function findBiomarkersInText(ocrText: string): AnchorResult {\n const startTime = Date.now();\n const normalizedText = normalize(ocrText);\n const bestByCode = new Map<string, AnchorMatch>();\n const geneticLines = new Map<number, boolean>();\n const valueLines = new Map<number, boolean>();\n const girthLines = new Map<number, boolean>();\n\n for (const candidate of resolveOverlaps(collectCandidates(normalizedText))) {\n const { end: lineEnd, start: lineStart } = getLineBounds(normalizedText, candidate.start);\n\n let genetic = geneticLines.get(lineStart);\n if (genetic === undefined) {\n genetic = hasGeneticContext(normalizedText.slice(lineStart, lineEnd));\n geneticLines.set(lineStart, genetic);\n }\n if (genetic) {\n continue;\n }\n\n let hasValue = valueLines.get(lineStart);\n if (hasValue === undefined) {\n hasValue = hasValueEvidence(normalizedText.slice(lineStart, lineEnd));\n valueLines.set(lineStart, hasValue);\n }\n\n let girth = girthLines.get(lineStart);\n if (girth === undefined) {\n girth = hasGirthContext(normalizedText.slice(lineStart, lineEnd));\n girthLines.set(lineStart, girth);\n }\n\n for (const entry of candidate.entries) {\n if (entry.ambiguous && !hasValue) {\n continue;\n }\n\n if (girth && SKINFOLD_SITE_CODES.has(entry.code)) {\n continue;\n }\n\n let confidence = CONFIDENCE_NAME_ONLY;\n if (entry.ambiguous) {\n confidence = CONFIDENCE_AMBIGUOUS;\n } else if (hasValue) {\n confidence = CONFIDENCE_VALUE_ADJACENT;\n }\n\n const existing = bestByCode.get(entry.code);\n const better =\n !existing ||\n confidence > existing.confidence ||\n (confidence === existing.confidence && candidate.start < existing.position);\n if (better) {\n bestByCode.set(entry.code, {\n code: entry.code,\n confidence,\n loinc: entry.loinc,\n matchedName: entry.original,\n position: candidate.start,\n });\n }\n }\n }\n\n const matches = Array.from(bestByCode.values()).sort((a, b) => a.position - b.position);\n const scanTimeMs = Date.now() - startTime;\n\n return {\n filteredReference: generateFilteredLLMReference(matches.map((m) => m.code)),\n matches,\n stats: {\n matchedCount: matches.length,\n scanTimeMs,\n totalPatterns: getPatterns().length,\n },\n };\n}\n\n/**\n * Get the list of matched biomarker codes from an anchor result.\n */\nexport function getMatchedCodes(result: AnchorResult): string[] {\n return result.matches.map((m) => m.code);\n}\n","/**\n * Contrato de saída para extração de laudo por modelo.\n *\n * Este é o **contrato de interoperabilidade**, e não um prompt. Ele descreve a\n * forma do JSON que qualquer modelo precisa devolver para o resto do toolkit\n * conseguir conferir e converter o resultado. Não diz como pedir isso ao\n * modelo, não traz instrução de comportamento e não depende de fornecedor:\n * quem usa liga do jeito que a plataforma dele permitir (saída estruturada,\n * gramática, tool use ou simples prompt com validação por cima).\n *\n * As descrições são deliberadamente neutras. Regra de comportamento (\"nunca\n * infira\", \"copie literalmente\") é ajuste de prompt, muda de modelo para\n * modelo e não pertence a um contrato público.\n *\n * As descrições são as únicas strings em inglês do pacote, e isso é\n * deliberado: o schema é contrato de integração lido por quem consome de fora\n * do Brasil, e uma descrição em pt-BR não ajuda ninguém em Colônia ou Madri.\n * O resto da documentação segue a regra do ecossistema.\n *\n * O campo `sourceText` existe porque é o que torna a conferência possível:\n * sem o trecho que originou o valor não dá para auditar a extração depois.\n *\n * Campo que aceita mais de um tipo usa `anyOf`, e não `type: [...]`. As duas\n * formas são JSON Schema válido, mas decodificador restrito não engole a\n * segunda: o LM Studio recusa a geração com `'type' must be a string`. Como o\n * ponto do contrato é servir a qualquer modelo, vale a forma mais aceita.\n */\nexport const LAB_EXTRACTION_SCHEMA = {\n $id: 'https://fhir-brasil.dev.br/schemas/lab-extraction.json',\n $schema: 'https://json-schema.org/draft/2020-12/schema',\n additionalProperties: false,\n properties: {\n biomarkers: {\n description: 'The measurements read from the report.',\n items: {\n additionalProperties: false,\n properties: {\n confidence: {\n description: 'Confidence in this reading, from 0 to 1.',\n maximum: 1,\n minimum: 0,\n type: 'number',\n },\n loinc: {\n anyOf: [{ type: 'string' }, { type: 'null' }],\n description: 'A LOINC code from the allowed list, or null when none of them applies.',\n },\n name: {\n description: 'The measurement name as the report prints it.',\n type: 'string',\n },\n referenceMax: {\n anyOf: [{ type: 'number' }, { type: 'null' }],\n description: 'Upper bound of the range printed on the report, or null.',\n },\n referenceMin: {\n anyOf: [{ type: 'number' }, { type: 'null' }],\n description: 'Lower bound of the range printed on the report, or null.',\n },\n sourceText: {\n description: 'The snippet of the report carrying this measurement and its value.',\n type: 'string',\n },\n unit: {\n description: 'Unit as the report prints it. Empty string when there is none.',\n type: 'string',\n },\n value: {\n anyOf: [{ type: 'number' }, { type: 'string' }],\n description: 'Numeric value, or text for a qualitative result.',\n },\n },\n required: ['name', 'value', 'unit', 'sourceText', 'confidence'],\n type: 'object',\n },\n type: 'array',\n },\n },\n required: ['biomarkers'],\n title: 'Laboratory report extraction',\n type: 'object',\n} as const;\n\n/** Uma grandeza como o modelo devolve, antes de qualquer conferência. */\nexport interface ExtractedBiomarker {\n confidence: number;\n loinc?: string | null;\n name: string;\n referenceMax?: number | null;\n referenceMin?: number | null;\n sourceText: string;\n unit: string;\n value: number | string;\n}\n\n/** O objeto inteiro que o modelo devolve. */\nexport interface ExtractionPayload {\n biomarkers: ExtractedBiomarker[];\n}\n","import { loincToCode } from '@precisa-saude/fhir';\n\nimport type { ExtractedBiomarker } from './extraction-schema.js';\n\n/**\n * Converte grandezas já conferidas no envelope que o `fhir-bio convert` come.\n *\n * O laudo e o paciente não vêm do modelo: o contrato de extração cobre só as\n * grandezas. Os dois saem daqui com valores sintéticos e óbvios, na mesma\n * linha do `fhir-rnds-sandbox`, para a demo rodar de ponta a ponta sem inventar\n * identidade de ninguém. Quem integra de verdade troca os dois pelo que já tem.\n */\nexport interface LabResultEnvelope {\n observations: {\n biomarkerCode: string;\n biomarkerName: string;\n flag: 'H' | 'L' | '';\n referenceMax?: number;\n referenceMin?: number;\n reportId: string;\n unit: string;\n value: number | string;\n }[];\n profile: { name: string; userId: string };\n report: {\n collectionDate: string;\n createdAt: string;\n overallStatus: 'ANORMAL' | 'NORMAL';\n processingStatus: 'complete';\n reportId: string;\n userId: string;\n };\n}\n\nexport interface ToLabResultOptions {\n collectionDate?: string;\n reportId?: string;\n userId?: string;\n}\n\n/** `H`/`L` só quando o próprio laudo trouxe a faixa. Nunca inferida daqui. */\nfunction flagFor(b: ExtractedBiomarker): 'H' | 'L' | '' {\n if (typeof b.value !== 'number') return '';\n if (typeof b.referenceMax === 'number' && b.value > b.referenceMax) return 'H';\n if (typeof b.referenceMin === 'number' && b.value < b.referenceMin) return 'L';\n return '';\n}\n\nexport function extractionToLabResult(\n biomarkers: ExtractedBiomarker[],\n options: ToLabResultOptions = {},\n): LabResultEnvelope {\n const reportId = options.reportId ?? 'laudo-demo';\n const userId = options.userId ?? 'paciente-demo';\n const collectionDate = options.collectionDate ?? new Date().toISOString().slice(0, 10);\n\n const observations = biomarkers.flatMap((b) => {\n // Sem código interno não há como converter, e o LOINC sozinho não basta\n // para o `convert`. Cai fora em silêncio porque a checagem de ancoragem já\n // rodou antes: o que chega aqui sem código é grandeza fora do catálogo.\n const code = b.loinc ? loincToCode(b.loinc) : undefined;\n if (!code) return [];\n\n return [\n {\n biomarkerCode: code,\n biomarkerName: b.name,\n flag: flagFor(b),\n ...(typeof b.referenceMax === 'number' ? { referenceMax: b.referenceMax } : {}),\n ...(typeof b.referenceMin === 'number' ? { referenceMin: b.referenceMin } : {}),\n reportId,\n unit: b.unit,\n value: b.value,\n },\n ];\n });\n\n return {\n observations,\n profile: { name: 'Paciente de Demonstração', userId },\n report: {\n collectionDate,\n createdAt: `${collectionDate}T00:00:00Z`,\n overallStatus: observations.some((o) => o.flag !== '') ? 'ANORMAL' : 'NORMAL',\n processingStatus: 'complete',\n reportId,\n userId,\n },\n };\n}\n","import { loincToCode } from '@precisa-saude/fhir';\n\nimport type { AnchorResult } from './anchor.js';\nimport type { ExtractedBiomarker, ExtractionPayload } from './extraction-schema.js';\n\n/**\n * Conferência da saída do modelo contra o contrato e contra a ancoragem.\n *\n * São duas checagens, e as duas são determinísticas:\n *\n * 1. **Forma.** O objeto bate com `LAB_EXTRACTION_SCHEMA`. Modelo que devolve\n * texto solto, campo faltando ou tipo errado é recusado aqui, o que deixa\n * a qualidade do modelo virar problema de cobertura e nunca de correção.\n * 2. **Ancoragem.** O código veio da lista que a varredura liberou. Código que\n * o laudo não mencionou é descartado, que é a falha cara: um valor\n * plausível pendurado num exame que não estava na página.\n *\n * A validação de citação, a correção de código contra nome impresso e a\n * política de confiança não moram aqui.\n *\n * Sem dependência de runtime além do `@precisa-saude/fhir`: a checagem de\n * forma é escrita à mão porque o schema é pequeno e o pacote não carrega\n * validador de JSON Schema.\n */\n\n/** Por que uma grandeza foi recusada. */\nexport type RejectionReason = 'not-anchored' | 'schema';\n\nexport interface RejectedBiomarker {\n /** Mensagem legível, já em pt-BR, dizendo o que falhou. */\n detail: string;\n /** O que o modelo devolveu, sem alteração, para o consumidor poder logar. */\n raw: unknown;\n reason: RejectionReason;\n}\n\nexport interface ExtractionValidationResult {\n accepted: ExtractedBiomarker[];\n /** Erros do objeto inteiro, quando nem dá para chegar nas grandezas. */\n errors: string[];\n rejected: RejectedBiomarker[];\n /** `true` quando o objeto tem forma válida, mesmo que toda grandeza caia. */\n valid: boolean;\n}\n\nexport interface ValidateExtractionOptions {\n /**\n * Resultado da ancoragem sobre o mesmo texto que foi ao modelo. Sem ele a\n * checagem de ancoragem não roda e só a forma é conferida, que é um modo\n * deliberadamente mais fraco: serve para inspecionar saída de modelo sem o\n * laudo em mãos.\n */\n anchors?: AnchorResult;\n}\n\nconst isRecord = (v: unknown): v is Record<string, unknown> =>\n typeof v === 'object' && v !== null && !Array.isArray(v);\n\n/** Confere uma grandeza contra o schema. Devolve a lista de problemas. */\nfunction schemaErrors(raw: unknown): string[] {\n if (!isRecord(raw)) return ['não é um objeto'];\n\n const errors: string[] = [];\n const { confidence, loinc, name, referenceMax, referenceMin, sourceText, unit, value } = raw;\n\n if (typeof name !== 'string' || name.length === 0) errors.push('`name` ausente ou vazio');\n if (typeof sourceText !== 'string' || sourceText.length === 0)\n errors.push('`sourceText` ausente ou vazio');\n if (typeof unit !== 'string') errors.push('`unit` ausente');\n if (typeof value !== 'number' && typeof value !== 'string') errors.push('`value` ausente');\n if (typeof confidence !== 'number' || confidence < 0 || confidence > 1)\n errors.push('`confidence` fora de 0..1');\n\n if (loinc !== undefined && loinc !== null && typeof loinc !== 'string')\n errors.push('`loinc` não é string nem null');\n\n for (const [key, v] of [\n ['referenceMax', referenceMax],\n ['referenceMin', referenceMin],\n ] as const) {\n if (v !== undefined && v !== null && typeof v !== 'number')\n errors.push(`\\`${key}\\` não é número nem null`);\n }\n\n return errors;\n}\n\n/**\n * O conjunto de códigos que a varredura liberou, pelos dois lados: o LOINC e o\n * código interno. O modelo devolve LOINC, mas aceitar o código interno também\n * evita recusar consumidor que prefira trabalhar com ele.\n */\nfunction allowedKeys(anchors: AnchorResult): Set<string> {\n const allowed = new Set<string>();\n for (const match of anchors.matches) {\n allowed.add(match.code);\n if (match.loinc) allowed.add(match.loinc);\n }\n return allowed;\n}\n\n/**\n * Confere a saída de um modelo contra o contrato e, quando a ancoragem é\n * fornecida, contra a lista de códigos que a varredura liberou.\n */\nexport function validateExtraction(\n raw: unknown,\n options: ValidateExtractionOptions = {},\n): ExtractionValidationResult {\n const { anchors } = options;\n\n if (!isRecord(raw)) {\n return { accepted: [], errors: ['a saída não é um objeto JSON'], rejected: [], valid: false };\n }\n if (!Array.isArray(raw.biomarkers)) {\n return {\n accepted: [],\n errors: ['`biomarkers` ausente ou não é lista'],\n rejected: [],\n valid: false,\n };\n }\n\n const allowed = anchors ? allowedKeys(anchors) : undefined;\n const accepted: ExtractedBiomarker[] = [];\n const rejected: RejectedBiomarker[] = [];\n\n for (const entry of raw.biomarkers) {\n const problems = schemaErrors(entry);\n if (problems.length > 0) {\n rejected.push({ detail: problems.join('; '), raw: entry, reason: 'schema' });\n continue;\n }\n\n const biomarker = entry as unknown as ExtractedBiomarker;\n\n if (allowed) {\n const loinc = biomarker.loinc ?? undefined;\n // Sem código não há o que conferir contra a ancoragem, e aceitar assim\n // deixaria passar justamente o caso que a varredura existe para pegar.\n if (!loinc) {\n rejected.push({\n detail: `\"${biomarker.name}\" veio sem código LOINC`,\n raw: entry,\n reason: 'not-anchored',\n });\n continue;\n }\n // O `?? ''` de antes nunca deixava código inválido passar, porque string\n // vazia não entra no conjunto de permitidos, mas obrigava quem lê a\n // provar isso. A forma explícita não precisa de prova.\n const internalCode = loincToCode(loinc);\n const isAnchored =\n allowed.has(loinc) || (internalCode !== undefined && allowed.has(internalCode));\n if (!isAnchored) {\n rejected.push({\n detail: `${loinc} não foi ancorado no texto de origem`,\n raw: entry,\n reason: 'not-anchored',\n });\n continue;\n }\n }\n\n accepted.push(biomarker);\n }\n\n return { accepted, errors: [], rejected, valid: true };\n}\n\n/** Só a lista de grandezas aprovadas, para quem não quer o relatório inteiro. */\nexport function acceptedBiomarkers(\n raw: unknown,\n options: ValidateExtractionOptions = {},\n): ExtractedBiomarker[] {\n return validateExtraction(raw, options).accepted;\n}\n\nexport type { ExtractedBiomarker, ExtractionPayload };\n","import { findBiomarkersInText } from './anchor.js';\nimport { LAB_EXTRACTION_SCHEMA } from './extraction-schema.js';\n\n/**\n * Cliente mínimo para endpoint compatível com OpenAI.\n *\n * Isto é **conveniência, não contrato**. O contrato é o\n * `LAB_EXTRACTION_SCHEMA`, e o toolkit funciona inteiro sem esta função: quem\n * integra chama o próprio modelo do jeito que a plataforma dele permitir e\n * entrega o JSON ao `validateExtraction`. Esta função existe para a demo rodar\n * de uma ponta à outra sem um `curl` no meio.\n *\n * `/v1/chat/completions` é o que praticamente todo mundo fala: LM Studio,\n * Ollama, llama.cpp, vLLM, OpenRouter, OpenAI, e a Anthropic pelo endpoint de\n * compatibilidade. Por isso não há SDK de fornecedor aqui, e por isso o pacote\n * continua sem dependência de runtime: `fetch` é do Node.\n *\n * A chave **nunca** entra por argumento de linha de comando, só por variável de\n * ambiente: argumento fica no histórico do shell e na lista de processos.\n */\nexport interface ExtractOptions {\n apiKey?: string;\n baseUrl: string;\n model: string;\n /**\n * Modo de saída estruturada. O padrão é negociar sozinho.\n *\n * Aqui é onde a compatibilidade quebra de verdade: o LM Studio recusa\n * `json_object` com 400 e só aceita `json_schema`, a OpenAI aceita os dois,\n * e servidor mais simples não conhece o campo. Como nenhum valor serve a\n * todos, a primeira tentativa vai com `json_schema` e, se o servidor recusar,\n * a segunda vai sem nada. Quem quiser fixar um modo passa ele aqui.\n *\n * Vale lembrar que isto mexe em **aproveitamento**, não em correção: saída\n * malformada é recusada pela conferência de qualquer jeito.\n */\n responseFormat?: 'auto' | 'json_object' | 'json_schema' | 'none';\n /** Milissegundos até desistir. Modelo local frio demora para carregar. */\n timeoutMs?: number;\n}\n\nexport interface ExtractResult {\n /** O JSON que o modelo devolveu, ainda sem conferência nenhuma. */\n payload: unknown;\n /** Texto cru da resposta, guardado para quando o parse falha. */\n raw: string;\n tookMs: number;\n}\n\n/**\n * O prompt é deliberadamente curto e neutro.\n *\n * Ele diz o que devolver e nada sobre como ler um laudo. Toda a instrução de\n * comportamento que um extrator de produção carrega é ajuste que muda de modelo\n * para modelo, e não pertence a um pacote público. O que sustenta a qualidade\n * aqui não é o prompt: é a conferência que roda depois.\n */\nfunction buildPrompt(text: string, allowed: string): string {\n return [\n 'Extract the laboratory results from the report below.',\n '',\n 'Return JSON matching this schema, and nothing else:',\n JSON.stringify(LAB_EXTRACTION_SCHEMA),\n '',\n 'Use only LOINC codes from this list:',\n allowed,\n '',\n 'REPORT:',\n text,\n ].join('\\n');\n}\n\n/** Modelo costuma embrulhar o JSON em cerca de markdown. Tira a cerca. */\nfunction stripFence(raw: string): string {\n const fenced = /```(?:json)?\\s*([\\s\\S]*?)```/.exec(raw);\n return (fenced?.[1] ?? raw).trim();\n}\n\n/**\n * Manda o laudo e a lista ancorada ao modelo e devolve o que ele respondeu.\n *\n * Não confere nada: a saída vai para o `validateExtraction`, que é onde a\n * ancoragem é cobrada. Separar os dois é proposital, porque é o que deixa\n * trocar de modelo sem mexer na parte que garante o resultado.\n */\nexport async function extractWithModel(\n text: string,\n options: ExtractOptions,\n): Promise<ExtractResult> {\n const { apiKey, baseUrl, model, responseFormat, timeoutMs = 300_000 } = options;\n\n // A própria ancoragem já monta a lista de permitidos, então o prompt e a\n // conferência bebem exatamente da mesma fonte.\n const allowed = findBiomarkersInText(text).filteredReference;\n\n const startedAt = Date.now();\n\n const formatBody = (mode: 'json_object' | 'json_schema' | 'none'): string =>\n JSON.stringify({\n messages: [{ content: buildPrompt(text, allowed), role: 'user' }],\n model,\n ...(mode === 'json_object' ? { response_format: { type: 'json_object' } } : {}),\n ...(mode === 'json_schema'\n ? {\n response_format: {\n json_schema: { name: 'lab_extraction', schema: LAB_EXTRACTION_SCHEMA, strict: true },\n type: 'json_schema',\n },\n }\n : {}),\n temperature: 0,\n });\n\n const post = async (mode: 'json_object' | 'json_schema' | 'none'): Promise<Response> =>\n fetch(`${baseUrl.replace(/\\/$/, '')}/chat/completions`, {\n body: formatBody(mode),\n headers: {\n 'content-type': 'application/json',\n ...(apiKey ? { authorization: `Bearer ${apiKey}` } : {}),\n },\n method: 'POST',\n signal: AbortSignal.timeout(timeoutMs),\n });\n\n const mode = responseFormat ?? 'auto';\n let response = await post(mode === 'auto' ? 'json_schema' : mode);\n\n // Servidor que não conhece o modo estrito responde 4xx. A segunda tentativa\n // vai sem nada, que é o denominador comum, e só então o erro sobe.\n if (!response.ok && mode === 'auto' && response.status >= 400 && response.status < 500) {\n response = await post('none');\n }\n\n if (!response.ok) {\n throw new Error(`${String(response.status)} de ${baseUrl}: ${await response.text()}`);\n }\n\n const body = (await response.json()) as {\n choices?: { message?: { content?: string } }[];\n };\n const raw = body.choices?.[0]?.message?.content ?? '';\n const tookMs = Date.now() - startedAt;\n\n try {\n return { payload: JSON.parse(stripFence(raw)), raw, tookMs };\n } catch {\n throw new Error(`O modelo não devolveu JSON analisável. Resposta crua:\\n${raw.slice(0, 500)}`);\n }\n}\n"]}
|