@sirdaspdf/core 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agentchunks.d.ts +18 -0
- package/dist/agentchunks.js +85 -0
- package/dist/anonymize.js +55 -7
- package/dist/classify.d.ts +12 -0
- package/dist/classify.js +89 -0
- package/dist/extract.d.ts +16 -0
- package/dist/extract.js +103 -0
- package/dist/index.d.ts +5 -0
- package/dist/index.js +5 -0
- package/dist/markdown.d.ts +10 -2
- package/dist/markdown.js +11 -4
- package/dist/ocr.js +34 -4
- package/dist/pages.d.ts +24 -0
- package/dist/pages.js +92 -0
- package/dist/security.d.ts +40 -0
- package/dist/security.js +100 -0
- package/package.json +2 -1
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
export interface AgentChunk {
|
|
2
|
+
/** Hash del contenido: el mismo texto da el mismo id en cualquier máquina. */
|
|
3
|
+
id: string;
|
|
4
|
+
index: number;
|
|
5
|
+
/** Primera y última página de las que sale el texto (base 1), o null si el Markdown no traía marcas. */
|
|
6
|
+
pages: [number, number] | null;
|
|
7
|
+
/** Ruta de encabezados, del más general al más concreto. */
|
|
8
|
+
section: string[];
|
|
9
|
+
text: string;
|
|
10
|
+
tokens: number;
|
|
11
|
+
}
|
|
12
|
+
/** Dos semillas: 64 bits de identificador bastan para no chocar en un corpus. */
|
|
13
|
+
export declare function contentId(text: string): string;
|
|
14
|
+
/**
|
|
15
|
+
* Parte Markdown (idealmente generado con `pageMarkers: true`) en trozos con
|
|
16
|
+
* procedencia. Las marcas de página se usan y se retiran del texto final.
|
|
17
|
+
*/
|
|
18
|
+
export declare function chunkForAgents(markdown: string, maxTokens?: number): AgentChunk[];
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
// Trozos para agentes: lo mismo que chunkMarkdown, más lo que un agente
|
|
2
|
+
// necesita para citar y deduplicar sin volver a leer el PDF: de qué páginas
|
|
3
|
+
// sale cada trozo, en qué sección cae y un identificador estable.
|
|
4
|
+
import { chunkMarkdown } from "./chunk.js";
|
|
5
|
+
import { estimateTokens, PAGE_MARKER } from "./markdown.js";
|
|
6
|
+
const HEADING = /^(#{1,6})\s+(.+)$/;
|
|
7
|
+
/** FNV-1a de 32 bits. Sin node:crypto, para que core siga cargando en el navegador. */
|
|
8
|
+
function fnv1a(text, seed) {
|
|
9
|
+
let h = seed >>> 0;
|
|
10
|
+
for (let i = 0; i < text.length; i++) {
|
|
11
|
+
h ^= text.charCodeAt(i);
|
|
12
|
+
h = Math.imul(h, 0x01000193) >>> 0;
|
|
13
|
+
}
|
|
14
|
+
return h.toString(16).padStart(8, "0");
|
|
15
|
+
}
|
|
16
|
+
/** Dos semillas: 64 bits de identificador bastan para no chocar en un corpus. */
|
|
17
|
+
export function contentId(text) {
|
|
18
|
+
return `fnv1a:${fnv1a(text, 0x811c9dc5)}${fnv1a(text, 0x050c5d1f)}`;
|
|
19
|
+
}
|
|
20
|
+
/** Ruta completa de cada encabezado, en el orden en que aparece. */
|
|
21
|
+
function headingPaths(markdown) {
|
|
22
|
+
const paths = new Map();
|
|
23
|
+
const stack = [];
|
|
24
|
+
for (const line of markdown.split("\n")) {
|
|
25
|
+
const m = line.match(HEADING);
|
|
26
|
+
if (!m)
|
|
27
|
+
continue;
|
|
28
|
+
const level = m[1].length;
|
|
29
|
+
while (stack.length && stack[stack.length - 1].level >= level)
|
|
30
|
+
stack.pop();
|
|
31
|
+
stack.push({ level, text: m[2].trim() });
|
|
32
|
+
if (!paths.has(m[2].trim()))
|
|
33
|
+
paths.set(m[2].trim(), stack.map((s) => s.text));
|
|
34
|
+
}
|
|
35
|
+
return paths;
|
|
36
|
+
}
|
|
37
|
+
const stripMarkers = (text) => text
|
|
38
|
+
.split("\n")
|
|
39
|
+
.filter((l) => !PAGE_MARKER.test(l.trim()))
|
|
40
|
+
.join("\n")
|
|
41
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
42
|
+
.trim();
|
|
43
|
+
/**
|
|
44
|
+
* Parte Markdown (idealmente generado con `pageMarkers: true`) en trozos con
|
|
45
|
+
* procedencia. Las marcas de página se usan y se retiran del texto final.
|
|
46
|
+
*/
|
|
47
|
+
export function chunkForAgents(markdown, maxTokens = 2000) {
|
|
48
|
+
const hasMarkers = markdown.split("\n").some((l) => PAGE_MARKER.test(l.trim()));
|
|
49
|
+
const paths = headingPaths(markdown);
|
|
50
|
+
// El texto va en orden, así que la página vigente al empezar un trozo es la
|
|
51
|
+
// última marca vista en los trozos anteriores.
|
|
52
|
+
let current = 1;
|
|
53
|
+
const out = [];
|
|
54
|
+
for (const c of chunkMarkdown(markdown, maxTokens)) {
|
|
55
|
+
const lines = c.text.split("\n");
|
|
56
|
+
let first = null;
|
|
57
|
+
let last = current;
|
|
58
|
+
for (const line of lines) {
|
|
59
|
+
const m = line.trim().match(PAGE_MARKER);
|
|
60
|
+
if (m) {
|
|
61
|
+
current = Number(m[1]);
|
|
62
|
+
continue;
|
|
63
|
+
}
|
|
64
|
+
// Una marca al final del trozo pertenece al siguiente: solo cuenta la
|
|
65
|
+
// página si hay texto real después de ella.
|
|
66
|
+
if (line.trim() && !HEADING.test(line)) {
|
|
67
|
+
if (first === null)
|
|
68
|
+
first = current;
|
|
69
|
+
last = current;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
const text = stripMarkers(c.text);
|
|
73
|
+
if (!text)
|
|
74
|
+
continue;
|
|
75
|
+
out.push({
|
|
76
|
+
id: contentId(text),
|
|
77
|
+
index: out.length + 1,
|
|
78
|
+
pages: hasMarkers ? [first ?? current, Math.max(first ?? current, last)] : null,
|
|
79
|
+
section: c.heading ? (paths.get(c.heading) ?? [c.heading]) : [],
|
|
80
|
+
text,
|
|
81
|
+
tokens: estimateTokens(text),
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
return out;
|
|
85
|
+
}
|
package/dist/anonymize.js
CHANGED
|
@@ -3,20 +3,52 @@ const MONTHS = "enero|febrero|marzo|abril|mayo|junio|julio|agosto|septiembre|set
|
|
|
3
3
|
// se filtra. Cubrimos las partículas españolas ("de", "del", "de la/las/los")
|
|
4
4
|
// y hasta siete palabras, que es lo que llega a tener un nombre compuesto
|
|
5
5
|
// colombiano con dos apellidos.
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
6
|
+
// Los espacios dentro de un nombre son de la MISMA línea. Con \s+ el patrón
|
|
7
|
+
// cruzaba el salto y «Paciente: Juan Carlos Perez Gomez» seguido de «Cedula de
|
|
8
|
+
// ciudadania…» capturaba «…Perez Gomez Cedula» como un solo nombre, dejando la
|
|
9
|
+
// palabra suelta sustituida en medio de la frase siguiente. Un párrafo ya llega
|
|
10
|
+
// con sus líneas unidas, así que no se pierde ningún nombre real por esto.
|
|
11
|
+
const SP = "[ \\t]+";
|
|
12
|
+
const PARTICLE = `(?:de${SP}(?:la|las|los)${SP}|del${SP}|de${SP})?`;
|
|
13
|
+
const NAME = `[A-ZÁÉÍÓÚÑ][a-záéíóúñü]+(?:${SP}${PARTICLE}[A-ZÁÉÍÓÚÑ][a-záéíóúñü]+){1,6}`;
|
|
14
|
+
const NAME_UPPER = `[A-ZÁÉÍÓÚÑ]{2,}(?:${SP}[A-ZÁÉÍÓÚÑ]{2,}){1,6}`;
|
|
9
15
|
/** Vuelve insensible a mayúsculas solo las etiquetas (el nombre sí distingue mayúsculas). */
|
|
10
16
|
const ci = (src) => src.replace(/[a-záéíóúñ]/g, (ch) => `[${ch}${ch.toUpperCase()}]`);
|
|
11
17
|
const NAME_LABELS = "nombres?(?:\\s+completo)?|paciente|usuario|afiliado|señora?|sr\\.|sra\\.|don|doña|comprador(?:a)?|vendedor(?:a)?|otorgante|compareciente|demandante|demandado|arrendador(?:a)?|arrendatario|apoderad[oa]|titular|firmado por|name|patient|client|signed by";
|
|
18
|
+
/** Algoritmo de Luhn: todas las tarjetas lo cumplen; la mayoría de números al azar, no. */
|
|
19
|
+
function luhn(value) {
|
|
20
|
+
const d = value.replace(/\D/g, "");
|
|
21
|
+
if (d.length < 13 || d.length > 19)
|
|
22
|
+
return false;
|
|
23
|
+
let sum = 0;
|
|
24
|
+
for (let i = 0; i < d.length; i++) {
|
|
25
|
+
let n = Number(d[d.length - 1 - i]);
|
|
26
|
+
if (i % 2)
|
|
27
|
+
n = n * 2 > 9 ? n * 2 - 9 : n * 2;
|
|
28
|
+
sum += n;
|
|
29
|
+
}
|
|
30
|
+
return sum % 10 === 0;
|
|
31
|
+
}
|
|
32
|
+
/** Números largos de documentos comerciales que no son tarjetas aunque pasen Luhn (1 de cada 10 lo hace por azar). */
|
|
33
|
+
const NOT_A_CARD = /(resoluci[oó]n|cufe|cude|autorizaci[oó]n|radicado|factura|consecutivo|orden|pedido|gu[ií]a|referencia|c[oó]digo|matr[ií]cula|expediente)[^\n]{0,30}$/i;
|
|
12
34
|
const RULES = [
|
|
13
35
|
{ type: "EMAIL", re: /[\w.+-]+@[\w-]+(?:\.[\w-]+)+/g },
|
|
14
|
-
|
|
36
|
+
// El separador NO puede cruzar una línea en blanco, y el número tiene que
|
|
37
|
+
// llevar al menos un dígito. Sin las dos condiciones, un título «Historia
|
|
38
|
+
// clínica» seguido de «Paciente: Fulano» capturaba la palabra «Paciente»
|
|
39
|
+
// como si fuera el número de historia, la sustituía en todo el documento y
|
|
40
|
+
// con eso rompía el patrón «Paciente: nombre»: el nombre quedaba a la vista.
|
|
41
|
+
{
|
|
42
|
+
type: "HISTORIA_CLINICA",
|
|
43
|
+
re: /\b(?:historia\s+cl[ií]nica|h\.?\s?c\.?|n[uú]mero\s+de\s+historia)[ \t]*(?:n[oº°.]*|#|:)?[ \t]*((?=[A-Z0-9-]*\d)[A-Z0-9-]{4,})/gi,
|
|
44
|
+
group: 1,
|
|
45
|
+
},
|
|
15
46
|
{ type: "NIT", re: /\b(?:NIT\.?\s*:?\s*)?\d{3}\.?\d{3}\.?\d{3}\s?-\s?\d\b/g },
|
|
16
|
-
{ type: "TARJETA", re: /\b(?:\d{4}[ -]?){3}\d{1,4}\b/g },
|
|
47
|
+
{ type: "TARJETA", re: /\b(?:\d{4}[ -]?){3}\d{1,4}\b/g, accept: (v, before) => luhn(v) && !NOT_A_CARD.test(before) },
|
|
17
48
|
{
|
|
18
49
|
type: "DOCUMENTO",
|
|
19
|
-
|
|
50
|
+
// Mismo motivo que arriba: la etiqueta y el número van en la misma línea.
|
|
51
|
+
re: /\b(?:C\.?\s?C\.?|c[eé]dula(?:\s+de\s+ciudadan[ií]a)?|T\.?\s?I\.?|C\.?\s?E\.?|NUIP|pasaporte|DNI|RUT|CURP|RFC|ID)[ \t]*(?:n[oº°.]*|#|:)?[ \t]*([A-Z]{0,4}\d[\d.\s-]{4,14}\d)/gi,
|
|
20
52
|
group: 1,
|
|
21
53
|
},
|
|
22
54
|
{ type: "DOCUMENTO", re: /(?<![$€£\d.,]\s?)\b\d{1,3}(?:\.\d{3}){2,3}\b(?!\s*(?:pesos|cop|usd|millones|m²|m2))/gi },
|
|
@@ -36,7 +68,14 @@ const RULES = [
|
|
|
36
68
|
type: "DIRECCION",
|
|
37
69
|
re: /\b(?:calle|cll?|carrera|cra|kr|kra|avenida|av|transversal|tv|diagonal|dg|autopista|street|st|avenue|ave)\.?\s*\d+[a-z]?(?:\s*bis)?(?:\s*(?:sur|norte|este))?\s*(?:#|n[oº°.]*|no\.?)\s*\d+[a-z]?\s*-\s*\d+(?:\s*(?:sur|norte|este))?(?:[,\s]+(?:apto|apartamento|oficina|of|int|interior|casa|torre)\.?\s*[\w-]+)*/gi,
|
|
38
70
|
},
|
|
39
|
-
|
|
71
|
+
// Se admite un único salto de línea entre la etiqueta y el nombre («Paciente:»
|
|
72
|
+
// al final de una línea y el nombre en la siguiente), pero no una línea en
|
|
73
|
+
// blanco: eso ya es otro párrafo y capturarlo produce falsos positivos.
|
|
74
|
+
{
|
|
75
|
+
type: "NOMBRE",
|
|
76
|
+
re: new RegExp(`(?:${ci(NAME_LABELS)})[ \\t]*[:\\-]?[ \\t]*(?:\\r?\\n[ \\t]*)?(${NAME_UPPER}|${NAME})(?![a-záéíóúñ])`, "g"),
|
|
77
|
+
group: 1,
|
|
78
|
+
},
|
|
40
79
|
];
|
|
41
80
|
export function anonymize(text) {
|
|
42
81
|
const map = new Map();
|
|
@@ -61,6 +100,15 @@ export function anonymize(text) {
|
|
|
61
100
|
const target = rule.group ? args[rule.group] : full;
|
|
62
101
|
if (!target || target.startsWith("["))
|
|
63
102
|
return full;
|
|
103
|
+
if (rule.accept) {
|
|
104
|
+
// Firma de replace: (coincidencia, ...grupos, offset, cadena[, grupos con nombre]).
|
|
105
|
+
const named = typeof args[args.length - 1] === "object";
|
|
106
|
+
const offset = args[args.length - (named ? 3 : 2)];
|
|
107
|
+
const source = args[args.length - (named ? 2 : 1)];
|
|
108
|
+
const before = source.slice(source.lastIndexOf("\n", offset) + 1, offset);
|
|
109
|
+
if (!rule.accept(target, before))
|
|
110
|
+
return full;
|
|
111
|
+
}
|
|
64
112
|
if (rule.type === "NOMBRE" && /\[/.test(target))
|
|
65
113
|
return full;
|
|
66
114
|
const label = labelFor(rule.type, target);
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
export type DocType = "factura" | "contrato" | "historia_clinica" | "cedula" | "rut" | "extracto_bancario" | "certificado" | "hoja_de_vida" | "desconocido";
|
|
2
|
+
export interface Classification {
|
|
3
|
+
type: DocType;
|
|
4
|
+
/** 0–1. Proporción de señales del ganador frente al total; no es una probabilidad calibrada. */
|
|
5
|
+
confidence: number;
|
|
6
|
+
/** Señales encontradas, para que un humano pueda comprobar por qué. */
|
|
7
|
+
signals: string[];
|
|
8
|
+
/** Segundo candidato, útil cuando la confianza es baja. */
|
|
9
|
+
runnerUp: DocType | null;
|
|
10
|
+
}
|
|
11
|
+
/** Clasifica por el texto (basta con las primeras páginas). */
|
|
12
|
+
export declare function classifyDocument(text: string): Classification;
|
package/dist/classify.js
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
// Tipo de documento por reglas, sin modelo: un agente decide qué hacer con un
|
|
2
|
+
// PDF (qué campos buscar, si anonimizar) antes de leerlo entero, y el contenido
|
|
3
|
+
// nunca sale de la máquina para averiguarlo.
|
|
4
|
+
const RULES = {
|
|
5
|
+
factura: [
|
|
6
|
+
[/factura\s+(electr[oó]nica\s+)?de\s+venta/i, 4],
|
|
7
|
+
[/\bcufe\b/i, 4],
|
|
8
|
+
[/\binvoice\b/i, 3],
|
|
9
|
+
[/\biva\b/i, 1],
|
|
10
|
+
[/subtotal/i, 2],
|
|
11
|
+
[/total\s+a\s+pagar/i, 2],
|
|
12
|
+
[/resoluci[oó]n\s+dian/i, 3],
|
|
13
|
+
],
|
|
14
|
+
contrato: [
|
|
15
|
+
[/\bcontrato\s+de\b/i, 4],
|
|
16
|
+
[/cl[aá]usula\s+(primera|segunda|tercera|\d+)/i, 3],
|
|
17
|
+
[/\bel\s+contratante\b|\bel\s+contratista\b/i, 3],
|
|
18
|
+
[/\bpartes\b.*\bacuerdan\b/is, 2],
|
|
19
|
+
[/\bagreement\b|\bwhereas\b/i, 3],
|
|
20
|
+
[/en\s+constancia\s+se\s+firma/i, 2],
|
|
21
|
+
],
|
|
22
|
+
historia_clinica: [
|
|
23
|
+
[/historia\s+cl[ií]nica/i, 4],
|
|
24
|
+
[/\bepicrisis\b/i, 4],
|
|
25
|
+
[/motivo\s+de\s+consulta/i, 3],
|
|
26
|
+
[/\bdiagn[oó]stico\b/i, 2],
|
|
27
|
+
[/\bcie-?10\b/i, 3],
|
|
28
|
+
[/signos\s+vitales|tensi[oó]n\s+arterial/i, 2],
|
|
29
|
+
],
|
|
30
|
+
cedula: [
|
|
31
|
+
[/c[eé]dula\s+de\s+ciudadan[ií]a/i, 4],
|
|
32
|
+
[/registradur[ií]a\s+nacional/i, 4],
|
|
33
|
+
[/fecha\s+y\s+lugar\s+de\s+nacimiento/i, 2],
|
|
34
|
+
[/\bgrupo\s+sangu[ií]neo\b|\bG\.?S\.?\s*RH\b/i, 2],
|
|
35
|
+
],
|
|
36
|
+
rut: [
|
|
37
|
+
[/registro\s+[uú]nico\s+tributario/i, 5],
|
|
38
|
+
[/formulario\s+del\s+registro/i, 2],
|
|
39
|
+
[/responsabilidades,?\s+calidades/i, 3],
|
|
40
|
+
[/direcci[oó]n\s+seccional/i, 2],
|
|
41
|
+
],
|
|
42
|
+
extracto_bancario: [
|
|
43
|
+
[/extracto/i, 3],
|
|
44
|
+
[/saldo\s+(anterior|final|disponible)/i, 3],
|
|
45
|
+
[/movimientos/i, 2],
|
|
46
|
+
[/\bbank\s+statement\b/i, 4],
|
|
47
|
+
[/n[uú]mero\s+de\s+cuenta/i, 1],
|
|
48
|
+
],
|
|
49
|
+
certificado: [
|
|
50
|
+
[/\bcertifica(do)?\b/i, 2],
|
|
51
|
+
[/hace\s+constar/i, 3],
|
|
52
|
+
[/certificado\s+de\s+(existencia|tradici[oó]n|ingresos|retenci[oó]n)/i, 4],
|
|
53
|
+
],
|
|
54
|
+
hoja_de_vida: [
|
|
55
|
+
[/hoja\s+de\s+vida|curr[ií]culum/i, 4],
|
|
56
|
+
[/experiencia\s+(laboral|profesional)/i, 3],
|
|
57
|
+
[/formaci[oó]n\s+acad[eé]mica|educaci[oó]n/i, 2],
|
|
58
|
+
[/referencias\s+(personales|laborales)/i, 2],
|
|
59
|
+
],
|
|
60
|
+
};
|
|
61
|
+
/** Clasifica por el texto (basta con las primeras páginas). */
|
|
62
|
+
export function classifyDocument(text) {
|
|
63
|
+
const sample = text.slice(0, 20000);
|
|
64
|
+
const scores = [];
|
|
65
|
+
for (const [type, rules] of Object.entries(RULES)) {
|
|
66
|
+
let score = 0;
|
|
67
|
+
const signals = [];
|
|
68
|
+
for (const [re, weight] of rules) {
|
|
69
|
+
const m = sample.match(re);
|
|
70
|
+
if (m) {
|
|
71
|
+
score += weight;
|
|
72
|
+
signals.push(m[0].replace(/\s+/g, " ").trim().slice(0, 40));
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
scores.push({ type, score, signals });
|
|
76
|
+
}
|
|
77
|
+
scores.sort((a, b) => b.score - a.score);
|
|
78
|
+
const [best, second] = scores;
|
|
79
|
+
const total = scores.reduce((n, s) => n + s.score, 0);
|
|
80
|
+
// Por debajo de 3 puntos es una palabra suelta, no un documento de ese tipo.
|
|
81
|
+
if (best.score < 3)
|
|
82
|
+
return { type: "desconocido", confidence: 0, signals: [], runnerUp: best.score ? best.type : null };
|
|
83
|
+
return {
|
|
84
|
+
type: best.type,
|
|
85
|
+
confidence: Math.round((best.score / total) * 100) / 100,
|
|
86
|
+
signals: best.signals,
|
|
87
|
+
runnerUp: second.score ? second.type : null,
|
|
88
|
+
};
|
|
89
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import type { DocType } from "./classify.js";
|
|
2
|
+
export interface ExtractedField {
|
|
3
|
+
/** Nombre estable en snake_case, pensado para JSON y hojas de cálculo. */
|
|
4
|
+
name: string;
|
|
5
|
+
value: string;
|
|
6
|
+
/** Página (base 1) donde apareció, si el Markdown traía marcas de página. */
|
|
7
|
+
page: number | null;
|
|
8
|
+
/** El fragmento que casó la regla: la cita para que una persona lo verifique. */
|
|
9
|
+
evidence: string;
|
|
10
|
+
/** Dato de una persona: se sustituye si se pide anonimizar. */
|
|
11
|
+
personal: boolean;
|
|
12
|
+
}
|
|
13
|
+
/** Tipos para los que hay reglas: el resto devuelve lista vacía, no un error. */
|
|
14
|
+
export declare const EXTRACTABLE_TYPES: DocType[];
|
|
15
|
+
/** Extrae los campos de un tipo. Mejor con Markdown generado con `pageMarkers: true`, para citar la página. */
|
|
16
|
+
export declare function extractFields(markdown: string, type: DocType): ExtractedField[];
|
package/dist/extract.js
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
import { PAGE_MARKER } from "./markdown.js";
|
|
2
|
+
const money = (v) => v.replace(/[\s$]/g, "");
|
|
3
|
+
const digits = (v) => v.replace(/[^\dkK-]/g, "");
|
|
4
|
+
const AMOUNT = String.raw `\$?\s*([\d]{1,3}(?:[.,\s]\d{3})*(?:[.,]\d{1,2})?)`;
|
|
5
|
+
const DATE = String.raw `(\d{1,2}[/-]\d{1,2}[/-]\d{2,4}|\d{4}-\d{2}-\d{2}|\d{1,2}\s+de\s+[a-záéíóú]+\s+de\s+\d{4})`;
|
|
6
|
+
const NIT = String.raw `(\d{3}\.?\d{3}\.?\d{3}\s*-?\s*\d)`;
|
|
7
|
+
const RULES = {
|
|
8
|
+
factura: [
|
|
9
|
+
{ name: "numero_factura", re: /factura[^\n]{0,40}?(?:n[oº°.]*|número|numero)\s*[:.]?\s*([A-Z]{0,6}-?\d{1,12})/i },
|
|
10
|
+
{ name: "cufe", re: /\bcufe\s*[:.]?\s*([0-9a-f]{8,128})/i },
|
|
11
|
+
{ name: "nit_emisor", re: new RegExp(String.raw `nit[^\n\d]{0,20}` + NIT, "i"), clean: digits },
|
|
12
|
+
{ name: "fecha_emision", re: new RegExp(String.raw `fecha(?:\s+de)?\s*(?:emisi[oó]n|expedici[oó]n|factura)?\s*[:.]?\s*` + DATE, "i") },
|
|
13
|
+
{ name: "fecha_vencimiento", re: new RegExp(String.raw `vencimiento\s*[:.]?\s*` + DATE, "i") },
|
|
14
|
+
{ name: "subtotal", re: new RegExp(String.raw `subtotal\s*[:.]?\s*` + AMOUNT, "i"), clean: money },
|
|
15
|
+
{ name: "iva", re: new RegExp(String.raw `\biva\b(?:\s*\d{1,2}\s*%)?\s*[:.]?\s*` + AMOUNT, "i"), clean: money },
|
|
16
|
+
// «Subtotal» también contiene «total»: se exige que no vaya pegado a otra palabra.
|
|
17
|
+
{ name: "total", re: new RegExp(String.raw `(?<![a-záéíóú])total\s*(?:a\s+pagar|factura|general)?\s*[:.]?\s*` + AMOUNT, "i"), clean: money },
|
|
18
|
+
{ name: "resolucion_dian", re: /resoluci[oó]n(?:\s+dian)?\s*(?:n[oº°.]*)?\s*[:.]?\s*(\d{8,20})/i },
|
|
19
|
+
],
|
|
20
|
+
rut: [
|
|
21
|
+
{ name: "nit", re: new RegExp(String.raw `(?:n[uú]mero\s+de\s+identificaci[oó]n\s+tributaria|\bnit\b)[^\n\d]{0,30}` + NIT, "i"), clean: digits },
|
|
22
|
+
{ name: "razon_social", re: /raz[oó]n\s+social\s*[:.]?\s*([^\n]{3,80})/i },
|
|
23
|
+
{ name: "direccion_seccional", re: /direcci[oó]n\s+seccional\s*[:.]?\s*([^\n]{3,60})/i },
|
|
24
|
+
{ name: "actividad_principal", re: /actividad\s+principal\s*[:.]?\s*(\d{4})/i },
|
|
25
|
+
],
|
|
26
|
+
cedula: [
|
|
27
|
+
{ name: "numero_documento", re: /(?:n[uú]mero|c\.?\s?c\.?|c[eé]dula(?:\s+de\s+ciudadan[ií]a)?)\s*[:.]?\s*(\d{1,3}(?:\.\d{3}){1,3}|\d{6,10})\b/i, personal: true, clean: digits },
|
|
28
|
+
{ name: "fecha_nacimiento", re: new RegExp(String.raw `fecha\s+de\s+nacimiento\s*[:.]?\s*` + DATE, "i"), personal: true },
|
|
29
|
+
{ name: "fecha_expedicion", re: new RegExp(String.raw `fecha\s+(?:y\s+lugar\s+)?de\s+expedici[oó]n\s*[:.]?\s*` + DATE, "i") },
|
|
30
|
+
],
|
|
31
|
+
historia_clinica: [
|
|
32
|
+
{ name: "paciente", re: /[Pp]aciente\s*[:.]?\s*([A-ZÁÉÍÓÚÑ][a-záéíóúñ]+(?:[ \t]+[A-ZÁÉÍÓÚÑ][a-záéíóúñ]+){1,4})/, personal: true },
|
|
33
|
+
{ name: "documento_paciente", re: /(?:c\.?\s?c\.?|documento|identificaci[oó]n)\s*[:.]?\s*(\d{1,3}(?:\.\d{3}){1,3}|\d{6,10})\b/i, personal: true, clean: digits },
|
|
34
|
+
{ name: "fecha_atencion", re: new RegExp(String.raw `fecha(?:\s+de)?\s*(?:atenci[oó]n|ingreso|consulta)?\s*[:.]?\s*` + DATE, "i") },
|
|
35
|
+
{ name: "motivo_consulta", re: /motivo\s+de\s+consulta\s*[:.]?\s*([^\n]{3,160})/i },
|
|
36
|
+
{ name: "diagnostico_cie10", re: /\b([A-TV-Z]\d{2}(?:\.?\d{1,2})?)\b(?=[^\n]{0,80}(?:diagn|cie|principal|relacionado)|\s*[-–:]\s*[A-ZÁÉÍÓÚ])/g, all: true },
|
|
37
|
+
],
|
|
38
|
+
extracto_bancario: [
|
|
39
|
+
{ name: "numero_cuenta", re: /(?:n[uú]mero\s+de\s+)?cuenta\s*(?:n[oº°.]*)?\s*[:.]?\s*([*\dX]{4,}[\d-]*)/i, personal: true },
|
|
40
|
+
{ name: "periodo", re: new RegExp(String.raw `(?:periodo|per[ií]odo)\s*[:.]?\s*` + DATE + String.raw `(?:\s*(?:al|a|-)\s*)` + DATE, "i") },
|
|
41
|
+
{ name: "saldo_anterior", re: new RegExp(String.raw `saldo\s+anterior\s*[:.]?\s*` + AMOUNT, "i"), clean: money },
|
|
42
|
+
{ name: "saldo_final", re: new RegExp(String.raw `saldo\s+(?:final|actual|disponible|total)\s*[:.]?\s*` + AMOUNT, "i"), clean: money },
|
|
43
|
+
],
|
|
44
|
+
contrato: [
|
|
45
|
+
{ name: "tipo_contrato", re: /contrato\s+de\s+([a-záéíóúñ ]{4,60}?)(?:\s+(?:entre|n[oº°.]|celebrado|suscrito)|[\n.,])/i },
|
|
46
|
+
{ name: "valor", re: new RegExp(String.raw `valor(?:\s+(?:total|del\s+contrato))?\s*[:.]?\s*(?:de\s+)?` + AMOUNT, "i"), clean: money },
|
|
47
|
+
{ name: "plazo", re: /(?:plazo|duraci[oó]n|t[eé]rmino)(?:\s+de\s+ejecuci[oó]n)?\s*[:.]?\s*(?:de\s+)?(\d{1,3}\s*\(?[a-z ]*\)?\s*(?:d[ií]as|meses|años))/i },
|
|
48
|
+
{ name: "fecha_firma", re: new RegExp(String.raw `(?:firma|suscribe|a\s+los)[^\n]{0,40}?` + DATE, "i") },
|
|
49
|
+
],
|
|
50
|
+
certificado: [
|
|
51
|
+
{ name: "entidad", re: /^([A-ZÁÉÍÓÚÑ][A-ZÁÉÍÓÚÑ .&]{5,80})$/m },
|
|
52
|
+
{ name: "fecha_expedicion", re: new RegExp(String.raw `(?:expedid[oa]|fecha)[^\n]{0,30}?` + DATE, "i") },
|
|
53
|
+
],
|
|
54
|
+
};
|
|
55
|
+
/** Página de una posición: la última marca `<!-- page N -->` antes de ella. */
|
|
56
|
+
function pageAt(markdown, index) {
|
|
57
|
+
let page = null;
|
|
58
|
+
const re = new RegExp(PAGE_MARKER.source.replace("^", "").replace("$", ""), "g");
|
|
59
|
+
for (const m of markdown.matchAll(re)) {
|
|
60
|
+
if ((m.index ?? 0) > index)
|
|
61
|
+
break;
|
|
62
|
+
page = Number(m[1]);
|
|
63
|
+
}
|
|
64
|
+
return page;
|
|
65
|
+
}
|
|
66
|
+
function evidenceAround(text, index, length) {
|
|
67
|
+
const start = text.lastIndexOf("\n", index) + 1;
|
|
68
|
+
const endNl = text.indexOf("\n", index + length);
|
|
69
|
+
return text
|
|
70
|
+
.slice(start, endNl === -1 ? undefined : endNl)
|
|
71
|
+
.trim()
|
|
72
|
+
.slice(0, 160);
|
|
73
|
+
}
|
|
74
|
+
/** Tipos para los que hay reglas: el resto devuelve lista vacía, no un error. */
|
|
75
|
+
export const EXTRACTABLE_TYPES = Object.keys(RULES);
|
|
76
|
+
/** Extrae los campos de un tipo. Mejor con Markdown generado con `pageMarkers: true`, para citar la página. */
|
|
77
|
+
export function extractFields(markdown, type) {
|
|
78
|
+
const rules = RULES[type] ?? [];
|
|
79
|
+
const out = [];
|
|
80
|
+
for (const rule of rules) {
|
|
81
|
+
const re = rule.all ? new RegExp(rule.re.source, rule.re.flags.includes("g") ? rule.re.flags : rule.re.flags + "g") : rule.re;
|
|
82
|
+
const matches = rule.all ? [...markdown.matchAll(re)] : [markdown.match(re)].filter((m) => !!m);
|
|
83
|
+
const seen = new Set();
|
|
84
|
+
for (const m of matches) {
|
|
85
|
+
const raw = (m[1] ?? "").trim();
|
|
86
|
+
if (!raw)
|
|
87
|
+
continue;
|
|
88
|
+
const value = rule.clean ? rule.clean(raw) : raw.replace(/\s+/g, " ");
|
|
89
|
+
if (seen.has(value))
|
|
90
|
+
continue;
|
|
91
|
+
seen.add(value);
|
|
92
|
+
const index = m.index ?? 0;
|
|
93
|
+
out.push({
|
|
94
|
+
name: rule.name,
|
|
95
|
+
value,
|
|
96
|
+
page: pageAt(markdown, index),
|
|
97
|
+
evidence: evidenceAround(markdown, index, m[0].length),
|
|
98
|
+
personal: !!rule.personal,
|
|
99
|
+
});
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
return out;
|
|
103
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -17,3 +17,8 @@ export * from "./datasplit.js";
|
|
|
17
17
|
export * from "./dataset.js";
|
|
18
18
|
export * from "./verifiable.js";
|
|
19
19
|
export * from "./paragraphs.js";
|
|
20
|
+
export * from "./agentchunks.js";
|
|
21
|
+
export * from "./classify.js";
|
|
22
|
+
export * from "./pages.js";
|
|
23
|
+
export * from "./security.js";
|
|
24
|
+
export * from "./extract.js";
|
package/dist/index.js
CHANGED
|
@@ -19,3 +19,8 @@ export * from "./datasplit.js";
|
|
|
19
19
|
export * from "./dataset.js";
|
|
20
20
|
export * from "./verifiable.js";
|
|
21
21
|
export * from "./paragraphs.js";
|
|
22
|
+
export * from "./agentchunks.js";
|
|
23
|
+
export * from "./classify.js";
|
|
24
|
+
export * from "./pages.js";
|
|
25
|
+
export * from "./security.js";
|
|
26
|
+
export * from "./extract.js";
|
package/dist/markdown.d.ts
CHANGED
|
@@ -13,9 +13,17 @@ type Progress = (ratio: number) => void;
|
|
|
13
13
|
/** Extrae las líneas de texto con tamaño y posición de un documento abierto. */
|
|
14
14
|
export declare function extractLines(pdf: PDFDocumentProxy, onProgress?: Progress): Promise<Line[]>;
|
|
15
15
|
/** Convierte líneas ya extraídas en Markdown (encabezados, listas, párrafos, campos). */
|
|
16
|
-
export
|
|
16
|
+
export interface MarkdownOptions {
|
|
17
|
+
/**
|
|
18
|
+
* Inserta `<!-- page N -->` donde empieza cada página. Invisible al renderizar,
|
|
19
|
+
* pero permite que los trozos sepan de qué página salen y el agente cite.
|
|
20
|
+
*/
|
|
21
|
+
pageMarkers?: boolean;
|
|
22
|
+
}
|
|
23
|
+
export declare const PAGE_MARKER: RegExp;
|
|
24
|
+
export declare function linesToMarkdown(lines: Line[], opts?: MarkdownOptions): string;
|
|
17
25
|
/** Documento PDF.js abierto → Markdown. */
|
|
18
|
-
export declare function pdfToMarkdown(pdf: PDFDocumentProxy, onProgress?: Progress): Promise<string>;
|
|
26
|
+
export declare function pdfToMarkdown(pdf: PDFDocumentProxy, onProgress?: Progress, opts?: MarkdownOptions): Promise<string>;
|
|
19
27
|
/** Estimación rápida de tokens (~4 caracteres por token). */
|
|
20
28
|
export declare function estimateTokens(text: string): number;
|
|
21
29
|
export {};
|
package/dist/markdown.js
CHANGED
|
@@ -86,8 +86,8 @@ function removeRepeated(lines, pages) {
|
|
|
86
86
|
return true;
|
|
87
87
|
});
|
|
88
88
|
}
|
|
89
|
-
|
|
90
|
-
export function linesToMarkdown(lines) {
|
|
89
|
+
export const PAGE_MARKER = /^<!-- page (\d+) -->$/;
|
|
90
|
+
export function linesToMarkdown(lines, opts = {}) {
|
|
91
91
|
if (!lines.length)
|
|
92
92
|
return "";
|
|
93
93
|
const body = mode(lines.map((l) => l.size));
|
|
@@ -105,7 +105,14 @@ export function linesToMarkdown(lines) {
|
|
|
105
105
|
out.push("");
|
|
106
106
|
inKv = false;
|
|
107
107
|
};
|
|
108
|
+
let markedPage = -1;
|
|
108
109
|
for (const l of lines) {
|
|
110
|
+
if (opts.pageMarkers && l.page !== markedPage) {
|
|
111
|
+
flush();
|
|
112
|
+
closeKv();
|
|
113
|
+
out.push(`<!-- page ${l.page} -->`, "");
|
|
114
|
+
markedPage = l.page;
|
|
115
|
+
}
|
|
109
116
|
const ratio = l.size / body;
|
|
110
117
|
const short = l.text.length < 90 && !/[.,;:]$/.test(l.text);
|
|
111
118
|
let heading = 0;
|
|
@@ -168,8 +175,8 @@ export function linesToMarkdown(lines) {
|
|
|
168
175
|
.trim();
|
|
169
176
|
}
|
|
170
177
|
/** Documento PDF.js abierto → Markdown. */
|
|
171
|
-
export async function pdfToMarkdown(pdf, onProgress) {
|
|
172
|
-
return linesToMarkdown(await extractLines(pdf, onProgress));
|
|
178
|
+
export async function pdfToMarkdown(pdf, onProgress, opts) {
|
|
179
|
+
return linesToMarkdown(await extractLines(pdf, onProgress), opts);
|
|
173
180
|
}
|
|
174
181
|
/** Estimación rápida de tokens (~4 caracteres por token). */
|
|
175
182
|
export function estimateTokens(text) {
|
package/dist/ocr.js
CHANGED
|
@@ -1,14 +1,40 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* El alto de una caja varía según tenga tildes o letras con cola ("Hola" vs
|
|
3
|
-
* "Hpqy"), así que agrupamos tamaños parecidos en cubos.
|
|
4
|
-
* de "tamaño de cuerpo" del Markdown vería decenas de tamaños distintos y no
|
|
5
|
-
* distinguiría un encabezado de un párrafo.
|
|
3
|
+
* "Hpqy"), así que agrupamos tamaños parecidos en cubos.
|
|
6
4
|
*/
|
|
7
5
|
const BUCKET = 2;
|
|
8
6
|
const bucket = (n) => Math.max(BUCKET, Math.round(n / BUCKET) * BUCKET);
|
|
7
|
+
/**
|
|
8
|
+
* Cuánto más alta que el cuerpo tiene que ser una línea para considerarla
|
|
9
|
+
* encabezado.
|
|
10
|
+
*
|
|
11
|
+
* El alto de una caja de OCR NO es el tamaño de la letra: es la envolvente de
|
|
12
|
+
* lo que se dibujó. En una página con todo el cuerpo al mismo tamaño, medimos
|
|
13
|
+
* 9,1 pt en "Cedula de ciudadania" y 12,0 pt en "Correo: juan.perez@…", solo
|
|
14
|
+
* porque la segunda tiene jota y arroba. Con un umbral bajo, cada línea con
|
|
15
|
+
* una cola se convertía en un «###», el documento salía troceado en secciones
|
|
16
|
+
* falsas y los párrafos se perdían.
|
|
17
|
+
*
|
|
18
|
+
* 1,35 es deliberadamente conservador: un encabezado sutil (1,15× el cuerpo)
|
|
19
|
+
* es indistinguible del ruido y se deja como párrafo. Perder un encabezado
|
|
20
|
+
* discreto molesta; inventar veinte destruye la estructura.
|
|
21
|
+
*/
|
|
22
|
+
const HEADING_RATIO = 1.35;
|
|
23
|
+
/** Mediana: resiste que una línea suelta salga enorme o diminuta. */
|
|
24
|
+
function median(values) {
|
|
25
|
+
if (values.length === 0)
|
|
26
|
+
return 0;
|
|
27
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
28
|
+
const mid = sorted.length >> 1;
|
|
29
|
+
return sorted.length % 2 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
|
|
30
|
+
}
|
|
9
31
|
/** Convierte las cajas de una página OCR en líneas del pipeline de Markdown. */
|
|
10
32
|
export function ocrPageToLines({ page, heightPx, scale, boxes }) {
|
|
11
33
|
const heightPt = heightPx / scale;
|
|
34
|
+
// El tamaño del cuerpo de ESTA página, no un valor absoluto: un escaneo a
|
|
35
|
+
// 150 dpi y otro a 300 tienen que producir el mismo Markdown.
|
|
36
|
+
const usable = boxes.filter((b) => b.text.trim() !== "");
|
|
37
|
+
const bodyPt = median(usable.map((b) => (b.y1 - b.y0) / scale));
|
|
12
38
|
const lines = [];
|
|
13
39
|
for (const b of boxes) {
|
|
14
40
|
const text = b.text.replace(/\s+/g, " ").trim();
|
|
@@ -17,9 +43,13 @@ export function ocrPageToLines({ page, heightPx, scale, boxes }) {
|
|
|
17
43
|
// PDF mide la y desde abajo; el OCR desde arriba. Invertimos para que el
|
|
18
44
|
// resto del pipeline (saltos de bloque, encabezados/pies) funcione igual.
|
|
19
45
|
const y = (heightPx - b.y1) / scale;
|
|
46
|
+
// Todo lo que no sea claramente más alto que el cuerpo SE IGUALA al cuerpo.
|
|
47
|
+
// Si no, el ruido de las cajas decide los encabezados.
|
|
48
|
+
const raw = (b.y1 - b.y0) / scale;
|
|
49
|
+
const size = bodyPt > 0 && raw < bodyPt * HEADING_RATIO ? bucket(bodyPt) : bucket(raw);
|
|
20
50
|
lines.push({
|
|
21
51
|
text,
|
|
22
|
-
size
|
|
52
|
+
size,
|
|
23
53
|
y,
|
|
24
54
|
x: b.x0 / scale,
|
|
25
55
|
page,
|
package/dist/pages.d.ts
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/** Quita las páginas indicadas (base 0). Negarse a vaciar el documento evita un PDF inválido. */
|
|
2
|
+
export declare function deletePages(buf: ArrayBuffer, remove: number[]): Promise<Uint8Array>;
|
|
3
|
+
/** Un PDF nuevo con solo esas páginas, en ese orden. */
|
|
4
|
+
export declare function extractPages(buf: ArrayBuffer, indices: number[]): Promise<Uint8Array>;
|
|
5
|
+
/** Gira páginas concretas (o todas si `indices` va vacío). */
|
|
6
|
+
export declare function rotatePages(buf: ArrayBuffer, angle: 90 | 180 | 270, indices?: number[]): Promise<Uint8Array>;
|
|
7
|
+
export type Corner = "bottom-center" | "bottom-right" | "bottom-left" | "top-center" | "top-right" | "top-left";
|
|
8
|
+
export interface PageNumberOptions {
|
|
9
|
+
position?: Corner;
|
|
10
|
+
/** `{n}` es la página y `{total}` el total. */
|
|
11
|
+
format?: string;
|
|
12
|
+
startAt?: number;
|
|
13
|
+
fontSize?: number;
|
|
14
|
+
/** Saltar las N primeras páginas (portadas). */
|
|
15
|
+
skipFirst?: number;
|
|
16
|
+
}
|
|
17
|
+
export declare function addPageNumbers(buf: ArrayBuffer, opts?: PageNumberOptions): Promise<Uint8Array>;
|
|
18
|
+
export interface WatermarkOptions {
|
|
19
|
+
opacity?: number;
|
|
20
|
+
fontSize?: number;
|
|
21
|
+
/** Diagonal por defecto, que es lo que se reconoce como marca de agua. */
|
|
22
|
+
angle?: number;
|
|
23
|
+
}
|
|
24
|
+
export declare function addWatermark(buf: ArrayBuffer, text: string, opts?: WatermarkOptions): Promise<Uint8Array>;
|
package/dist/pages.js
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
// Operaciones de página que tienen todas las suites de la competencia: quitar,
|
|
2
|
+
// extraer, girar, numerar y marca de agua. Todas con pdf-lib, en local.
|
|
3
|
+
import { PDFDocument, StandardFonts, degrees, rgb } from "pdf-lib";
|
|
4
|
+
async function copyOf(buf, indices) {
|
|
5
|
+
const src = await PDFDocument.load(buf);
|
|
6
|
+
const out = await PDFDocument.create();
|
|
7
|
+
for (const p of await out.copyPages(src, indices))
|
|
8
|
+
out.addPage(p);
|
|
9
|
+
return out.save({ useObjectStreams: true });
|
|
10
|
+
}
|
|
11
|
+
function assertIndices(indices, total) {
|
|
12
|
+
for (const i of indices) {
|
|
13
|
+
if (!Number.isInteger(i) || i < 0 || i >= total)
|
|
14
|
+
throw new Error(`Página fuera del documento: ${i + 1}`);
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
/** Quita las páginas indicadas (base 0). Negarse a vaciar el documento evita un PDF inválido. */
|
|
18
|
+
export async function deletePages(buf, remove) {
|
|
19
|
+
const total = (await PDFDocument.load(buf)).getPageCount();
|
|
20
|
+
assertIndices(remove, total);
|
|
21
|
+
const drop = new Set(remove);
|
|
22
|
+
const keep = Array.from({ length: total }, (_, i) => i).filter((i) => !drop.has(i));
|
|
23
|
+
if (!keep.length)
|
|
24
|
+
throw new Error("No se pueden quitar todas las páginas: el PDF quedaría vacío.");
|
|
25
|
+
return copyOf(buf, keep);
|
|
26
|
+
}
|
|
27
|
+
/** Un PDF nuevo con solo esas páginas, en ese orden. */
|
|
28
|
+
export async function extractPages(buf, indices) {
|
|
29
|
+
const total = (await PDFDocument.load(buf)).getPageCount();
|
|
30
|
+
assertIndices(indices, total);
|
|
31
|
+
if (!indices.length)
|
|
32
|
+
throw new Error("Indica al menos una página.");
|
|
33
|
+
return copyOf(buf, indices);
|
|
34
|
+
}
|
|
35
|
+
/** Gira páginas concretas (o todas si `indices` va vacío). */
|
|
36
|
+
export async function rotatePages(buf, angle, indices = []) {
|
|
37
|
+
const doc = await PDFDocument.load(buf);
|
|
38
|
+
const pages = doc.getPages();
|
|
39
|
+
assertIndices(indices, pages.length);
|
|
40
|
+
const targets = indices.length ? indices : pages.map((_, i) => i);
|
|
41
|
+
for (const i of targets)
|
|
42
|
+
pages[i].setRotation(degrees((pages[i].getRotation().angle + angle) % 360));
|
|
43
|
+
return doc.save({ useObjectStreams: true });
|
|
44
|
+
}
|
|
45
|
+
/** Helvetica estándar solo codifica WinAnsi: mejor un error claro que un PDF corrupto. */
|
|
46
|
+
function safeText(font, text) {
|
|
47
|
+
try {
|
|
48
|
+
font.encodeText(text);
|
|
49
|
+
return text;
|
|
50
|
+
}
|
|
51
|
+
catch {
|
|
52
|
+
throw new Error(`El texto «${text}» tiene caracteres que la fuente estándar del PDF no admite.`);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
export async function addPageNumbers(buf, opts = {}) {
|
|
56
|
+
const { position = "bottom-center", format = "{n}", startAt = 1, fontSize = 10, skipFirst = 0 } = opts;
|
|
57
|
+
const doc = await PDFDocument.load(buf);
|
|
58
|
+
const font = await doc.embedFont(StandardFonts.Helvetica);
|
|
59
|
+
const pages = doc.getPages();
|
|
60
|
+
const numbered = pages.length - skipFirst;
|
|
61
|
+
pages.forEach((page, i) => {
|
|
62
|
+
if (i < skipFirst)
|
|
63
|
+
return;
|
|
64
|
+
const label = safeText(font, format.replace("{n}", String(i - skipFirst + startAt)).replace("{total}", String(numbered + startAt - 1)));
|
|
65
|
+
const { width, height } = page.getSize();
|
|
66
|
+
const w = font.widthOfTextAtSize(label, fontSize);
|
|
67
|
+
const margin = 28;
|
|
68
|
+
const x = position.endsWith("left") ? margin : position.endsWith("right") ? width - margin - w : (width - w) / 2;
|
|
69
|
+
const y = position.startsWith("top") ? height - margin : margin - fontSize / 2;
|
|
70
|
+
page.drawText(label, { x, y, size: fontSize, font, color: rgb(0.25, 0.25, 0.25) });
|
|
71
|
+
});
|
|
72
|
+
return doc.save({ useObjectStreams: true });
|
|
73
|
+
}
|
|
74
|
+
export async function addWatermark(buf, text, opts = {}) {
|
|
75
|
+
if (!text.trim())
|
|
76
|
+
throw new Error("La marca de agua necesita un texto.");
|
|
77
|
+
const { opacity = 0.15, angle = 45 } = opts;
|
|
78
|
+
const doc = await PDFDocument.load(buf);
|
|
79
|
+
const font = await doc.embedFont(StandardFonts.HelveticaBold);
|
|
80
|
+
const label = safeText(font, text.trim());
|
|
81
|
+
for (const page of doc.getPages()) {
|
|
82
|
+
const { width, height } = page.getSize();
|
|
83
|
+
const size = opts.fontSize ?? Math.min(96, (Math.hypot(width, height) * 0.7) / Math.max(label.length * 0.6, 1));
|
|
84
|
+
const w = font.widthOfTextAtSize(label, size);
|
|
85
|
+
const rad = (angle * Math.PI) / 180;
|
|
86
|
+
// Centrar el texto girado: se desplaza el origen media anchura sobre su propio eje.
|
|
87
|
+
const x = width / 2 - (Math.cos(rad) * w) / 2 + (Math.sin(rad) * size) / 3;
|
|
88
|
+
const y = height / 2 - (Math.sin(rad) * w) / 2 - (Math.cos(rad) * size) / 3;
|
|
89
|
+
page.drawText(label, { x, y, size, font, color: rgb(0.55, 0.1, 0.25), opacity, rotate: degrees(angle) });
|
|
90
|
+
}
|
|
91
|
+
return doc.save({ useObjectStreams: true });
|
|
92
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/** Lo mínimo que usamos del módulo Emscripten de qpdf. */
|
|
2
|
+
export interface QpdfModule {
|
|
3
|
+
FS: {
|
|
4
|
+
writeFile(path: string, data: Uint8Array): void;
|
|
5
|
+
readFile(path: string): Uint8Array;
|
|
6
|
+
};
|
|
7
|
+
callMain(args: string[]): number;
|
|
8
|
+
}
|
|
9
|
+
export type QpdfFactory = () => Promise<QpdfModule>;
|
|
10
|
+
export type LockState =
|
|
11
|
+
/** Sin cifrado. */
|
|
12
|
+
"none"
|
|
13
|
+
/** Cifrado solo con contraseña de propietario: se abre, pero con permisos limitados (imprimir, copiar…). */
|
|
14
|
+
| "restricted"
|
|
15
|
+
/** Hace falta contraseña para abrirlo. */
|
|
16
|
+
| "password";
|
|
17
|
+
export declare class WrongPasswordError extends Error {
|
|
18
|
+
constructor();
|
|
19
|
+
}
|
|
20
|
+
export declare function lockState(factory: QpdfFactory, input: Uint8Array): Promise<LockState>;
|
|
21
|
+
export interface ProtectOptions {
|
|
22
|
+
/** Contraseña para abrir. */
|
|
23
|
+
password: string;
|
|
24
|
+
/**
|
|
25
|
+
* Contraseña de propietario (para cambiar permisos). Si no se da, se genera
|
|
26
|
+
* una aleatoria: reutilizar la de apertura haría que cualquiera que abre el
|
|
27
|
+
* archivo pudiera también quitarle las restricciones.
|
|
28
|
+
*/
|
|
29
|
+
ownerPassword?: string;
|
|
30
|
+
allowPrint?: boolean;
|
|
31
|
+
allowCopy?: boolean;
|
|
32
|
+
allowEdit?: boolean;
|
|
33
|
+
}
|
|
34
|
+
/** Cifra con AES-256, el estándar que abren Acrobat, Chrome, Preview y cualquier lector actual. */
|
|
35
|
+
export declare function protectPdf(factory: QpdfFactory, input: Uint8Array, opts: ProtectOptions): Promise<Uint8Array>;
|
|
36
|
+
/**
|
|
37
|
+
* Quita el cifrado. Sin contraseña funciona con los PDF «restringidos» (se
|
|
38
|
+
* abren pero no dejan copiar o imprimir), que son la mayoría de los casos.
|
|
39
|
+
*/
|
|
40
|
+
export declare function unlockPdf(factory: QpdfFactory, input: Uint8Array, password?: string): Promise<Uint8Array>;
|
package/dist/security.js
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
// Poner y quitar contraseñas con qpdf compilado a WebAssembly. core no depende
|
|
2
|
+
// de qpdf: quien llama inyecta el módulo, porque en el navegador el .wasm se
|
|
3
|
+
// sirve desde nuestro propio origen y en Node se lee del disco. Así core sigue
|
|
4
|
+
// cargando sin 1,3 MB que la mayoría de herramientas no necesita.
|
|
5
|
+
import { PDFDocument } from "pdf-lib";
|
|
6
|
+
export class WrongPasswordError extends Error {
|
|
7
|
+
constructor() {
|
|
8
|
+
super("Contraseña incorrecta.");
|
|
9
|
+
this.name = "WrongPasswordError";
|
|
10
|
+
}
|
|
11
|
+
}
|
|
12
|
+
const IN = "/in.pdf";
|
|
13
|
+
const OUT = "/out.pdf";
|
|
14
|
+
/** qpdf termina con `exit()`: Emscripten lo lanza como excepción con `status`. */
|
|
15
|
+
function exitCode(q, args) {
|
|
16
|
+
try {
|
|
17
|
+
return q.callMain(args);
|
|
18
|
+
}
|
|
19
|
+
catch (e) {
|
|
20
|
+
const status = e.status;
|
|
21
|
+
if (typeof status === "number")
|
|
22
|
+
return status;
|
|
23
|
+
throw e;
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
async function run(factory, input, args) {
|
|
27
|
+
// Un módulo por operación: callMain deja estado global y reutilizarlo falla
|
|
28
|
+
// de forma intermitente en la segunda llamada.
|
|
29
|
+
const q = await factory();
|
|
30
|
+
q.FS.writeFile(IN, input);
|
|
31
|
+
const code = exitCode(q, args);
|
|
32
|
+
let output = null;
|
|
33
|
+
try {
|
|
34
|
+
output = q.FS.readFile(OUT);
|
|
35
|
+
}
|
|
36
|
+
catch {
|
|
37
|
+
output = null;
|
|
38
|
+
}
|
|
39
|
+
return { code, output };
|
|
40
|
+
}
|
|
41
|
+
export async function lockState(factory, input) {
|
|
42
|
+
// Esta build de qpdf sale con 2 tanto si el PDF no está cifrado como si pide
|
|
43
|
+
// una contraseña que no tiene, así que pdf-lib decide si hay cifrado y qpdf
|
|
44
|
+
// solo desempata: --requires-password da 3 cuando se abre sin contraseña.
|
|
45
|
+
try {
|
|
46
|
+
await PDFDocument.load(input, { updateMetadata: false });
|
|
47
|
+
return "none";
|
|
48
|
+
}
|
|
49
|
+
catch (e) {
|
|
50
|
+
// Por nombre y no con instanceof: en Node pueden convivir la copia CJS y la ESM de pdf-lib.
|
|
51
|
+
if (e.name !== "EncryptedPDFError" && !/is encrypted/.test(e.message))
|
|
52
|
+
throw e;
|
|
53
|
+
}
|
|
54
|
+
const { code } = await run(factory, input, ["--requires-password", IN]);
|
|
55
|
+
return code === 3 ? "restricted" : "password";
|
|
56
|
+
}
|
|
57
|
+
function randomSecret() {
|
|
58
|
+
const bytes = new Uint8Array(24);
|
|
59
|
+
globalThis.crypto.getRandomValues(bytes);
|
|
60
|
+
return Array.from(bytes, (b) => b.toString(16).padStart(2, "0")).join("");
|
|
61
|
+
}
|
|
62
|
+
/** Cifra con AES-256, el estándar que abren Acrobat, Chrome, Preview y cualquier lector actual. */
|
|
63
|
+
export async function protectPdf(factory, input, opts) {
|
|
64
|
+
if (!opts.password)
|
|
65
|
+
throw new Error("La contraseña no puede estar vacía.");
|
|
66
|
+
const { allowPrint = true, allowCopy = true, allowEdit = true } = opts;
|
|
67
|
+
const { code, output } = await run(factory, input, [
|
|
68
|
+
// Forma con nombre: una contraseña que empiece por «--» no se confunde con una opción.
|
|
69
|
+
"--encrypt",
|
|
70
|
+
`--user-password=${opts.password}`,
|
|
71
|
+
`--owner-password=${opts.ownerPassword || randomSecret()}`,
|
|
72
|
+
"--bits=256",
|
|
73
|
+
`--print=${allowPrint ? "full" : "none"}`,
|
|
74
|
+
`--extract=${allowCopy ? "y" : "n"}`,
|
|
75
|
+
`--modify=${allowEdit ? "all" : "none"}`,
|
|
76
|
+
"--",
|
|
77
|
+
IN,
|
|
78
|
+
OUT,
|
|
79
|
+
]);
|
|
80
|
+
// 3 = terminó con avisos (PDF algo irregular) pero la salida es válida.
|
|
81
|
+
if ((code !== 0 && code !== 3) || !output) {
|
|
82
|
+
const state = await lockState(factory, input).catch(() => "none");
|
|
83
|
+
if (state !== "none")
|
|
84
|
+
throw new Error("Este PDF ya tiene contraseña. Quítala primero y vuelve a protegerlo.");
|
|
85
|
+
throw new Error("No se pudo cifrar el PDF.");
|
|
86
|
+
}
|
|
87
|
+
return output;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Quita el cifrado. Sin contraseña funciona con los PDF «restringidos» (se
|
|
91
|
+
* abren pero no dejan copiar o imprimir), que son la mayoría de los casos.
|
|
92
|
+
*/
|
|
93
|
+
export async function unlockPdf(factory, input, password = "") {
|
|
94
|
+
const { code, output } = await run(factory, input, [`--password=${password}`, "--decrypt", IN, OUT]);
|
|
95
|
+
if ((code === 0 || code === 3) && output)
|
|
96
|
+
return output;
|
|
97
|
+
if ((await lockState(factory, input)) === "password")
|
|
98
|
+
throw new WrongPasswordError();
|
|
99
|
+
throw new Error("No se pudo quitar la contraseña del PDF.");
|
|
100
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@sirdaspdf/core",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.2.0",
|
|
4
4
|
"description": "Sırdaş core: PDF operations, PDF→Markdown for AI and local PII anonymization. Runs in the browser and in Node; files never leave the machine.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pdf",
|
|
@@ -66,6 +66,7 @@
|
|
|
66
66
|
}
|
|
67
67
|
},
|
|
68
68
|
"devDependencies": {
|
|
69
|
+
"@neslinesli93/qpdf-wasm": "0.3.0",
|
|
69
70
|
"pdfjs-dist": "4.10.38",
|
|
70
71
|
"typescript": "5.9.2"
|
|
71
72
|
}
|