@sirdaspdf/core 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agentchunks.d.ts +18 -0
- package/dist/agentchunks.js +85 -0
- package/dist/anonymize.js +55 -7
- package/dist/classify.d.ts +12 -0
- package/dist/classify.js +89 -0
- package/dist/dataset.d.ts +25 -1
- package/dist/dataset.js +68 -2
- package/dist/extract.d.ts +16 -0
- package/dist/extract.js +103 -0
- package/dist/formats.d.ts +30 -0
- package/dist/formats.js +521 -0
- package/dist/forms.d.ts +19 -0
- package/dist/forms.js +57 -0
- package/dist/index.d.ts +9 -0
- package/dist/index.js +9 -0
- package/dist/locate.d.ts +13 -0
- package/dist/locate.js +35 -0
- package/dist/markdown.d.ts +16 -2
- package/dist/markdown.js +25 -6
- package/dist/ocr.js +34 -4
- package/dist/pages.d.ts +24 -0
- package/dist/pages.js +92 -0
- package/dist/security.d.ts +40 -0
- package/dist/security.js +100 -0
- package/dist/segment.d.ts +10 -0
- package/dist/segment.js +46 -0
- package/package.json +3 -1
package/dist/locate.js
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
const norm = (s) => s.toLowerCase().normalize("NFD").replace(/\p{Diacritic}/gu, "").replace(/[^\p{L}\p{N}]+/gu, " ").trim();
|
|
2
|
+
/**
|
|
3
|
+
* Ubica `text` entre las líneas extraídas. Tolera diferencias de espacios,
|
|
4
|
+
* tildes y puntuación, y citas que ocupan hasta tres líneas seguidas.
|
|
5
|
+
*/
|
|
6
|
+
export function locateText(lines, text, page) {
|
|
7
|
+
const needle = norm(text);
|
|
8
|
+
if (needle.length < 3)
|
|
9
|
+
return null;
|
|
10
|
+
const pool = lines.map((l, i) => ({ l, i })).filter(({ l }) => l.bbox && (!page || l.page === page));
|
|
11
|
+
for (let span = 1; span <= 3; span++) {
|
|
12
|
+
for (let k = 0; k + span <= pool.length; k++) {
|
|
13
|
+
const group = pool.slice(k, k + span);
|
|
14
|
+
if (group.some((g) => g.l.page !== group[0].l.page))
|
|
15
|
+
continue;
|
|
16
|
+
if (span > 1 && group.some((g, j) => j > 0 && g.i !== group[j - 1].i + 1))
|
|
17
|
+
continue;
|
|
18
|
+
const hay = norm(group.map((g) => g.l.text).join(" "));
|
|
19
|
+
if (!hay.includes(needle))
|
|
20
|
+
continue;
|
|
21
|
+
const boxes = group.map((g) => g.l.bbox);
|
|
22
|
+
return {
|
|
23
|
+
page: group[0].l.page,
|
|
24
|
+
bbox: [
|
|
25
|
+
Math.min(...boxes.map((b) => b[0])),
|
|
26
|
+
Math.min(...boxes.map((b) => b[1])),
|
|
27
|
+
Math.max(...boxes.map((b) => b[2])),
|
|
28
|
+
Math.max(...boxes.map((b) => b[3])),
|
|
29
|
+
],
|
|
30
|
+
lines: span,
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
return null;
|
|
35
|
+
}
|
package/dist/markdown.d.ts
CHANGED
|
@@ -8,14 +8,28 @@ export interface Line {
|
|
|
8
8
|
bold: boolean;
|
|
9
9
|
/** Posición vertical relativa: 0 = pie de página, 1 = parte superior. */
|
|
10
10
|
rel: number;
|
|
11
|
+
/**
|
|
12
|
+
* Caja de la línea normalizada 0–1 con origen arriba a la izquierda
|
|
13
|
+
* [x0, y0, x1, y1]: lo que necesita un visor para resaltar una cita sin
|
|
14
|
+
* saber nada de coordenadas PDF. Ausente en líneas que no vienen de un PDF.
|
|
15
|
+
*/
|
|
16
|
+
bbox?: [number, number, number, number];
|
|
11
17
|
}
|
|
12
18
|
type Progress = (ratio: number) => void;
|
|
13
19
|
/** Extrae las líneas de texto con tamaño y posición de un documento abierto. */
|
|
14
20
|
export declare function extractLines(pdf: PDFDocumentProxy, onProgress?: Progress): Promise<Line[]>;
|
|
15
21
|
/** Convierte líneas ya extraídas en Markdown (encabezados, listas, párrafos, campos). */
|
|
16
|
-
export
|
|
22
|
+
export interface MarkdownOptions {
|
|
23
|
+
/**
|
|
24
|
+
* Inserta `<!-- page N -->` donde empieza cada página. Invisible al renderizar,
|
|
25
|
+
* pero permite que los trozos sepan de qué página salen y el agente cite.
|
|
26
|
+
*/
|
|
27
|
+
pageMarkers?: boolean;
|
|
28
|
+
}
|
|
29
|
+
export declare const PAGE_MARKER: RegExp;
|
|
30
|
+
export declare function linesToMarkdown(lines: Line[], opts?: MarkdownOptions): string;
|
|
17
31
|
/** Documento PDF.js abierto → Markdown. */
|
|
18
|
-
export declare function pdfToMarkdown(pdf: PDFDocumentProxy, onProgress?: Progress): Promise<string>;
|
|
32
|
+
export declare function pdfToMarkdown(pdf: PDFDocumentProxy, onProgress?: Progress, opts?: MarkdownOptions): Promise<string>;
|
|
19
33
|
/** Estimación rápida de tokens (~4 caracteres por token). */
|
|
20
34
|
export declare function estimateTokens(text: string): number;
|
|
21
35
|
export {};
|
package/dist/markdown.js
CHANGED
|
@@ -23,7 +23,9 @@ export async function extractLines(pdf, onProgress) {
|
|
|
23
23
|
for (let p = 1; p <= pdf.numPages; p++) {
|
|
24
24
|
const page = await pdf.getPage(p);
|
|
25
25
|
const content = await page.getTextContent();
|
|
26
|
-
const
|
|
26
|
+
const viewport = page.getViewport({ scale: 1 });
|
|
27
|
+
const pageHeight = viewport.height || 792;
|
|
28
|
+
const pageWidth = viewport.width || 612;
|
|
27
29
|
const styles = content.styles;
|
|
28
30
|
const rows = new Map();
|
|
29
31
|
for (const it of content.items) {
|
|
@@ -55,7 +57,17 @@ export async function extractLines(pdf, onProgress) {
|
|
|
55
57
|
continue;
|
|
56
58
|
const size = Math.max(...row.map((r) => r.size));
|
|
57
59
|
const bold = row.filter((r) => r.bold).reduce((s, r) => s + r.str.length, 0) > text.length * 0.6;
|
|
58
|
-
|
|
60
|
+
const x0 = row[0].x;
|
|
61
|
+
const x1 = Math.max(...row.map((r) => r.x + r.w));
|
|
62
|
+
const r4 = (n) => Math.round(Math.min(1, Math.max(0, n)) * 10000) / 10000;
|
|
63
|
+
// y es la línea base: la caja sube un cuerpo de letra y baja lo que ocupan los descendentes.
|
|
64
|
+
const bbox = [
|
|
65
|
+
r4(x0 / pageWidth),
|
|
66
|
+
r4(1 - (y + size * 0.9) / pageHeight),
|
|
67
|
+
r4(x1 / pageWidth),
|
|
68
|
+
r4(1 - (y - size * 0.25) / pageHeight),
|
|
69
|
+
];
|
|
70
|
+
lines.push({ text, size, y, x: x0, page: p, bold, rel: y / pageHeight, bbox });
|
|
59
71
|
}
|
|
60
72
|
page.cleanup();
|
|
61
73
|
onProgress?.(p / pdf.numPages);
|
|
@@ -86,8 +98,8 @@ function removeRepeated(lines, pages) {
|
|
|
86
98
|
return true;
|
|
87
99
|
});
|
|
88
100
|
}
|
|
89
|
-
|
|
90
|
-
export function linesToMarkdown(lines) {
|
|
101
|
+
export const PAGE_MARKER = /^<!-- page (\d+) -->$/;
|
|
102
|
+
export function linesToMarkdown(lines, opts = {}) {
|
|
91
103
|
if (!lines.length)
|
|
92
104
|
return "";
|
|
93
105
|
const body = mode(lines.map((l) => l.size));
|
|
@@ -105,7 +117,14 @@ export function linesToMarkdown(lines) {
|
|
|
105
117
|
out.push("");
|
|
106
118
|
inKv = false;
|
|
107
119
|
};
|
|
120
|
+
let markedPage = -1;
|
|
108
121
|
for (const l of lines) {
|
|
122
|
+
if (opts.pageMarkers && l.page !== markedPage) {
|
|
123
|
+
flush();
|
|
124
|
+
closeKv();
|
|
125
|
+
out.push(`<!-- page ${l.page} -->`, "");
|
|
126
|
+
markedPage = l.page;
|
|
127
|
+
}
|
|
109
128
|
const ratio = l.size / body;
|
|
110
129
|
const short = l.text.length < 90 && !/[.,;:]$/.test(l.text);
|
|
111
130
|
let heading = 0;
|
|
@@ -168,8 +187,8 @@ export function linesToMarkdown(lines) {
|
|
|
168
187
|
.trim();
|
|
169
188
|
}
|
|
170
189
|
/** Documento PDF.js abierto → Markdown. */
|
|
171
|
-
export async function pdfToMarkdown(pdf, onProgress) {
|
|
172
|
-
return linesToMarkdown(await extractLines(pdf, onProgress));
|
|
190
|
+
export async function pdfToMarkdown(pdf, onProgress, opts) {
|
|
191
|
+
return linesToMarkdown(await extractLines(pdf, onProgress), opts);
|
|
173
192
|
}
|
|
174
193
|
/** Estimación rápida de tokens (~4 caracteres por token). */
|
|
175
194
|
export function estimateTokens(text) {
|
package/dist/ocr.js
CHANGED
|
@@ -1,14 +1,40 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* El alto de una caja varía según tenga tildes o letras con cola ("Hola" vs
|
|
3
|
-
* "Hpqy"), así que agrupamos tamaños parecidos en cubos.
|
|
4
|
-
* de "tamaño de cuerpo" del Markdown vería decenas de tamaños distintos y no
|
|
5
|
-
* distinguiría un encabezado de un párrafo.
|
|
3
|
+
* "Hpqy"), así que agrupamos tamaños parecidos en cubos.
|
|
6
4
|
*/
|
|
7
5
|
const BUCKET = 2;
|
|
8
6
|
const bucket = (n) => Math.max(BUCKET, Math.round(n / BUCKET) * BUCKET);
|
|
7
|
+
/**
|
|
8
|
+
* Cuánto más alta que el cuerpo tiene que ser una línea para considerarla
|
|
9
|
+
* encabezado.
|
|
10
|
+
*
|
|
11
|
+
* El alto de una caja de OCR NO es el tamaño de la letra: es la envolvente de
|
|
12
|
+
* lo que se dibujó. En una página con todo el cuerpo al mismo tamaño, medimos
|
|
13
|
+
* 9,1 pt en "Cedula de ciudadania" y 12,0 pt en "Correo: juan.perez@…", solo
|
|
14
|
+
* porque la segunda tiene jota y arroba. Con un umbral bajo, cada línea con
|
|
15
|
+
* una cola se convertía en un «###», el documento salía troceado en secciones
|
|
16
|
+
* falsas y los párrafos se perdían.
|
|
17
|
+
*
|
|
18
|
+
* 1,35 es deliberadamente conservador: un encabezado sutil (1,15× el cuerpo)
|
|
19
|
+
* es indistinguible del ruido y se deja como párrafo. Perder un encabezado
|
|
20
|
+
* discreto molesta; inventar veinte destruye la estructura.
|
|
21
|
+
*/
|
|
22
|
+
const HEADING_RATIO = 1.35;
|
|
23
|
+
/** Mediana: resiste que una línea suelta salga enorme o diminuta. */
|
|
24
|
+
function median(values) {
|
|
25
|
+
if (values.length === 0)
|
|
26
|
+
return 0;
|
|
27
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
28
|
+
const mid = sorted.length >> 1;
|
|
29
|
+
return sorted.length % 2 ? sorted[mid] : (sorted[mid - 1] + sorted[mid]) / 2;
|
|
30
|
+
}
|
|
9
31
|
/** Convierte las cajas de una página OCR en líneas del pipeline de Markdown. */
|
|
10
32
|
export function ocrPageToLines({ page, heightPx, scale, boxes }) {
|
|
11
33
|
const heightPt = heightPx / scale;
|
|
34
|
+
// El tamaño del cuerpo de ESTA página, no un valor absoluto: un escaneo a
|
|
35
|
+
// 150 dpi y otro a 300 tienen que producir el mismo Markdown.
|
|
36
|
+
const usable = boxes.filter((b) => b.text.trim() !== "");
|
|
37
|
+
const bodyPt = median(usable.map((b) => (b.y1 - b.y0) / scale));
|
|
12
38
|
const lines = [];
|
|
13
39
|
for (const b of boxes) {
|
|
14
40
|
const text = b.text.replace(/\s+/g, " ").trim();
|
|
@@ -17,9 +43,13 @@ export function ocrPageToLines({ page, heightPx, scale, boxes }) {
|
|
|
17
43
|
// PDF mide la y desde abajo; el OCR desde arriba. Invertimos para que el
|
|
18
44
|
// resto del pipeline (saltos de bloque, encabezados/pies) funcione igual.
|
|
19
45
|
const y = (heightPx - b.y1) / scale;
|
|
46
|
+
// Todo lo que no sea claramente más alto que el cuerpo SE IGUALA al cuerpo.
|
|
47
|
+
// Si no, el ruido de las cajas decide los encabezados.
|
|
48
|
+
const raw = (b.y1 - b.y0) / scale;
|
|
49
|
+
const size = bodyPt > 0 && raw < bodyPt * HEADING_RATIO ? bucket(bodyPt) : bucket(raw);
|
|
20
50
|
lines.push({
|
|
21
51
|
text,
|
|
22
|
-
size
|
|
52
|
+
size,
|
|
23
53
|
y,
|
|
24
54
|
x: b.x0 / scale,
|
|
25
55
|
page,
|
package/dist/pages.d.ts
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
/** Quita las páginas indicadas (base 0). Negarse a vaciar el documento evita un PDF inválido. */
|
|
2
|
+
export declare function deletePages(buf: ArrayBuffer, remove: number[]): Promise<Uint8Array>;
|
|
3
|
+
/** Un PDF nuevo con solo esas páginas, en ese orden. */
|
|
4
|
+
export declare function extractPages(buf: ArrayBuffer, indices: number[]): Promise<Uint8Array>;
|
|
5
|
+
/** Gira páginas concretas (o todas si `indices` va vacío). */
|
|
6
|
+
export declare function rotatePages(buf: ArrayBuffer, angle: 90 | 180 | 270, indices?: number[]): Promise<Uint8Array>;
|
|
7
|
+
export type Corner = "bottom-center" | "bottom-right" | "bottom-left" | "top-center" | "top-right" | "top-left";
|
|
8
|
+
export interface PageNumberOptions {
|
|
9
|
+
position?: Corner;
|
|
10
|
+
/** `{n}` es la página y `{total}` el total. */
|
|
11
|
+
format?: string;
|
|
12
|
+
startAt?: number;
|
|
13
|
+
fontSize?: number;
|
|
14
|
+
/** Saltar las N primeras páginas (portadas). */
|
|
15
|
+
skipFirst?: number;
|
|
16
|
+
}
|
|
17
|
+
export declare function addPageNumbers(buf: ArrayBuffer, opts?: PageNumberOptions): Promise<Uint8Array>;
|
|
18
|
+
export interface WatermarkOptions {
|
|
19
|
+
opacity?: number;
|
|
20
|
+
fontSize?: number;
|
|
21
|
+
/** Diagonal por defecto, que es lo que se reconoce como marca de agua. */
|
|
22
|
+
angle?: number;
|
|
23
|
+
}
|
|
24
|
+
export declare function addWatermark(buf: ArrayBuffer, text: string, opts?: WatermarkOptions): Promise<Uint8Array>;
|
package/dist/pages.js
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
// Operaciones de página que tienen todas las suites de la competencia: quitar,
|
|
2
|
+
// extraer, girar, numerar y marca de agua. Todas con pdf-lib, en local.
|
|
3
|
+
import { PDFDocument, StandardFonts, degrees, rgb } from "pdf-lib";
|
|
4
|
+
async function copyOf(buf, indices) {
|
|
5
|
+
const src = await PDFDocument.load(buf);
|
|
6
|
+
const out = await PDFDocument.create();
|
|
7
|
+
for (const p of await out.copyPages(src, indices))
|
|
8
|
+
out.addPage(p);
|
|
9
|
+
return out.save({ useObjectStreams: true });
|
|
10
|
+
}
|
|
11
|
+
function assertIndices(indices, total) {
|
|
12
|
+
for (const i of indices) {
|
|
13
|
+
if (!Number.isInteger(i) || i < 0 || i >= total)
|
|
14
|
+
throw new Error(`Página fuera del documento: ${i + 1}`);
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
/** Quita las páginas indicadas (base 0). Negarse a vaciar el documento evita un PDF inválido. */
|
|
18
|
+
export async function deletePages(buf, remove) {
|
|
19
|
+
const total = (await PDFDocument.load(buf)).getPageCount();
|
|
20
|
+
assertIndices(remove, total);
|
|
21
|
+
const drop = new Set(remove);
|
|
22
|
+
const keep = Array.from({ length: total }, (_, i) => i).filter((i) => !drop.has(i));
|
|
23
|
+
if (!keep.length)
|
|
24
|
+
throw new Error("No se pueden quitar todas las páginas: el PDF quedaría vacío.");
|
|
25
|
+
return copyOf(buf, keep);
|
|
26
|
+
}
|
|
27
|
+
/** Un PDF nuevo con solo esas páginas, en ese orden. */
|
|
28
|
+
export async function extractPages(buf, indices) {
|
|
29
|
+
const total = (await PDFDocument.load(buf)).getPageCount();
|
|
30
|
+
assertIndices(indices, total);
|
|
31
|
+
if (!indices.length)
|
|
32
|
+
throw new Error("Indica al menos una página.");
|
|
33
|
+
return copyOf(buf, indices);
|
|
34
|
+
}
|
|
35
|
+
/** Gira páginas concretas (o todas si `indices` va vacío). */
|
|
36
|
+
export async function rotatePages(buf, angle, indices = []) {
|
|
37
|
+
const doc = await PDFDocument.load(buf);
|
|
38
|
+
const pages = doc.getPages();
|
|
39
|
+
assertIndices(indices, pages.length);
|
|
40
|
+
const targets = indices.length ? indices : pages.map((_, i) => i);
|
|
41
|
+
for (const i of targets)
|
|
42
|
+
pages[i].setRotation(degrees((pages[i].getRotation().angle + angle) % 360));
|
|
43
|
+
return doc.save({ useObjectStreams: true });
|
|
44
|
+
}
|
|
45
|
+
/** Helvetica estándar solo codifica WinAnsi: mejor un error claro que un PDF corrupto. */
|
|
46
|
+
function safeText(font, text) {
|
|
47
|
+
try {
|
|
48
|
+
font.encodeText(text);
|
|
49
|
+
return text;
|
|
50
|
+
}
|
|
51
|
+
catch {
|
|
52
|
+
throw new Error(`El texto «${text}» tiene caracteres que la fuente estándar del PDF no admite.`);
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
export async function addPageNumbers(buf, opts = {}) {
|
|
56
|
+
const { position = "bottom-center", format = "{n}", startAt = 1, fontSize = 10, skipFirst = 0 } = opts;
|
|
57
|
+
const doc = await PDFDocument.load(buf);
|
|
58
|
+
const font = await doc.embedFont(StandardFonts.Helvetica);
|
|
59
|
+
const pages = doc.getPages();
|
|
60
|
+
const numbered = pages.length - skipFirst;
|
|
61
|
+
pages.forEach((page, i) => {
|
|
62
|
+
if (i < skipFirst)
|
|
63
|
+
return;
|
|
64
|
+
const label = safeText(font, format.replace("{n}", String(i - skipFirst + startAt)).replace("{total}", String(numbered + startAt - 1)));
|
|
65
|
+
const { width, height } = page.getSize();
|
|
66
|
+
const w = font.widthOfTextAtSize(label, fontSize);
|
|
67
|
+
const margin = 28;
|
|
68
|
+
const x = position.endsWith("left") ? margin : position.endsWith("right") ? width - margin - w : (width - w) / 2;
|
|
69
|
+
const y = position.startsWith("top") ? height - margin : margin - fontSize / 2;
|
|
70
|
+
page.drawText(label, { x, y, size: fontSize, font, color: rgb(0.25, 0.25, 0.25) });
|
|
71
|
+
});
|
|
72
|
+
return doc.save({ useObjectStreams: true });
|
|
73
|
+
}
|
|
74
|
+
export async function addWatermark(buf, text, opts = {}) {
|
|
75
|
+
if (!text.trim())
|
|
76
|
+
throw new Error("La marca de agua necesita un texto.");
|
|
77
|
+
const { opacity = 0.15, angle = 45 } = opts;
|
|
78
|
+
const doc = await PDFDocument.load(buf);
|
|
79
|
+
const font = await doc.embedFont(StandardFonts.HelveticaBold);
|
|
80
|
+
const label = safeText(font, text.trim());
|
|
81
|
+
for (const page of doc.getPages()) {
|
|
82
|
+
const { width, height } = page.getSize();
|
|
83
|
+
const size = opts.fontSize ?? Math.min(96, (Math.hypot(width, height) * 0.7) / Math.max(label.length * 0.6, 1));
|
|
84
|
+
const w = font.widthOfTextAtSize(label, size);
|
|
85
|
+
const rad = (angle * Math.PI) / 180;
|
|
86
|
+
// Centrar el texto girado: se desplaza el origen media anchura sobre su propio eje.
|
|
87
|
+
const x = width / 2 - (Math.cos(rad) * w) / 2 + (Math.sin(rad) * size) / 3;
|
|
88
|
+
const y = height / 2 - (Math.sin(rad) * w) / 2 - (Math.cos(rad) * size) / 3;
|
|
89
|
+
page.drawText(label, { x, y, size, font, color: rgb(0.55, 0.1, 0.25), opacity, rotate: degrees(angle) });
|
|
90
|
+
}
|
|
91
|
+
return doc.save({ useObjectStreams: true });
|
|
92
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/** Lo mínimo que usamos del módulo Emscripten de qpdf. */
|
|
2
|
+
export interface QpdfModule {
|
|
3
|
+
FS: {
|
|
4
|
+
writeFile(path: string, data: Uint8Array): void;
|
|
5
|
+
readFile(path: string): Uint8Array;
|
|
6
|
+
};
|
|
7
|
+
callMain(args: string[]): number;
|
|
8
|
+
}
|
|
9
|
+
export type QpdfFactory = () => Promise<QpdfModule>;
|
|
10
|
+
export type LockState =
|
|
11
|
+
/** Sin cifrado. */
|
|
12
|
+
"none"
|
|
13
|
+
/** Cifrado solo con contraseña de propietario: se abre, pero con permisos limitados (imprimir, copiar…). */
|
|
14
|
+
| "restricted"
|
|
15
|
+
/** Hace falta contraseña para abrirlo. */
|
|
16
|
+
| "password";
|
|
17
|
+
export declare class WrongPasswordError extends Error {
|
|
18
|
+
constructor();
|
|
19
|
+
}
|
|
20
|
+
export declare function lockState(factory: QpdfFactory, input: Uint8Array): Promise<LockState>;
|
|
21
|
+
export interface ProtectOptions {
|
|
22
|
+
/** Contraseña para abrir. */
|
|
23
|
+
password: string;
|
|
24
|
+
/**
|
|
25
|
+
* Contraseña de propietario (para cambiar permisos). Si no se da, se genera
|
|
26
|
+
* una aleatoria: reutilizar la de apertura haría que cualquiera que abre el
|
|
27
|
+
* archivo pudiera también quitarle las restricciones.
|
|
28
|
+
*/
|
|
29
|
+
ownerPassword?: string;
|
|
30
|
+
allowPrint?: boolean;
|
|
31
|
+
allowCopy?: boolean;
|
|
32
|
+
allowEdit?: boolean;
|
|
33
|
+
}
|
|
34
|
+
/** Cifra con AES-256, el estándar que abren Acrobat, Chrome, Preview y cualquier lector actual. */
|
|
35
|
+
export declare function protectPdf(factory: QpdfFactory, input: Uint8Array, opts: ProtectOptions): Promise<Uint8Array>;
|
|
36
|
+
/**
|
|
37
|
+
* Quita el cifrado. Sin contraseña funciona con los PDF «restringidos» (se
|
|
38
|
+
* abren pero no dejan copiar o imprimir), que son la mayoría de los casos.
|
|
39
|
+
*/
|
|
40
|
+
export declare function unlockPdf(factory: QpdfFactory, input: Uint8Array, password?: string): Promise<Uint8Array>;
|
package/dist/security.js
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
// Poner y quitar contraseñas con qpdf compilado a WebAssembly. core no depende
|
|
2
|
+
// de qpdf: quien llama inyecta el módulo, porque en el navegador el .wasm se
|
|
3
|
+
// sirve desde nuestro propio origen y en Node se lee del disco. Así core sigue
|
|
4
|
+
// cargando sin 1,3 MB que la mayoría de herramientas no necesita.
|
|
5
|
+
import { PDFDocument } from "pdf-lib";
|
|
6
|
+
export class WrongPasswordError extends Error {
|
|
7
|
+
constructor() {
|
|
8
|
+
super("Contraseña incorrecta.");
|
|
9
|
+
this.name = "WrongPasswordError";
|
|
10
|
+
}
|
|
11
|
+
}
|
|
12
|
+
const IN = "/in.pdf";
|
|
13
|
+
const OUT = "/out.pdf";
|
|
14
|
+
/** qpdf termina con `exit()`: Emscripten lo lanza como excepción con `status`. */
|
|
15
|
+
function exitCode(q, args) {
|
|
16
|
+
try {
|
|
17
|
+
return q.callMain(args);
|
|
18
|
+
}
|
|
19
|
+
catch (e) {
|
|
20
|
+
const status = e.status;
|
|
21
|
+
if (typeof status === "number")
|
|
22
|
+
return status;
|
|
23
|
+
throw e;
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
async function run(factory, input, args) {
|
|
27
|
+
// Un módulo por operación: callMain deja estado global y reutilizarlo falla
|
|
28
|
+
// de forma intermitente en la segunda llamada.
|
|
29
|
+
const q = await factory();
|
|
30
|
+
q.FS.writeFile(IN, input);
|
|
31
|
+
const code = exitCode(q, args);
|
|
32
|
+
let output = null;
|
|
33
|
+
try {
|
|
34
|
+
output = q.FS.readFile(OUT);
|
|
35
|
+
}
|
|
36
|
+
catch {
|
|
37
|
+
output = null;
|
|
38
|
+
}
|
|
39
|
+
return { code, output };
|
|
40
|
+
}
|
|
41
|
+
export async function lockState(factory, input) {
|
|
42
|
+
// Esta build de qpdf sale con 2 tanto si el PDF no está cifrado como si pide
|
|
43
|
+
// una contraseña que no tiene, así que pdf-lib decide si hay cifrado y qpdf
|
|
44
|
+
// solo desempata: --requires-password da 3 cuando se abre sin contraseña.
|
|
45
|
+
try {
|
|
46
|
+
await PDFDocument.load(input, { updateMetadata: false });
|
|
47
|
+
return "none";
|
|
48
|
+
}
|
|
49
|
+
catch (e) {
|
|
50
|
+
// Por nombre y no con instanceof: en Node pueden convivir la copia CJS y la ESM de pdf-lib.
|
|
51
|
+
if (e.name !== "EncryptedPDFError" && !/is encrypted/.test(e.message))
|
|
52
|
+
throw e;
|
|
53
|
+
}
|
|
54
|
+
const { code } = await run(factory, input, ["--requires-password", IN]);
|
|
55
|
+
return code === 3 ? "restricted" : "password";
|
|
56
|
+
}
|
|
57
|
+
function randomSecret() {
|
|
58
|
+
const bytes = new Uint8Array(24);
|
|
59
|
+
globalThis.crypto.getRandomValues(bytes);
|
|
60
|
+
return Array.from(bytes, (b) => b.toString(16).padStart(2, "0")).join("");
|
|
61
|
+
}
|
|
62
|
+
/** Cifra con AES-256, el estándar que abren Acrobat, Chrome, Preview y cualquier lector actual. */
|
|
63
|
+
export async function protectPdf(factory, input, opts) {
|
|
64
|
+
if (!opts.password)
|
|
65
|
+
throw new Error("La contraseña no puede estar vacía.");
|
|
66
|
+
const { allowPrint = true, allowCopy = true, allowEdit = true } = opts;
|
|
67
|
+
const { code, output } = await run(factory, input, [
|
|
68
|
+
// Forma con nombre: una contraseña que empiece por «--» no se confunde con una opción.
|
|
69
|
+
"--encrypt",
|
|
70
|
+
`--user-password=${opts.password}`,
|
|
71
|
+
`--owner-password=${opts.ownerPassword || randomSecret()}`,
|
|
72
|
+
"--bits=256",
|
|
73
|
+
`--print=${allowPrint ? "full" : "none"}`,
|
|
74
|
+
`--extract=${allowCopy ? "y" : "n"}`,
|
|
75
|
+
`--modify=${allowEdit ? "all" : "none"}`,
|
|
76
|
+
"--",
|
|
77
|
+
IN,
|
|
78
|
+
OUT,
|
|
79
|
+
]);
|
|
80
|
+
// 3 = terminó con avisos (PDF algo irregular) pero la salida es válida.
|
|
81
|
+
if ((code !== 0 && code !== 3) || !output) {
|
|
82
|
+
const state = await lockState(factory, input).catch(() => "none");
|
|
83
|
+
if (state !== "none")
|
|
84
|
+
throw new Error("Este PDF ya tiene contraseña. Quítala primero y vuelve a protegerlo.");
|
|
85
|
+
throw new Error("No se pudo cifrar el PDF.");
|
|
86
|
+
}
|
|
87
|
+
return output;
|
|
88
|
+
}
|
|
89
|
+
/**
|
|
90
|
+
* Quita el cifrado. Sin contraseña funciona con los PDF «restringidos» (se
|
|
91
|
+
* abren pero no dejan copiar o imprimir), que son la mayoría de los casos.
|
|
92
|
+
*/
|
|
93
|
+
export async function unlockPdf(factory, input, password = "") {
|
|
94
|
+
const { code, output } = await run(factory, input, [`--password=${password}`, "--decrypt", IN, OUT]);
|
|
95
|
+
if ((code === 0 || code === 3) && output)
|
|
96
|
+
return output;
|
|
97
|
+
if ((await lockState(factory, input)) === "password")
|
|
98
|
+
throw new WrongPasswordError();
|
|
99
|
+
throw new Error("No se pudo quitar la contraseña del PDF.");
|
|
100
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { type DocType } from "./classify.js";
|
|
2
|
+
export interface Segment {
|
|
3
|
+
/** Páginas base 1, inclusivas. */
|
|
4
|
+
from: number;
|
|
5
|
+
to: number;
|
|
6
|
+
type: DocType;
|
|
7
|
+
/** Por qué empieza aquí un documento nuevo (vacío en el primero). */
|
|
8
|
+
reasons: string[];
|
|
9
|
+
}
|
|
10
|
+
export declare function segmentPages(pages: string[]): Segment[];
|
package/dist/segment.js
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
// Separar un PDF que en realidad son varios documentos pegados: el lote que
|
|
2
|
+
// sale del escáner de la recepción, un expediente con factura + historia +
|
|
3
|
+
// autorización. Sin modelo: se buscan los cortes que una persona vería.
|
|
4
|
+
import { classifyDocument } from "./classify.js";
|
|
5
|
+
/** «Página 1 de 5», «Pág. 1/3», «Page 1 of 2», «1 de 4» al pie. */
|
|
6
|
+
const FIRST_PAGE = /\b(?:p[aá]g(?:ina)?\.?|page)\s*1\s*(?:de|of|\/)\s*\d+\b/i;
|
|
7
|
+
const OTHER_PAGE = /\b(?:p[aá]g(?:ina)?\.?|page)\s*([2-9]|\d{2,})\s*(?:de|of|\/)\s*\d+\b/i;
|
|
8
|
+
/** Encabezados con los que suele abrir un documento. */
|
|
9
|
+
const OPENING = /^(?:#+\s*)?(factura(?:\s+electr[oó]nica)?(?:\s+de\s+venta)?|historia\s+cl[ií]nica|epicrisis|certificado|certifica(?:ci[oó]n)?|contrato\s+de|registro\s+[uú]nico\s+tributario|extracto|autorizaci[oó]n|orden\s+m[eé]dica|f[oó]rmula\s+m[eé]dica|acta\s+de|poder\s+especial|constancia|cuenta\s+de\s+cobro|resultados?\s+de\s+laboratorio)\b/im;
|
|
10
|
+
export function segmentPages(pages) {
|
|
11
|
+
if (!pages.length)
|
|
12
|
+
return [];
|
|
13
|
+
const kinds = pages.map((t) => classifyDocument(t));
|
|
14
|
+
const segments = [];
|
|
15
|
+
let current = { from: 1, to: 1, type: kinds[0].type, reasons: [] };
|
|
16
|
+
for (let i = 1; i < pages.length; i++) {
|
|
17
|
+
const text = pages[i];
|
|
18
|
+
const head = text.split("\n").slice(0, 8).join("\n");
|
|
19
|
+
const reasons = [];
|
|
20
|
+
const prevType = current.type;
|
|
21
|
+
const k = kinds[i];
|
|
22
|
+
if (FIRST_PAGE.test(text))
|
|
23
|
+
reasons.push("numeración reinicia en «página 1»");
|
|
24
|
+
if (OPENING.test(head))
|
|
25
|
+
reasons.push(`abre con «${head.match(OPENING)[1]}»`);
|
|
26
|
+
// Un cambio de tipo solo cuenta con señales fuertes en ambos lados: una
|
|
27
|
+
// página de anexos sin palabras clave no debe partir un contrato.
|
|
28
|
+
if (k.type !== "desconocido" && prevType !== "desconocido" && k.type !== prevType && k.confidence >= 0.6) {
|
|
29
|
+
reasons.push(`cambia de ${prevType} a ${k.type}`);
|
|
30
|
+
}
|
|
31
|
+
// «Página 3 de 5» contradice un corte: sigue el mismo documento.
|
|
32
|
+
const continues = OTHER_PAGE.test(text) && !FIRST_PAGE.test(text);
|
|
33
|
+
const cut = !continues && (reasons.length >= 2 || reasons.some((r) => r.startsWith("numeración")));
|
|
34
|
+
if (cut) {
|
|
35
|
+
segments.push(current);
|
|
36
|
+
current = { from: i + 1, to: i + 1, type: k.type, reasons };
|
|
37
|
+
}
|
|
38
|
+
else {
|
|
39
|
+
current.to = i + 1;
|
|
40
|
+
if (current.type === "desconocido" && k.type !== "desconocido")
|
|
41
|
+
current.type = k.type;
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
segments.push(current);
|
|
45
|
+
return segments;
|
|
46
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@sirdaspdf/core",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"description": "Sırdaş core: PDF operations, PDF→Markdown for AI and local PII anonymization. Runs in the browser and in Node; files never leave the machine.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pdf",
|
|
@@ -55,6 +55,7 @@
|
|
|
55
55
|
"prepublishOnly": "npm run test"
|
|
56
56
|
},
|
|
57
57
|
"dependencies": {
|
|
58
|
+
"fflate": "0.8.3",
|
|
58
59
|
"pdf-lib": "1.17.1"
|
|
59
60
|
},
|
|
60
61
|
"peerDependencies": {
|
|
@@ -66,6 +67,7 @@
|
|
|
66
67
|
}
|
|
67
68
|
},
|
|
68
69
|
"devDependencies": {
|
|
70
|
+
"@neslinesli93/qpdf-wasm": "0.3.0",
|
|
69
71
|
"pdfjs-dist": "4.10.38",
|
|
70
72
|
"typescript": "5.9.2"
|
|
71
73
|
}
|