@sirdaspdf/core 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sırdaş
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,56 @@
1
+ # @sirdaspdf/core
2
+
3
+ The engine behind [Sırdaş](https://sirdas.app): PDF operations, PDF→Markdown for
4
+ LLMs, local anonymization of personal data, and the pieces needed to turn
5
+ documents into a training set. Runs in the browser and in Node, with the same
6
+ code path.
7
+
8
+ ```bash
9
+ npm install @sirdaspdf/core
10
+ ```
11
+
12
+ ## Use
13
+
14
+ ```js
15
+ import { pdfToMarkdown, anonymize, deanonymize } from "@sirdaspdf/core";
16
+
17
+ const markdown = await pdfToMarkdown(pdfjsDocument);
18
+ const { text, findings } = anonymize(markdown);
19
+ // …send `text` to a model, then put the real data back into its answer:
20
+ const real = deanonymize(answer, findings);
21
+ ```
22
+
23
+ PDF.js is a peer dependency and optional: the text functions take a document you
24
+ have already opened, so the browser uses its own worker and Node uses the legacy
25
+ build. Structural operations (merge, split, compress) only need `pdf-lib`.
26
+
27
+ Subpath imports are available and the package is `sideEffects: false`, so a
28
+ bundler only pulls in what you use:
29
+
30
+ ```js
31
+ import { chunkMarkdown } from "@sirdaspdf/core/chunk";
32
+ import { dedupe } from "@sirdaspdf/core/dedupe";
33
+ ```
34
+
35
+ ## What is in it
36
+
37
+ | Module | |
38
+ |---|---|
39
+ | `markdown` | Layout-aware PDF→Markdown: font size becomes heading level, running headers removed, token estimate |
40
+ | `anonymize` | Rule-based detection with consistent, **reversible** placeholders. Tuned for Spanish and Colombian documents |
41
+ | `tables` | Table detection and CSV/Markdown export; columns rebuilt from recurring whitespace |
42
+ | `chunk` | Split at headings into context-sized pieces, each carrying its heading |
43
+ | `diff` | Line-level diff of two documents with a similarity score |
44
+ | `dataset` | Records for fine-tuning in TRL's field names, with provenance |
45
+ | `dedupe` | MinHash + LSH near-duplicate detection |
46
+ | `paragraphs` | CCNet-style repeated-paragraph detection |
47
+ | `contamination` | n-gram overlap between training and evaluation sets |
48
+ | `datasplit` | Train/validation/test split **by document**, never by chunk |
49
+ | `verifiable` | (prompt, ground_truth) pairs from tables, for RL with verifiable rewards |
50
+ | `merge` · `split` · `organize` · `lossless` · `ranges` · `ocr` · `errors` | |
51
+
52
+ Anonymization detection is rule-based and can miss unusual formats. Review
53
+ before sharing a sensitive document.
54
+
55
+ Full documentation: https://sirdas.app/docs/agents/core.md ·
56
+ Source: https://github.com/Brayan15p/SIRDAS-APP-PDF · MIT
@@ -0,0 +1,35 @@
1
+ /**
2
+ * Motor de anonimización local basado en reglas (v1).
3
+ * Pseudonimiza: el mismo valor recibe siempre la misma etiqueta,
4
+ * para que la IA conserve la coherencia del documento.
5
+ */
6
+ export type EntityType = "EMAIL" | "NIT" | "DOCUMENTO" | "TARJETA" | "TELEFONO" | "FECHA" | "DIRECCION" | "NOMBRE" | "HISTORIA_CLINICA";
7
+ export interface Finding {
8
+ type: EntityType;
9
+ value: string;
10
+ label: string;
11
+ }
12
+ export declare function anonymize(text: string): {
13
+ text: string;
14
+ findings: Finding[];
15
+ };
16
+ /**
17
+ * Devuelve los datos reales a un texto anonimizado: el viaje de vuelta.
18
+ *
19
+ * El flujo pensado es: anonimizar → pegar el texto con etiquetas en la IA →
20
+ * traer aquí la respuesta → recuperar los nombres y números reales. El
21
+ * documento original nunca se modifica y nada de esto sale del dispositivo.
22
+ *
23
+ * `unknown` recoge las etiquetas que aparecen en el texto pero no están en el
24
+ * mapa: casi siempre significa que la IA se inventó una etiqueta, o que el
25
+ * mapa no corresponde a este documento.
26
+ */
27
+ export declare function deanonymize(text: string, findings: Finding[]): {
28
+ text: string;
29
+ restored: number;
30
+ unknown: string[];
31
+ };
32
+ /** Serializa el mapa de etiquetas para guardarlo o pasarlo entre herramientas. */
33
+ export declare function findingsToMap(findings: Finding[]): Record<string, string>;
34
+ /** Reconstruye los hallazgos desde un mapa guardado. */
35
+ export declare function mapToFindings(map: Record<string, string>): Finding[];
@@ -0,0 +1,113 @@
1
+ const MONTHS = "enero|febrero|marzo|abril|mayo|junio|julio|agosto|septiembre|setiembre|octubre|noviembre|diciembre|january|february|march|april|may|june|july|august|september|october|november|december";
2
+ // Un nombre debe capturarse ENTERO: si se corta, el apellido que queda fuera
3
+ // se filtra. Cubrimos las partículas españolas ("de", "del", "de la/las/los")
4
+ // y hasta siete palabras, que es lo que llega a tener un nombre compuesto
5
+ // colombiano con dos apellidos.
6
+ const PARTICLE = "(?:de\\s+(?:la|las|los)\\s+|del\\s+|de\\s+)?";
7
+ const NAME = `[A-ZÁÉÍÓÚÑ][a-záéíóúñü]+(?:\\s+${PARTICLE}[A-ZÁÉÍÓÚÑ][a-záéíóúñü]+){1,6}`;
8
+ const NAME_UPPER = "[A-ZÁÉÍÓÚÑ]{2,}(?:\\s+[A-ZÁÉÍÓÚÑ]{2,}){1,6}";
9
+ /** Vuelve insensible a mayúsculas solo las etiquetas (el nombre sí distingue mayúsculas). */
10
+ const ci = (src) => src.replace(/[a-záéíóúñ]/g, (ch) => `[${ch}${ch.toUpperCase()}]`);
11
+ const NAME_LABELS = "nombres?(?:\\s+completo)?|paciente|usuario|afiliado|señora?|sr\\.|sra\\.|don|doña|comprador(?:a)?|vendedor(?:a)?|otorgante|compareciente|demandante|demandado|arrendador(?:a)?|arrendatario|apoderad[oa]|titular|firmado por|name|patient|client|signed by";
12
+ const RULES = [
13
+ { type: "EMAIL", re: /[\w.+-]+@[\w-]+(?:\.[\w-]+)+/g },
14
+ { type: "HISTORIA_CLINICA", re: /\b(?:historia\s+cl[ií]nica|h\.?\s?c\.?|n[uú]mero\s+de\s+historia)\s*(?:n[oº°.]*|#|:)?\s*([A-Z0-9-]{4,})/gi, group: 1 },
15
+ { type: "NIT", re: /\b(?:NIT\.?\s*:?\s*)?\d{3}\.?\d{3}\.?\d{3}\s?-\s?\d\b/g },
16
+ { type: "TARJETA", re: /\b(?:\d{4}[ -]?){3}\d{1,4}\b/g },
17
+ {
18
+ type: "DOCUMENTO",
19
+ re: /\b(?:C\.?\s?C\.?|c[eé]dula(?:\s+de\s+ciudadan[ií]a)?|T\.?\s?I\.?|C\.?\s?E\.?|NUIP|pasaporte|DNI|RUT|CURP|RFC|ID)\s*(?:n[oº°.]*|#|:)?\s*([A-Z]{0,4}\d[\d.\s-]{4,14}\d)/gi,
20
+ group: 1,
21
+ },
22
+ { type: "DOCUMENTO", re: /(?<![$€£\d.,]\s?)\b\d{1,3}(?:\.\d{3}){2,3}\b(?!\s*(?:pesos|cop|usd|millones|m²|m2))/gi },
23
+ {
24
+ type: "TELEFONO",
25
+ re: /(?:\+\d{1,3}[\s.-]?)?(?:\(\d{1,4}\)[\s.-]?)?\b(?:3\d{2}[\s.-]?\d{3}[\s.-]?\d{4}|60\d[\s.-]?\d{3}[\s.-]?\d{4}|\d{3}[\s.-]\d{3}[\s.-]\d{4})\b/g,
26
+ },
27
+ // Fijo colombiano con indicativo entre paréntesis: "(601) 742 1234", "(1) 742-1234".
28
+ {
29
+ type: "TELEFONO",
30
+ re: /(?:\+\d{1,3}[\s.-]?)?\(\d{1,3}\)[\s.-]?\d{3}[\s.-]?\d{4}\b/g,
31
+ },
32
+ { type: "FECHA", re: /\b\d{1,2}[/.-]\d{1,2}[/.-](?:\d{4}|\d{2})\b/g },
33
+ { type: "FECHA", re: /\b\d{4}-\d{2}-\d{2}\b/g },
34
+ { type: "FECHA", re: new RegExp(`\\b\\d{1,2}\\s+(?:de\\s+)?(?:${MONTHS})(?:\\s+(?:de|del)?\\s*\\d{4})?\\b`, "gi") },
35
+ {
36
+ type: "DIRECCION",
37
+ re: /\b(?:calle|cll?|carrera|cra|kr|kra|avenida|av|transversal|tv|diagonal|dg|autopista|street|st|avenue|ave)\.?\s*\d+[a-z]?(?:\s*bis)?(?:\s*(?:sur|norte|este))?\s*(?:#|n[oº°.]*|no\.?)\s*\d+[a-z]?\s*-\s*\d+(?:\s*(?:sur|norte|este))?(?:[,\s]+(?:apto|apartamento|oficina|of|int|interior|casa|torre)\.?\s*[\w-]+)*/gi,
38
+ },
39
+ { type: "NOMBRE", re: new RegExp(`(?:${ci(NAME_LABELS)})\\s*[:\\-]?\\s*(${NAME_UPPER}|${NAME})(?![a-záéíóúñ])`, "g"), group: 1 },
40
+ ];
41
+ export function anonymize(text) {
42
+ const map = new Map();
43
+ const counters = new Map();
44
+ const findings = [];
45
+ const labelFor = (type, value) => {
46
+ const key = `${type}:${value.replace(/\s+/g, " ").trim().toLowerCase()}`;
47
+ const found = map.get(key);
48
+ if (found)
49
+ return found;
50
+ const n = (counters.get(type) ?? 0) + 1;
51
+ counters.set(type, n);
52
+ const label = `[${type}_${n}]`;
53
+ map.set(key, label);
54
+ findings.push({ type, value: value.trim(), label });
55
+ return label;
56
+ };
57
+ let out = text;
58
+ for (const rule of RULES) {
59
+ out = out.replace(rule.re, (...args) => {
60
+ const full = args[0];
61
+ const target = rule.group ? args[rule.group] : full;
62
+ if (!target || target.startsWith("["))
63
+ return full;
64
+ if (rule.type === "NOMBRE" && /\[/.test(target))
65
+ return full;
66
+ const label = labelFor(rule.type, target);
67
+ return rule.group ? full.replace(target, label) : label;
68
+ });
69
+ }
70
+ // Segunda pasada: cualquier aparición suelta de un nombre ya detectado.
71
+ for (const f of findings.filter((x) => x.type === "NOMBRE")) {
72
+ const esc = f.value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&").replace(/\s+/g, "\\s+");
73
+ out = out.replace(new RegExp(`\\b${esc}\\b`, "gi"), f.label);
74
+ }
75
+ return { text: out, findings };
76
+ }
77
+ /**
78
+ * Devuelve los datos reales a un texto anonimizado: el viaje de vuelta.
79
+ *
80
+ * El flujo pensado es: anonimizar → pegar el texto con etiquetas en la IA →
81
+ * traer aquí la respuesta → recuperar los nombres y números reales. El
82
+ * documento original nunca se modifica y nada de esto sale del dispositivo.
83
+ *
84
+ * `unknown` recoge las etiquetas que aparecen en el texto pero no están en el
85
+ * mapa: casi siempre significa que la IA se inventó una etiqueta, o que el
86
+ * mapa no corresponde a este documento.
87
+ */
88
+ export function deanonymize(text, findings) {
89
+ const byLabel = new Map(findings.map((f) => [f.label, f.value]));
90
+ let restored = 0;
91
+ const unknown = new Set();
92
+ const out = text.replace(/\[([A-Z_]+)_(\d+)\]/g, (label) => {
93
+ const value = byLabel.get(label);
94
+ if (value === undefined) {
95
+ unknown.add(label);
96
+ return label;
97
+ }
98
+ restored++;
99
+ return value;
100
+ });
101
+ return { text: out, restored, unknown: [...unknown] };
102
+ }
103
+ /** Serializa el mapa de etiquetas para guardarlo o pasarlo entre herramientas. */
104
+ export function findingsToMap(findings) {
105
+ return Object.fromEntries(findings.map((f) => [f.label, f.value]));
106
+ }
107
+ /** Reconstruye los hallazgos desde un mapa guardado. */
108
+ export function mapToFindings(map) {
109
+ return Object.entries(map).map(([label, value]) => {
110
+ const type = (label.match(/^\[([A-Z_]+)_\d+\]$/)?.[1] ?? "NOMBRE");
111
+ return { type, value, label };
112
+ });
113
+ }
@@ -0,0 +1,13 @@
1
+ export interface Chunk {
2
+ /** Posición, base 1. */
3
+ index: number;
4
+ /** Encabezado bajo el que cae este trozo, si lo hay. */
5
+ heading: string | null;
6
+ text: string;
7
+ tokens: number;
8
+ }
9
+ /**
10
+ * Divide el Markdown en trozos de como máximo `maxTokens` (estimados).
11
+ * Secciones consecutivas pequeñas se agrupan para no generar ruido.
12
+ */
13
+ export declare function chunkMarkdown(markdown: string, maxTokens?: number): Chunk[];
package/dist/chunk.js ADDED
@@ -0,0 +1,93 @@
1
+ // Divide un documento en trozos del tamaño que acepta un modelo de IA.
2
+ //
3
+ // Cada trozo se corta en un encabezado y arrastra el encabezado que lo
4
+ // contiene, de modo que sea comprensible por sí solo: eso es lo que hace que
5
+ // un fragmento sirva tanto para pegarlo en un chat como para indexarlo en RAG.
6
+ import { estimateTokens } from "./markdown.js";
7
+ const HEADING = /^(#{1,3})\s+(.+)$/;
8
+ /** Parte el Markdown por encabezados, conservando el texto previo al primero. */
9
+ function sections(markdown) {
10
+ const out = [];
11
+ let heading = null;
12
+ let body = [];
13
+ const flush = () => {
14
+ const text = body.join("\n").trim();
15
+ if (text || heading)
16
+ out.push({ heading, body: text });
17
+ body = [];
18
+ };
19
+ for (const line of markdown.split("\n")) {
20
+ const m = line.match(HEADING);
21
+ if (m) {
22
+ flush();
23
+ heading = m[2].trim();
24
+ }
25
+ else {
26
+ body.push(line);
27
+ }
28
+ }
29
+ flush();
30
+ return out;
31
+ }
32
+ /** Parte un bloque largo por párrafos sin pasarse del presupuesto. */
33
+ function splitLong(body, maxTokens) {
34
+ const paragraphs = body.split(/\n{2,}/);
35
+ const parts = [];
36
+ let current = [];
37
+ const size = () => estimateTokens(current.join("\n\n"));
38
+ for (const p of paragraphs) {
39
+ if (current.length && size() + estimateTokens(p) > maxTokens) {
40
+ parts.push(current.join("\n\n"));
41
+ current = [];
42
+ }
43
+ // Un solo párrafo más grande que el presupuesto va solo: partirlo a la
44
+ // mitad de una frase sería peor que entregarlo entero.
45
+ current.push(p);
46
+ }
47
+ if (current.length)
48
+ parts.push(current.join("\n\n"));
49
+ return parts.filter((p) => p.trim());
50
+ }
51
+ /**
52
+ * Divide el Markdown en trozos de como máximo `maxTokens` (estimados).
53
+ * Secciones consecutivas pequeñas se agrupan para no generar ruido.
54
+ */
55
+ export function chunkMarkdown(markdown, maxTokens = 2000) {
56
+ const text = markdown.trim();
57
+ if (!text)
58
+ return [];
59
+ if (estimateTokens(text) <= maxTokens) {
60
+ return [{ index: 1, heading: null, text, tokens: estimateTokens(text) }];
61
+ }
62
+ const chunks = [];
63
+ const push = (heading, body) => {
64
+ const withHeading = heading ? `# ${heading}\n\n${body}`.trim() : body.trim();
65
+ if (!withHeading)
66
+ return;
67
+ chunks.push({ index: chunks.length + 1, heading, text: withHeading, tokens: estimateTokens(withHeading) });
68
+ };
69
+ let pending = null;
70
+ const flushPending = () => {
71
+ if (pending)
72
+ push(pending.heading, pending.body);
73
+ pending = null;
74
+ };
75
+ for (const sec of sections(text)) {
76
+ const whole = sec.heading ? `# ${sec.heading}\n\n${sec.body}` : sec.body;
77
+ if (estimateTokens(whole) > maxTokens) {
78
+ flushPending();
79
+ for (const part of splitLong(sec.body, maxTokens))
80
+ push(sec.heading, part);
81
+ continue;
82
+ }
83
+ if (pending && estimateTokens(`${pending.body}\n\n${whole}`) <= maxTokens) {
84
+ // Cabe con la sección anterior: agrupamos para no fragmentar de más.
85
+ pending.body = `${pending.body}\n\n${whole}`.trim();
86
+ continue;
87
+ }
88
+ flushPending();
89
+ pending = { heading: sec.heading, body: sec.body };
90
+ }
91
+ flushPending();
92
+ return chunks.map((c, i) => ({ ...c, index: i + 1 }));
93
+ }
@@ -0,0 +1,45 @@
1
+ export interface ContaminationOptions {
2
+ /** Palabras por n-grama. 13 es la heurística habitual. */
3
+ n?: number;
4
+ /** Fracción de n-gramas compartidos a partir de la cual se marca. */
5
+ threshold?: number;
6
+ /** Cuántos ejemplos de coincidencia devolver por registro. */
7
+ samples?: number;
8
+ /**
9
+ * Ignorar los n-gramas que aparezcan en más de este número de documentos de
10
+ * referencia. Es la regla de GPT-3 contra el texto repetido: una cláusula
11
+ * legal o un pie de página que sale en media empresa no es contaminación,
12
+ * es formulario. Poner 0 desactiva el filtro.
13
+ */
14
+ ignoreCommon?: number;
15
+ }
16
+ export interface Contaminated {
17
+ /** Índice en el conjunto revisado. */
18
+ index: number;
19
+ /** N-gramas de este registro que aparecen en el otro conjunto. */
20
+ shared: number;
21
+ /** Total de n-gramas del registro. */
22
+ total: number;
23
+ /** shared / total. */
24
+ ratio: number;
25
+ /** Fragmentos concretos que coincidieron. */
26
+ examples: string[];
27
+ }
28
+ export interface ContaminationReport {
29
+ contaminated: Contaminated[];
30
+ /** Índices que se pueden usar sin problema. */
31
+ clean: number[];
32
+ /** Fracción de registros marcados. */
33
+ ratio: number;
34
+ n: number;
35
+ /** N-gramas descartados por salir en demasiados documentos de referencia. */
36
+ ignoredAsBoilerplate: number;
37
+ }
38
+ /**
39
+ * Marca los registros de `check` que comparten texto con `reference`.
40
+ *
41
+ * El uso normal es `check` = entrenamiento y `reference` = evaluación: así
42
+ * sabes qué quitar del entrenamiento para que la evaluación siga valiendo.
43
+ * Al revés también sirve, para saber qué preguntas del test ya están gastadas.
44
+ */
45
+ export declare function findContamination(check: string[], reference: string[], options?: ContaminationOptions): ContaminationReport;
@@ -0,0 +1,74 @@
1
+ // Detección de contaminación: qué parte de tu conjunto de entrenamiento
2
+ // aparece también en el de evaluación.
3
+ //
4
+ // Si un fragmento del test está en el train, la métrica de evaluación deja de
5
+ // medir si el modelo aprendió y pasa a medir si recordó. El método estándar en
6
+ // la industria es el solapamiento de n-gramas largos: un n-grama de 13 palabras
7
+ // que aparece en los dos lados no es una coincidencia del idioma, es el mismo
8
+ // texto.
9
+ import { fnv1a, ngrams, ngramHashes } from "./text.js";
10
+ const DEFAULTS = { n: 13, threshold: 0.01, samples: 3, ignoreCommon: 10 };
11
+ /**
12
+ * Marca los registros de `check` que comparten texto con `reference`.
13
+ *
14
+ * El uso normal es `check` = entrenamiento y `reference` = evaluación: así
15
+ * sabes qué quitar del entrenamiento para que la evaluación siga valiendo.
16
+ * Al revés también sirve, para saber qué preguntas del test ya están gastadas.
17
+ */
18
+ export function findContamination(check, reference, options = {}) {
19
+ const { n, threshold, samples, ignoreCommon } = { ...DEFAULTS, ...options };
20
+ // Se cuenta en cuántos documentos distintos sale cada n-grama, no cuántas
21
+ // veces en total: repetir una frase dentro de un mismo documento no la
22
+ // convierte en formulario.
23
+ const docsWith = new Map();
24
+ for (const text of reference) {
25
+ for (const h of ngramHashes(text, n))
26
+ docsWith.set(h, (docsWith.get(h) ?? 0) + 1);
27
+ }
28
+ const refHashes = new Set();
29
+ let ignored = 0;
30
+ for (const [h, count] of docsWith) {
31
+ if (ignoreCommon > 0 && count > ignoreCommon)
32
+ ignored++;
33
+ else
34
+ refHashes.add(h);
35
+ }
36
+ const contaminated = [];
37
+ const clean = [];
38
+ for (let i = 0; i < check.length; i++) {
39
+ const grams = ngrams(check[i], n);
40
+ if (grams.length === 0) {
41
+ clean.push(i);
42
+ continue;
43
+ }
44
+ // Se recorre el texto, no el conjunto, para poder devolver el fragmento
45
+ // literal que coincidió: un informe que solo da un número no se puede
46
+ // comprobar.
47
+ const seen = new Set();
48
+ const examples = [];
49
+ let shared = 0;
50
+ for (const g of grams) {
51
+ if (!refHashes.has(fnv1a(g)))
52
+ continue;
53
+ shared++;
54
+ if (examples.length < samples && !seen.has(g)) {
55
+ seen.add(g);
56
+ examples.push(g);
57
+ }
58
+ }
59
+ const ratio = shared / grams.length;
60
+ if (shared > 0 && ratio >= threshold) {
61
+ contaminated.push({ index: i, shared, total: grams.length, ratio, examples });
62
+ }
63
+ else {
64
+ clean.push(i);
65
+ }
66
+ }
67
+ return {
68
+ contaminated,
69
+ clean,
70
+ ratio: check.length === 0 ? 0 : contaminated.length / check.length,
71
+ n,
72
+ ignoredAsBoilerplate: ignored,
73
+ };
74
+ }
@@ -0,0 +1,69 @@
1
+ /**
2
+ * Formatos de salida. Los nombres de campo son los que consumen los
3
+ * entrenadores; cambiarlos rompe la carga del dataset, así que no se tocan.
4
+ */
5
+ export type DatasetFormat = "text" | "chat" | "prompt-completion";
6
+ export interface SourceDocument {
7
+ /** Nombre del archivo: es la clave del reparto y de la trazabilidad. */
8
+ name: string;
9
+ /** El documento ya convertido a Markdown. */
10
+ markdown: string;
11
+ }
12
+ export interface BuildOptions {
13
+ format?: DatasetFormat;
14
+ /** Tokens por registro antes de trocear. */
15
+ maxTokens?: number;
16
+ /** Anonimizar los datos personales. Activado por defecto, y con motivo. */
17
+ anonymize?: boolean;
18
+ /**
19
+ * Plantilla de la instrucción para los formatos `chat` y
20
+ * `prompt-completion`. Admite {heading} y {document}.
21
+ */
22
+ template?: string;
23
+ /** Instrucción de sistema, solo para `chat`. */
24
+ system?: string;
25
+ }
26
+ /** Procedencia: sin esto un dataset no se puede auditar ni corregir. */
27
+ export interface Provenance {
28
+ document: string;
29
+ /** Índice del trozo dentro del documento, empezando en 1. */
30
+ chunk: number;
31
+ heading: string;
32
+ tokens: number;
33
+ /** Cuántos datos personales se sustituyeron en este registro. */
34
+ redacted: number;
35
+ }
36
+ export interface DatasetRecord {
37
+ /** Los campos del formato elegido, tal cual van al JSONL. */
38
+ data: Record<string, unknown>;
39
+ meta: Provenance;
40
+ }
41
+ /**
42
+ * Convierte documentos en registros de entrenamiento.
43
+ *
44
+ * Cada trozo de cada documento da un registro. El troceado es por encabezados,
45
+ * así que un registro no parte una idea por la mitad, y el encabezado viaja
46
+ * dentro del texto para que el fragmento se entienda solo.
47
+ */
48
+ export declare function buildRecords(docs: SourceDocument[], options?: BuildOptions): DatasetRecord[];
49
+ /**
50
+ * Serializa a JSONL. Una línea por registro, sin sangrado: es lo que leen los
51
+ * cargadores de datasets.
52
+ *
53
+ * `withMeta` añade la procedencia a cada línea. Los entrenadores ignoran los
54
+ * campos que no conocen, así que dejarla puesta sale gratis y permite rastrear
55
+ * de qué página salió cada ejemplo cuando el modelo diga algo raro.
56
+ */
57
+ export declare function toJsonl(records: DatasetRecord[], withMeta?: boolean): string;
58
+ export interface DatasetStats {
59
+ records: number;
60
+ documents: number;
61
+ tokens: number;
62
+ /** Total de datos personales sustituidos en todo el conjunto. */
63
+ redacted: number;
64
+ /** Registros en los que se sustituyó al menos un dato personal. */
65
+ recordsWithPii: number;
66
+ medianTokens: number;
67
+ maxTokens: number;
68
+ }
69
+ export declare function datasetStats(records: DatasetRecord[]): DatasetStats;
@@ -0,0 +1,95 @@
1
+ // Construcción del conjunto de datos: de documentos a JSONL listo para
2
+ // entrenar.
3
+ //
4
+ // Lo que NO hace, a propósito: inventar preguntas y respuestas con un modelo.
5
+ // Eso exigiría mandar tus documentos a una API, que es exactamente lo que este
6
+ // producto existe para evitar. Aquí todo sale del documento por reglas, y cada
7
+ // registro dice de dónde salió.
8
+ import { chunkMarkdown } from "./chunk.js";
9
+ import { anonymize as redactPii } from "./anonymize.js";
10
+ import { estimateTokens } from "./markdown.js";
11
+ const DEFAULT_TEMPLATE = "Resume el contenido de «{heading}» del documento {document}.";
12
+ /** Un trozo sin encabezado no puede pedir «resume la sección “archivo.pdf”». */
13
+ const DEFAULT_TEMPLATE_NO_HEADING = "Resume el contenido del documento {document}.";
14
+ function fillTemplate(tpl, values) {
15
+ return tpl.replace(/\{(heading|document)\}/g, (_, k) => values[k]);
16
+ }
17
+ /**
18
+ * Convierte documentos en registros de entrenamiento.
19
+ *
20
+ * Cada trozo de cada documento da un registro. El troceado es por encabezados,
21
+ * así que un registro no parte una idea por la mitad, y el encabezado viaja
22
+ * dentro del texto para que el fragmento se entienda solo.
23
+ */
24
+ export function buildRecords(docs, options = {}) {
25
+ const { format = "text", maxTokens = 2000, anonymize = true, system, } = options;
26
+ const records = [];
27
+ for (const doc of docs) {
28
+ const chunks = chunkMarkdown(doc.markdown, maxTokens);
29
+ chunks.forEach((chunk, i) => {
30
+ let text = chunk.text;
31
+ let redacted = 0;
32
+ if (anonymize) {
33
+ const result = redactPii(text);
34
+ text = result.text;
35
+ redacted = result.findings.length;
36
+ }
37
+ if (text.trim() === "")
38
+ return;
39
+ const meta = {
40
+ document: doc.name,
41
+ chunk: i + 1,
42
+ heading: chunk.heading ?? "",
43
+ tokens: estimateTokens(text),
44
+ redacted,
45
+ };
46
+ const hasHeading = (chunk.heading ?? "").trim() !== "";
47
+ const tpl = options.template ?? (hasHeading ? DEFAULT_TEMPLATE : DEFAULT_TEMPLATE_NO_HEADING);
48
+ const instruction = fillTemplate(tpl, { heading: chunk.heading ?? "", document: doc.name });
49
+ records.push({ data: shape(format, text, instruction, system), meta });
50
+ });
51
+ }
52
+ return records;
53
+ }
54
+ function shape(format, text, instruction, system) {
55
+ switch (format) {
56
+ case "text":
57
+ return { text };
58
+ case "prompt-completion":
59
+ return { prompt: instruction, completion: text };
60
+ case "chat": {
61
+ const messages = [];
62
+ if (system)
63
+ messages.push({ role: "system", content: system });
64
+ messages.push({ role: "user", content: instruction });
65
+ messages.push({ role: "assistant", content: text });
66
+ return { messages };
67
+ }
68
+ }
69
+ }
70
+ /**
71
+ * Serializa a JSONL. Una línea por registro, sin sangrado: es lo que leen los
72
+ * cargadores de datasets.
73
+ *
74
+ * `withMeta` añade la procedencia a cada línea. Los entrenadores ignoran los
75
+ * campos que no conocen, así que dejarla puesta sale gratis y permite rastrear
76
+ * de qué página salió cada ejemplo cuando el modelo diga algo raro.
77
+ */
78
+ export function toJsonl(records, withMeta = true) {
79
+ return records
80
+ .map((r) => JSON.stringify(withMeta ? { ...r.data, sirdas: r.meta } : r.data))
81
+ .join("\n");
82
+ }
83
+ export function datasetStats(records) {
84
+ const tokens = records.map((r) => r.meta.tokens).sort((a, b) => a - b);
85
+ const total = tokens.reduce((a, b) => a + b, 0);
86
+ return {
87
+ records: records.length,
88
+ documents: new Set(records.map((r) => r.meta.document)).size,
89
+ tokens: total,
90
+ redacted: records.reduce((n, r) => n + r.meta.redacted, 0),
91
+ recordsWithPii: records.filter((r) => r.meta.redacted > 0).length,
92
+ medianTokens: tokens.length === 0 ? 0 : tokens[Math.floor(tokens.length / 2)],
93
+ maxTokens: tokens.length === 0 ? 0 : tokens[tokens.length - 1],
94
+ };
95
+ }
@@ -0,0 +1,30 @@
1
+ export interface SplitRatios {
2
+ train: number;
3
+ validation: number;
4
+ test: number;
5
+ }
6
+ export interface SplitOptions {
7
+ ratios?: SplitRatios;
8
+ /** Cambiarla reordena el reparto; la misma semilla da siempre lo mismo. */
9
+ seed?: string;
10
+ }
11
+ export interface SplitResult<T> {
12
+ train: T[];
13
+ validation: T[];
14
+ test: T[];
15
+ /** Cuántos documentos distintos fueron a cada partición. */
16
+ documents: {
17
+ train: number;
18
+ validation: number;
19
+ test: number;
20
+ };
21
+ }
22
+ /**
23
+ * Reparte registros agrupándolos por documento de origen.
24
+ *
25
+ * `docOf` dice a qué documento pertenece cada registro. El reparto se decide
26
+ * hasheando el nombre del documento junto con la semilla, así que es estable:
27
+ * volver a ejecutarlo con los mismos archivos da exactamente el mismo reparto,
28
+ * y añadir un documento nuevo no mueve a los demás de partición.
29
+ */
30
+ export declare function splitByDocument<T>(records: T[], docOf: (record: T) => string, options?: SplitOptions): SplitResult<T>;