@sirdaspdf/core 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/tables.js ADDED
@@ -0,0 +1,186 @@
1
+ /** Extrae cada trozo de texto con sus coordenadas, sin fusionar filas. */
2
+ export async function extractCells(pdf, onProgress) {
3
+ const cells = [];
4
+ for (let p = 1; p <= pdf.numPages; p++) {
5
+ const page = await pdf.getPage(p);
6
+ const content = await page.getTextContent();
7
+ for (const it of content.items) {
8
+ if (!("str" in it) || !it.str.trim())
9
+ continue;
10
+ const t = it.transform;
11
+ cells.push({
12
+ text: it.str.trim(),
13
+ x0: t[4],
14
+ x1: t[4] + (it.width || 0),
15
+ y: Math.round(t[5] / 2) * 2,
16
+ page: p,
17
+ });
18
+ }
19
+ page.cleanup();
20
+ onProgress?.(p / pdf.numPages);
21
+ }
22
+ return cells;
23
+ }
24
+ /** Agrupa los trozos en filas por página y altura. */
25
+ function toRows(cells) {
26
+ const byKey = new Map();
27
+ for (const c of cells) {
28
+ const key = `${c.page}:${c.y}`;
29
+ const row = byKey.get(key) ?? { page: c.page, y: c.y, items: [] };
30
+ row.items.push(c);
31
+ byKey.set(key, row);
32
+ }
33
+ return [...byKey.values()]
34
+ .map((r) => ({ ...r, items: r.items.sort((a, b) => a.x0 - b.x0) }))
35
+ .sort((a, b) => (a.page !== b.page ? a.page - b.page : b.y - a.y));
36
+ }
37
+ /** ¿La posición x cae en un hueco de esta fila (o fuera de su extensión)? */
38
+ function isBlankAt(row, x) {
39
+ for (const it of row.items) {
40
+ if (x > it.x0 && x < it.x1)
41
+ return false;
42
+ }
43
+ return true;
44
+ }
45
+ /**
46
+ * Ancho típico de carácter en la fila. Sirve de referencia: un espacio entre
47
+ * palabras mide alrededor de un tercio de carácter, mientras que el hueco
48
+ * entre columnas es varias veces mayor.
49
+ */
50
+ function charWidth(row) {
51
+ const widths = row.items
52
+ .filter((i) => i.text.length > 0)
53
+ .map((i) => (i.x1 - i.x0) / i.text.length)
54
+ .filter((w) => w > 0)
55
+ .sort((a, b) => a - b);
56
+ return widths.length ? widths[Math.floor(widths.length / 2)] : 5;
57
+ }
58
+ /**
59
+ * Posiciones candidatas a separador: el centro de cada hueco entre celdas que
60
+ * sea claramente mayor que un espacio entre palabras.
61
+ */
62
+ function gapCentres(row) {
63
+ const minGap = Math.max(6, charWidth(row) * 2.5);
64
+ const out = [];
65
+ for (let i = 1; i < row.items.length; i++) {
66
+ const gap = row.items[i].x0 - row.items[i - 1].x1;
67
+ if (gap >= minGap)
68
+ out.push((row.items[i - 1].x1 + row.items[i].x0) / 2);
69
+ }
70
+ return out;
71
+ }
72
+ /** Corta un bloque de filas en columnas y devuelve la tabla, o null si no lo es. */
73
+ function buildTable(block, opts) {
74
+ const candidates = block.flatMap((r) => gapCentres(r)).sort((a, b) => a - b);
75
+ if (!candidates.length)
76
+ return null;
77
+ // Agrupamos candidatos cercanos: son el mismo corte visto en varias filas.
78
+ const clusters = [];
79
+ for (const x of candidates) {
80
+ const last = clusters[clusters.length - 1];
81
+ if (last && x - last[last.length - 1] <= 8)
82
+ last.push(x);
83
+ else
84
+ clusters.push([x]);
85
+ }
86
+ // Un corte vale solo si está en blanco en casi todas las filas del bloque.
87
+ const separators = [];
88
+ for (const cluster of clusters) {
89
+ const x = cluster.reduce((s, v) => s + v, 0) / cluster.length;
90
+ const support = block.filter((r) => isBlankAt(r, x)).length / block.length;
91
+ if (support >= opts.minSeparatorSupport)
92
+ separators.push(x);
93
+ }
94
+ if (separators.length < opts.minCols - 1)
95
+ return null;
96
+ const slots = separators.length + 1;
97
+ let rows = [];
98
+ for (const row of block) {
99
+ const cellsOfRow = Array.from({ length: slots }, () => []);
100
+ for (const it of row.items) {
101
+ const mid = (it.x0 + it.x1) / 2;
102
+ let col = 0;
103
+ while (col < separators.length && mid > separators[col])
104
+ col++;
105
+ cellsOfRow[col].push(it.text);
106
+ }
107
+ rows.push(cellsOfRow.map((parts) => parts.join(" ").trim()));
108
+ }
109
+ // Una misma banda de espacio en blanco puede producir varios cortes (los
110
+ // importes alineados a la derecha empiezan en una x distinta por fila), y
111
+ // entre ellos queda una columna vacía en todas las filas. Se descartan.
112
+ const used = Array.from({ length: slots }, (_, i) => rows.some((r) => r[i] !== ""));
113
+ const columns = used.filter(Boolean).length;
114
+ if (columns < opts.minCols)
115
+ return null;
116
+ rows = rows.map((r) => r.filter((_, i) => used[i]));
117
+ // Una "tabla" donde casi todas las filas tienen una sola celda es, en
118
+ // realidad, un párrafo con sangría.
119
+ const filled = rows.filter((r) => r.filter(Boolean).length >= 2).length;
120
+ if (filled / rows.length < 0.6)
121
+ return null;
122
+ return { page: block[0].page, rows, columns };
123
+ }
124
+ /** Encuentra las tablas de un documento a partir de sus trozos de texto. */
125
+ export function detectTables(cells, options = {}) {
126
+ const opts = {
127
+ minRows: options.minRows ?? 3,
128
+ minCols: options.minCols ?? 2,
129
+ maxRowGap: options.maxRowGap ?? 28,
130
+ minSeparatorSupport: options.minSeparatorSupport ?? 0.8,
131
+ };
132
+ const rows = toRows(cells);
133
+ const tables = [];
134
+ let block = [];
135
+ const flush = () => {
136
+ if (block.length >= opts.minRows) {
137
+ const table = buildTable(block, opts);
138
+ if (table)
139
+ tables.push(table);
140
+ }
141
+ block = [];
142
+ };
143
+ for (const row of rows) {
144
+ const prev = block[block.length - 1];
145
+ const breaks = prev !== undefined && (prev.page !== row.page || prev.y - row.y > opts.maxRowGap);
146
+ // Una fila de una sola celda no participa en una tabla: corta el bloque.
147
+ if (breaks || row.items.length < 2) {
148
+ flush();
149
+ if (row.items.length >= 2)
150
+ block.push(row);
151
+ continue;
152
+ }
153
+ block.push(row);
154
+ }
155
+ flush();
156
+ return tables;
157
+ }
158
+ /** Documento abierto → tablas detectadas. */
159
+ export async function pdfToTables(pdf, options = {}, onProgress) {
160
+ return detectTables(await extractCells(pdf, onProgress), options);
161
+ }
162
+ /** Escapa un valor según RFC 4180. */
163
+ function escapeCsv(value, delimiter) {
164
+ if (value.includes(delimiter) || /["\r\n]/.test(value)) {
165
+ return `"${value.replace(/"/g, '""')}"`;
166
+ }
167
+ return value;
168
+ }
169
+ export function tableToCsv(table, { delimiter = ",", bom = false } = {}) {
170
+ const body = table.rows.map((r) => r.map((c) => escapeCsv(c, delimiter)).join(delimiter)).join("\r\n");
171
+ return (bom ? "" : "") + body;
172
+ }
173
+ /** Tabla en Markdown, usando la primera fila como encabezado. */
174
+ export function tableToMarkdown(table) {
175
+ if (!table.rows.length)
176
+ return "";
177
+ const esc = (s) => s.replace(/\|/g, "\\|");
178
+ const [head, ...body] = table.rows;
179
+ const width = table.columns;
180
+ const pad = (r) => Array.from({ length: width }, (_, i) => esc(r[i] ?? ""));
181
+ return [
182
+ `| ${pad(head).join(" | ")} |`,
183
+ `| ${Array.from({ length: width }, () => "---").join(" | ")} |`,
184
+ ...body.map((r) => `| ${pad(r).join(" | ")} |`),
185
+ ].join("\n");
186
+ }
package/dist/text.d.ts ADDED
@@ -0,0 +1,19 @@
1
+ /**
2
+ * Normaliza para comparar: minúsculas, sin acentos, sin puntuación y con los
3
+ * espacios colapsados. Dos textos que solo difieren en formato deben verse
4
+ * iguales, porque para un modelo de lenguaje lo son.
5
+ */
6
+ export declare function normalizeForCompare(text: string): string;
7
+ export declare function words(text: string): string[];
8
+ /** FNV-1a de 32 bits: rápido, sin dependencias y estable entre ejecuciones. */
9
+ export declare function fnv1a(s: string): number;
10
+ /**
11
+ * Conjunto de n-gramas de palabras, como hashes. Si el texto es más corto que
12
+ * `n`, devuelve un único elemento con todo el texto: un documento de tres
13
+ * palabras sigue pudiendo ser un duplicado.
14
+ */
15
+ export declare function ngramHashes(text: string, n: number): Set<number>;
16
+ /** Los n-gramas como texto, para poder enseñar cuál coincidió. */
17
+ export declare function ngrams(text: string, n: number): string[];
18
+ /** Jaccard exacto entre dos conjuntos. */
19
+ export declare function jaccard(a: Set<number>, b: Set<number>): number;
package/dist/text.js ADDED
@@ -0,0 +1,74 @@
1
+ // Utilidades de texto compartidas por la deduplicación y la detección de
2
+ // contaminación. Todo es determinista: la misma entrada da el mismo resultado
3
+ // en cualquier máquina, que es lo que permite auditar un dataset.
4
+ /**
5
+ * Normaliza para comparar: minúsculas, sin acentos, sin puntuación y con los
6
+ * espacios colapsados. Dos textos que solo difieren en formato deben verse
7
+ * iguales, porque para un modelo de lenguaje lo son.
8
+ */
9
+ export function normalizeForCompare(text) {
10
+ return text
11
+ .toLowerCase()
12
+ .normalize("NFD")
13
+ .replace(/[̀-ͯ]/g, "")
14
+ .replace(/[^\p{L}\p{N}\s]/gu, " ")
15
+ .replace(/\s+/g, " ")
16
+ .trim();
17
+ }
18
+ export function words(text) {
19
+ const n = normalizeForCompare(text);
20
+ return n === "" ? [] : n.split(" ");
21
+ }
22
+ /** FNV-1a de 32 bits: rápido, sin dependencias y estable entre ejecuciones. */
23
+ export function fnv1a(s) {
24
+ let h = 0x811c9dc5;
25
+ for (let i = 0; i < s.length; i++) {
26
+ h ^= s.charCodeAt(i);
27
+ // h *= 16777619 en aritmética de 32 bits sin desbordar el doble de JS.
28
+ h = (h + ((h << 1) + (h << 4) + (h << 7) + (h << 8) + (h << 24))) >>> 0;
29
+ }
30
+ return h >>> 0;
31
+ }
32
+ /**
33
+ * Conjunto de n-gramas de palabras, como hashes. Si el texto es más corto que
34
+ * `n`, devuelve un único elemento con todo el texto: un documento de tres
35
+ * palabras sigue pudiendo ser un duplicado.
36
+ */
37
+ export function ngramHashes(text, n) {
38
+ const w = words(text);
39
+ const out = new Set();
40
+ if (w.length === 0)
41
+ return out;
42
+ if (w.length < n) {
43
+ out.add(fnv1a(w.join(" ")));
44
+ return out;
45
+ }
46
+ for (let i = 0; i + n <= w.length; i++)
47
+ out.add(fnv1a(w.slice(i, i + n).join(" ")));
48
+ return out;
49
+ }
50
+ /** Los n-gramas como texto, para poder enseñar cuál coincidió. */
51
+ export function ngrams(text, n) {
52
+ const w = words(text);
53
+ if (w.length === 0)
54
+ return [];
55
+ if (w.length < n)
56
+ return [w.join(" ")];
57
+ const out = [];
58
+ for (let i = 0; i + n <= w.length; i++)
59
+ out.push(w.slice(i, i + n).join(" "));
60
+ return out;
61
+ }
62
+ /** Jaccard exacto entre dos conjuntos. */
63
+ export function jaccard(a, b) {
64
+ if (a.size === 0 && b.size === 0)
65
+ return 1;
66
+ if (a.size === 0 || b.size === 0)
67
+ return 0;
68
+ const [small, large] = a.size <= b.size ? [a, b] : [b, a];
69
+ let shared = 0;
70
+ for (const x of small)
71
+ if (large.has(x))
72
+ shared++;
73
+ return shared / (a.size + b.size - shared);
74
+ }
@@ -0,0 +1,60 @@
1
+ import type { Table } from "./tables.js";
2
+ export type VerifiableKind = "lookup" | "count" | "extreme";
3
+ export interface VerifiablePair {
4
+ /** La pregunta, con la tabla incluida como contexto. */
5
+ prompt: string;
6
+ /** La respuesta correcta, copiada literalmente de la tabla. */
7
+ answer: string;
8
+ kind: VerifiableKind;
9
+ source: {
10
+ document: string;
11
+ page: number;
12
+ /** Fila y columna de donde salió la respuesta, base 1 sin contar cabecera. */
13
+ row?: number;
14
+ column?: string;
15
+ };
16
+ }
17
+ export interface VerifiableOptions {
18
+ /** Incluir la tabla en el enunciado. Sin ella la tarea no es resoluble. */
19
+ includeTable?: boolean;
20
+ /** Máximo de pares por tabla, para no ahogar el conjunto con una sola. */
21
+ maxPerTable?: number;
22
+ /** Idioma de los enunciados. */
23
+ lang?: "es" | "en";
24
+ }
25
+ /**
26
+ * Genera pares verificables a partir de una tabla.
27
+ *
28
+ * Solo se generan preguntas cuya respuesta sea inequívoca: si una clave de fila
29
+ * se repite, o si una celda está vacía, esa pregunta no se emite. Un par
30
+ * ambiguo en un conjunto de refuerzo es peor que no tenerlo, porque castiga al
31
+ * modelo por acertar.
32
+ */
33
+ export declare function tableToVerifiable(table: Table, document: string, options?: VerifiableOptions): VerifiablePair[];
34
+ /**
35
+ * Comprueba una respuesta contra la esperada.
36
+ *
37
+ * Es la función de recompensa: devuelve 1 o 0, nada intermedio. Se comparan
38
+ * números como números cuando ambos lo son, para que «1.120.000» y «1120000»
39
+ * cuenten como la misma respuesta; en otro caso se comparan como texto sin
40
+ * distinguir mayúsculas ni espacios de sobra.
41
+ */
42
+ export declare function verifyAnswer(expected: string, given: string): boolean;
43
+ /**
44
+ * JSONL de pares verificables, un par por línea.
45
+ *
46
+ * El campo se llama `ground_truth` y no `answer` porque es el nombre que usa el
47
+ * conjunto de referencia de RLVR (allenai/RLVR-GSM-MATH-IF-Mixed-Constraints) y
48
+ * el que espera el ejemplo de función de recompensa de GRPOTrainer. `prompt` es
49
+ * la única columna que el entrenador exige; las demás le llegan a la función de
50
+ * recompensa como argumentos con nombre, así que la procedencia viaja gratis.
51
+ */
52
+ export declare function verifiableToJsonl(pairs: VerifiablePair[]): string;
53
+ /**
54
+ * La función de recompensa que acompaña a ese JSONL, lista para pegar.
55
+ *
56
+ * Se entrega junto a los datos porque la mitad del trabajo de montar RLVR es
57
+ * escribir el verificador, y aquí ya sabemos exactamente cómo se comparan estas
58
+ * respuestas: es la misma regla que aplica verifyAnswer.
59
+ */
60
+ export declare function rewardFunctionPython(): string;
@@ -0,0 +1,244 @@
1
+ import { tableToMarkdown } from "./tables.js";
2
+ const PHRASES = {
3
+ es: {
4
+ lookup: (col, key) => `¿Cuál es el valor de «${col}» para «${key}»?`,
5
+ count: () => "¿Cuántas filas de datos tiene la tabla, sin contar la cabecera?",
6
+ max: (col) => `¿Cuál es el valor más alto de la columna «${col}»?`,
7
+ min: (col) => `¿Cuál es el valor más bajo de la columna «${col}»?`,
8
+ context: (doc, page) => `Tabla de la página ${page} de ${doc}:`,
9
+ ask: "Responde únicamente con el valor, sin explicación.",
10
+ },
11
+ en: {
12
+ lookup: (col, key) => `What is the value of "${col}" for "${key}"?`,
13
+ count: () => "How many data rows does the table have, not counting the header?",
14
+ max: (col) => `What is the highest value in the "${col}" column?`,
15
+ min: (col) => `What is the lowest value in the "${col}" column?`,
16
+ context: (doc, page) => `Table from page ${page} of ${doc}:`,
17
+ ask: "Answer with the value only, no explanation.",
18
+ },
19
+ };
20
+ /**
21
+ * Lee un número escrito a la colombiana (1.120.000,50) o a la inglesa
22
+ * (1,120,000.50), o devuelve null si no hay número.
23
+ *
24
+ * El caso ambiguo es un único separador seguido de exactamente tres cifras:
25
+ * «1.120» son mil ciento veinte en Bogotá y uno coma doce en Londres. Se
26
+ * resuelve como millar, porque estos pares salen de facturas y extractos donde
27
+ * esa es la lectura correcta. No afecta a comparar una respuesta consigo misma:
28
+ * dos textos idénticos ya coinciden antes de llegar aquí.
29
+ */
30
+ function parseNumber(raw) {
31
+ const s = raw.replace(/[^\d.,-]/g, "").trim();
32
+ if (s === "" || !/\d/.test(s))
33
+ return null;
34
+ const dots = (s.match(/\./g) ?? []).length;
35
+ const commas = (s.match(/,/g) ?? []).length;
36
+ let normalized;
37
+ if (dots > 0 && commas > 0) {
38
+ // Los dos presentes: el último que aparece es el decimal.
39
+ normalized =
40
+ s.lastIndexOf(",") > s.lastIndexOf(".")
41
+ ? s.replace(/\./g, "").replace(",", ".")
42
+ : s.replace(/,/g, "");
43
+ }
44
+ else if (dots > 1 || commas > 1) {
45
+ // Repetido: solo puede ser separador de millares.
46
+ normalized = s.replace(/[.,]/g, "");
47
+ }
48
+ else if (dots === 1 || commas === 1) {
49
+ const sep = dots === 1 ? "." : ",";
50
+ const after = s.slice(s.indexOf(sep) + 1);
51
+ normalized = /^\d{3}$/.test(after)
52
+ ? s.replace(/[.,]/g, "") // millares
53
+ : s.replace(",", ".");
54
+ }
55
+ else {
56
+ normalized = s;
57
+ }
58
+ const n = Number(normalized);
59
+ return Number.isFinite(n) ? n : null;
60
+ }
61
+ /** Una columna es numérica si lo son casi todas sus celdas no vacías. */
62
+ function numericColumn(rows, col) {
63
+ const values = rows.map((r) => r[col] ?? "").filter((v) => v.trim() !== "");
64
+ if (values.length < 2)
65
+ return false;
66
+ return values.filter((v) => parseNumber(v) !== null).length / values.length >= 0.8;
67
+ }
68
+ /**
69
+ * Genera pares verificables a partir de una tabla.
70
+ *
71
+ * Solo se generan preguntas cuya respuesta sea inequívoca: si una clave de fila
72
+ * se repite, o si una celda está vacía, esa pregunta no se emite. Un par
73
+ * ambiguo en un conjunto de refuerzo es peor que no tenerlo, porque castiga al
74
+ * modelo por acertar.
75
+ */
76
+ export function tableToVerifiable(table, document, options = {}) {
77
+ const { includeTable = true, maxPerTable = 10, lang = "es" } = options;
78
+ const p = PHRASES[lang];
79
+ const rows = table.rows;
80
+ if (rows.length < 2)
81
+ return [];
82
+ const header = rows[0];
83
+ const body = rows.slice(1);
84
+ if (header.length < 2)
85
+ return [];
86
+ const context = includeTable
87
+ ? `${p.context(document, table.page)}\n\n${tableToMarkdown(table)}\n\n`
88
+ : "";
89
+ const ask = (question) => `${context}${question} ${p.ask}`;
90
+ // Claves de fila que aparecen una sola vez: si «Consulta general» está dos
91
+ // veces, preguntar por ella no tiene respuesta única.
92
+ const keyCount = new Map();
93
+ for (const row of body) {
94
+ const key = (row[0] ?? "").trim();
95
+ if (key)
96
+ keyCount.set(key, (keyCount.get(key) ?? 0) + 1);
97
+ }
98
+ const pairs = [];
99
+ for (let r = 0; r < body.length && pairs.length < maxPerTable; r++) {
100
+ const key = (body[r][0] ?? "").trim();
101
+ if (!key || keyCount.get(key) !== 1)
102
+ continue;
103
+ for (let c = 1; c < header.length && pairs.length < maxPerTable; c++) {
104
+ const col = (header[c] ?? "").trim();
105
+ const value = (body[r][c] ?? "").trim();
106
+ if (!col || !value)
107
+ continue;
108
+ pairs.push({
109
+ prompt: ask(p.lookup(col, key)),
110
+ answer: value,
111
+ kind: "lookup",
112
+ source: { document, page: table.page, row: r + 1, column: col },
113
+ });
114
+ }
115
+ }
116
+ if (pairs.length < maxPerTable) {
117
+ pairs.push({
118
+ prompt: ask(p.count()),
119
+ answer: String(body.length),
120
+ kind: "count",
121
+ source: { document, page: table.page },
122
+ });
123
+ }
124
+ // Máximo y mínimo por columna numérica: exige leer toda la columna, no una
125
+ // celda, así que es una tarea de otro tipo aunque salga de la misma tabla.
126
+ for (let c = 1; c < header.length && pairs.length < maxPerTable; c++) {
127
+ if (!numericColumn(body, c))
128
+ continue;
129
+ const col = (header[c] ?? "").trim();
130
+ const cells = body
131
+ .map((row) => ({ raw: (row[c] ?? "").trim(), n: parseNumber(row[c] ?? "") }))
132
+ .filter((x) => x.n !== null);
133
+ if (cells.length < 2)
134
+ continue;
135
+ const top = cells.reduce((a, b) => (b.n > a.n ? b : a));
136
+ const bottom = cells.reduce((a, b) => (b.n < a.n ? b : a));
137
+ // Si hay empate en el extremo, la respuesta no es única: se descarta.
138
+ if (cells.filter((x) => x.n === top.n).length === 1) {
139
+ pairs.push({
140
+ prompt: ask(p.max(col)),
141
+ answer: top.raw,
142
+ kind: "extreme",
143
+ source: { document, page: table.page, column: col },
144
+ });
145
+ }
146
+ if (pairs.length < maxPerTable && cells.filter((x) => x.n === bottom.n).length === 1) {
147
+ pairs.push({
148
+ prompt: ask(p.min(col)),
149
+ answer: bottom.raw,
150
+ kind: "extreme",
151
+ source: { document, page: table.page, column: col },
152
+ });
153
+ }
154
+ }
155
+ return pairs;
156
+ }
157
+ /**
158
+ * Comprueba una respuesta contra la esperada.
159
+ *
160
+ * Es la función de recompensa: devuelve 1 o 0, nada intermedio. Se comparan
161
+ * números como números cuando ambos lo son, para que «1.120.000» y «1120000»
162
+ * cuenten como la misma respuesta; en otro caso se comparan como texto sin
163
+ * distinguir mayúsculas ni espacios de sobra.
164
+ */
165
+ export function verifyAnswer(expected, given) {
166
+ const a = expected.trim();
167
+ const b = given.trim();
168
+ if (a === b)
169
+ return true;
170
+ const na = parseNumber(a);
171
+ const nb = parseNumber(b);
172
+ if (na !== null && nb !== null)
173
+ return na === nb;
174
+ return a.toLowerCase().replace(/\s+/g, " ") === b.toLowerCase().replace(/\s+/g, " ");
175
+ }
176
+ /**
177
+ * JSONL de pares verificables, un par por línea.
178
+ *
179
+ * El campo se llama `ground_truth` y no `answer` porque es el nombre que usa el
180
+ * conjunto de referencia de RLVR (allenai/RLVR-GSM-MATH-IF-Mixed-Constraints) y
181
+ * el que espera el ejemplo de función de recompensa de GRPOTrainer. `prompt` es
182
+ * la única columna que el entrenador exige; las demás le llegan a la función de
183
+ * recompensa como argumentos con nombre, así que la procedencia viaja gratis.
184
+ */
185
+ export function verifiableToJsonl(pairs) {
186
+ return pairs
187
+ .map((p) => JSON.stringify({ prompt: p.prompt, ground_truth: p.answer, kind: p.kind, sirdas: p.source }))
188
+ .join("\n");
189
+ }
190
+ /**
191
+ * La función de recompensa que acompaña a ese JSONL, lista para pegar.
192
+ *
193
+ * Se entrega junto a los datos porque la mitad del trabajo de montar RLVR es
194
+ * escribir el verificador, y aquí ya sabemos exactamente cómo se comparan estas
195
+ * respuestas: es la misma regla que aplica verifyAnswer.
196
+ */
197
+ export function rewardFunctionPython() {
198
+ return `# Función de recompensa para los pares generados por \`sirdas rl\`.
199
+ # Uso con TRL: el dataset necesita la columna "prompt"; las demás columnas
200
+ # llegan aquí como argumentos con nombre.
201
+ #
202
+ # from trl import GRPOConfig, GRPOTrainer
203
+ # trainer = GRPOTrainer(model=..., reward_funcs=sirdas_reward, args=..., train_dataset=ds)
204
+ import re
205
+
206
+
207
+ def _to_number(raw):
208
+ """Lee 1.120.000 (es) y 1,120,000.50 (en). Devuelve None si no hay número."""
209
+ s = re.sub(r"[^\\d.,-]", "", str(raw)).strip()
210
+ if not s or not re.search(r"\\d", s):
211
+ return None
212
+ dots, commas = s.count("."), s.count(",")
213
+ if dots and commas:
214
+ s = s.replace(".", "").replace(",", ".") if s.rfind(",") > s.rfind(".") else s.replace(",", "")
215
+ elif dots > 1 or commas > 1:
216
+ s = re.sub(r"[.,]", "", s)
217
+ elif dots == 1 or commas == 1:
218
+ sep = "." if dots == 1 else ","
219
+ after = s.split(sep, 1)[1]
220
+ s = re.sub(r"[.,]", "", s) if re.fullmatch(r"\\d{3}", after) else s.replace(",", ".")
221
+ try:
222
+ return float(s)
223
+ except ValueError:
224
+ return None
225
+
226
+
227
+ def sirdas_reward(completions, ground_truth, **kwargs):
228
+ """1.0 si la respuesta coincide con el valor impreso en el documento, 0.0 si no."""
229
+ rewards = []
230
+ for completion, truth in zip(completions, ground_truth):
231
+ given = completion if isinstance(completion, str) else completion[-1]["content"]
232
+ given, truth = given.strip(), str(truth).strip()
233
+ if given == truth:
234
+ rewards.append(1.0)
235
+ continue
236
+ a, b = _to_number(truth), _to_number(given)
237
+ if a is not None and b is not None:
238
+ rewards.append(1.0 if a == b else 0.0)
239
+ else:
240
+ norm = lambda x: " ".join(x.lower().split())
241
+ rewards.append(1.0 if norm(a if a else truth) == norm(given) else 0.0)
242
+ return rewards
243
+ `;
244
+ }
package/package.json ADDED
@@ -0,0 +1,72 @@
1
+ {
2
+ "name": "@sirdaspdf/core",
3
+ "version": "0.1.0",
4
+ "description": "Sırdaş core: PDF operations, PDF→Markdown for AI and local PII anonymization. Runs in the browser and in Node; files never leave the machine.",
5
+ "keywords": [
6
+ "pdf",
7
+ "markdown",
8
+ "llm",
9
+ "anonymization",
10
+ "pii",
11
+ "privacy",
12
+ "local-first",
13
+ "dataset",
14
+ "fine-tuning",
15
+ "deduplication"
16
+ ],
17
+ "homepage": "https://sirdas.app",
18
+ "bugs": {
19
+ "url": "https://github.com/Brayan15p/SIRDAS-APP-PDF/issues"
20
+ },
21
+ "repository": {
22
+ "type": "git",
23
+ "url": "git+https://github.com/Brayan15p/SIRDAS-APP-PDF.git",
24
+ "directory": "packages/core"
25
+ },
26
+ "license": "MIT",
27
+ "author": "Sırdaş",
28
+ "type": "module",
29
+ "main": "./dist/index.js",
30
+ "types": "./dist/index.d.ts",
31
+ "exports": {
32
+ ".": {
33
+ "types": "./dist/index.d.ts",
34
+ "default": "./dist/index.js"
35
+ },
36
+ "./*": {
37
+ "types": "./dist/*.d.ts",
38
+ "default": "./dist/*.js"
39
+ }
40
+ },
41
+ "files": [
42
+ "dist",
43
+ "LICENSE"
44
+ ],
45
+ "engines": {
46
+ "node": ">=20"
47
+ },
48
+ "publishConfig": {
49
+ "access": "public"
50
+ },
51
+ "sideEffects": false,
52
+ "scripts": {
53
+ "build": "tsc -p tsconfig.json",
54
+ "test": "npm run build && node --test test/*.test.ts",
55
+ "prepublishOnly": "npm run test"
56
+ },
57
+ "dependencies": {
58
+ "pdf-lib": "1.17.1"
59
+ },
60
+ "peerDependencies": {
61
+ "pdfjs-dist": ">=4"
62
+ },
63
+ "peerDependenciesMeta": {
64
+ "pdfjs-dist": {
65
+ "optional": true
66
+ }
67
+ },
68
+ "devDependencies": {
69
+ "pdfjs-dist": "4.10.38",
70
+ "typescript": "5.9.2"
71
+ }
72
+ }