@sirdaspdf/core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +56 -0
- package/dist/anonymize.d.ts +35 -0
- package/dist/anonymize.js +113 -0
- package/dist/chunk.d.ts +13 -0
- package/dist/chunk.js +93 -0
- package/dist/contamination.d.ts +45 -0
- package/dist/contamination.js +74 -0
- package/dist/dataset.d.ts +69 -0
- package/dist/dataset.js +95 -0
- package/dist/datasplit.d.ts +30 -0
- package/dist/datasplit.js +49 -0
- package/dist/dedupe.d.ts +62 -0
- package/dist/dedupe.js +159 -0
- package/dist/diff.d.ts +44 -0
- package/dist/diff.js +133 -0
- package/dist/errors.d.ts +12 -0
- package/dist/errors.js +24 -0
- package/dist/index.d.ts +19 -0
- package/dist/index.js +21 -0
- package/dist/lossless.d.ts +2 -0
- package/dist/lossless.js +10 -0
- package/dist/markdown.d.ts +21 -0
- package/dist/markdown.js +177 -0
- package/dist/merge.d.ts +1 -0
- package/dist/merge.js +10 -0
- package/dist/ocr.d.ts +24 -0
- package/dist/ocr.js +46 -0
- package/dist/organize.d.ts +5 -0
- package/dist/organize.js +13 -0
- package/dist/paragraphs.d.ts +68 -0
- package/dist/paragraphs.js +138 -0
- package/dist/ranges.d.ts +2 -0
- package/dist/ranges.js +23 -0
- package/dist/split.d.ts +3 -0
- package/dist/split.js +17 -0
- package/dist/tables.d.ts +44 -0
- package/dist/tables.js +186 -0
- package/dist/text.d.ts +19 -0
- package/dist/text.js +74 -0
- package/dist/verifiable.d.ts +60 -0
- package/dist/verifiable.js +244 -0
- package/package.json +72 -0
package/dist/tables.js
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
/** Extrae cada trozo de texto con sus coordenadas, sin fusionar filas. */
|
|
2
|
+
export async function extractCells(pdf, onProgress) {
|
|
3
|
+
const cells = [];
|
|
4
|
+
for (let p = 1; p <= pdf.numPages; p++) {
|
|
5
|
+
const page = await pdf.getPage(p);
|
|
6
|
+
const content = await page.getTextContent();
|
|
7
|
+
for (const it of content.items) {
|
|
8
|
+
if (!("str" in it) || !it.str.trim())
|
|
9
|
+
continue;
|
|
10
|
+
const t = it.transform;
|
|
11
|
+
cells.push({
|
|
12
|
+
text: it.str.trim(),
|
|
13
|
+
x0: t[4],
|
|
14
|
+
x1: t[4] + (it.width || 0),
|
|
15
|
+
y: Math.round(t[5] / 2) * 2,
|
|
16
|
+
page: p,
|
|
17
|
+
});
|
|
18
|
+
}
|
|
19
|
+
page.cleanup();
|
|
20
|
+
onProgress?.(p / pdf.numPages);
|
|
21
|
+
}
|
|
22
|
+
return cells;
|
|
23
|
+
}
|
|
24
|
+
/** Agrupa los trozos en filas por página y altura. */
|
|
25
|
+
function toRows(cells) {
|
|
26
|
+
const byKey = new Map();
|
|
27
|
+
for (const c of cells) {
|
|
28
|
+
const key = `${c.page}:${c.y}`;
|
|
29
|
+
const row = byKey.get(key) ?? { page: c.page, y: c.y, items: [] };
|
|
30
|
+
row.items.push(c);
|
|
31
|
+
byKey.set(key, row);
|
|
32
|
+
}
|
|
33
|
+
return [...byKey.values()]
|
|
34
|
+
.map((r) => ({ ...r, items: r.items.sort((a, b) => a.x0 - b.x0) }))
|
|
35
|
+
.sort((a, b) => (a.page !== b.page ? a.page - b.page : b.y - a.y));
|
|
36
|
+
}
|
|
37
|
+
/** ¿La posición x cae en un hueco de esta fila (o fuera de su extensión)? */
|
|
38
|
+
function isBlankAt(row, x) {
|
|
39
|
+
for (const it of row.items) {
|
|
40
|
+
if (x > it.x0 && x < it.x1)
|
|
41
|
+
return false;
|
|
42
|
+
}
|
|
43
|
+
return true;
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Ancho típico de carácter en la fila. Sirve de referencia: un espacio entre
|
|
47
|
+
* palabras mide alrededor de un tercio de carácter, mientras que el hueco
|
|
48
|
+
* entre columnas es varias veces mayor.
|
|
49
|
+
*/
|
|
50
|
+
function charWidth(row) {
|
|
51
|
+
const widths = row.items
|
|
52
|
+
.filter((i) => i.text.length > 0)
|
|
53
|
+
.map((i) => (i.x1 - i.x0) / i.text.length)
|
|
54
|
+
.filter((w) => w > 0)
|
|
55
|
+
.sort((a, b) => a - b);
|
|
56
|
+
return widths.length ? widths[Math.floor(widths.length / 2)] : 5;
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Posiciones candidatas a separador: el centro de cada hueco entre celdas que
|
|
60
|
+
* sea claramente mayor que un espacio entre palabras.
|
|
61
|
+
*/
|
|
62
|
+
function gapCentres(row) {
|
|
63
|
+
const minGap = Math.max(6, charWidth(row) * 2.5);
|
|
64
|
+
const out = [];
|
|
65
|
+
for (let i = 1; i < row.items.length; i++) {
|
|
66
|
+
const gap = row.items[i].x0 - row.items[i - 1].x1;
|
|
67
|
+
if (gap >= minGap)
|
|
68
|
+
out.push((row.items[i - 1].x1 + row.items[i].x0) / 2);
|
|
69
|
+
}
|
|
70
|
+
return out;
|
|
71
|
+
}
|
|
72
|
+
/** Corta un bloque de filas en columnas y devuelve la tabla, o null si no lo es. */
|
|
73
|
+
function buildTable(block, opts) {
|
|
74
|
+
const candidates = block.flatMap((r) => gapCentres(r)).sort((a, b) => a - b);
|
|
75
|
+
if (!candidates.length)
|
|
76
|
+
return null;
|
|
77
|
+
// Agrupamos candidatos cercanos: son el mismo corte visto en varias filas.
|
|
78
|
+
const clusters = [];
|
|
79
|
+
for (const x of candidates) {
|
|
80
|
+
const last = clusters[clusters.length - 1];
|
|
81
|
+
if (last && x - last[last.length - 1] <= 8)
|
|
82
|
+
last.push(x);
|
|
83
|
+
else
|
|
84
|
+
clusters.push([x]);
|
|
85
|
+
}
|
|
86
|
+
// Un corte vale solo si está en blanco en casi todas las filas del bloque.
|
|
87
|
+
const separators = [];
|
|
88
|
+
for (const cluster of clusters) {
|
|
89
|
+
const x = cluster.reduce((s, v) => s + v, 0) / cluster.length;
|
|
90
|
+
const support = block.filter((r) => isBlankAt(r, x)).length / block.length;
|
|
91
|
+
if (support >= opts.minSeparatorSupport)
|
|
92
|
+
separators.push(x);
|
|
93
|
+
}
|
|
94
|
+
if (separators.length < opts.minCols - 1)
|
|
95
|
+
return null;
|
|
96
|
+
const slots = separators.length + 1;
|
|
97
|
+
let rows = [];
|
|
98
|
+
for (const row of block) {
|
|
99
|
+
const cellsOfRow = Array.from({ length: slots }, () => []);
|
|
100
|
+
for (const it of row.items) {
|
|
101
|
+
const mid = (it.x0 + it.x1) / 2;
|
|
102
|
+
let col = 0;
|
|
103
|
+
while (col < separators.length && mid > separators[col])
|
|
104
|
+
col++;
|
|
105
|
+
cellsOfRow[col].push(it.text);
|
|
106
|
+
}
|
|
107
|
+
rows.push(cellsOfRow.map((parts) => parts.join(" ").trim()));
|
|
108
|
+
}
|
|
109
|
+
// Una misma banda de espacio en blanco puede producir varios cortes (los
|
|
110
|
+
// importes alineados a la derecha empiezan en una x distinta por fila), y
|
|
111
|
+
// entre ellos queda una columna vacía en todas las filas. Se descartan.
|
|
112
|
+
const used = Array.from({ length: slots }, (_, i) => rows.some((r) => r[i] !== ""));
|
|
113
|
+
const columns = used.filter(Boolean).length;
|
|
114
|
+
if (columns < opts.minCols)
|
|
115
|
+
return null;
|
|
116
|
+
rows = rows.map((r) => r.filter((_, i) => used[i]));
|
|
117
|
+
// Una "tabla" donde casi todas las filas tienen una sola celda es, en
|
|
118
|
+
// realidad, un párrafo con sangría.
|
|
119
|
+
const filled = rows.filter((r) => r.filter(Boolean).length >= 2).length;
|
|
120
|
+
if (filled / rows.length < 0.6)
|
|
121
|
+
return null;
|
|
122
|
+
return { page: block[0].page, rows, columns };
|
|
123
|
+
}
|
|
124
|
+
/** Encuentra las tablas de un documento a partir de sus trozos de texto. */
|
|
125
|
+
export function detectTables(cells, options = {}) {
|
|
126
|
+
const opts = {
|
|
127
|
+
minRows: options.minRows ?? 3,
|
|
128
|
+
minCols: options.minCols ?? 2,
|
|
129
|
+
maxRowGap: options.maxRowGap ?? 28,
|
|
130
|
+
minSeparatorSupport: options.minSeparatorSupport ?? 0.8,
|
|
131
|
+
};
|
|
132
|
+
const rows = toRows(cells);
|
|
133
|
+
const tables = [];
|
|
134
|
+
let block = [];
|
|
135
|
+
const flush = () => {
|
|
136
|
+
if (block.length >= opts.minRows) {
|
|
137
|
+
const table = buildTable(block, opts);
|
|
138
|
+
if (table)
|
|
139
|
+
tables.push(table);
|
|
140
|
+
}
|
|
141
|
+
block = [];
|
|
142
|
+
};
|
|
143
|
+
for (const row of rows) {
|
|
144
|
+
const prev = block[block.length - 1];
|
|
145
|
+
const breaks = prev !== undefined && (prev.page !== row.page || prev.y - row.y > opts.maxRowGap);
|
|
146
|
+
// Una fila de una sola celda no participa en una tabla: corta el bloque.
|
|
147
|
+
if (breaks || row.items.length < 2) {
|
|
148
|
+
flush();
|
|
149
|
+
if (row.items.length >= 2)
|
|
150
|
+
block.push(row);
|
|
151
|
+
continue;
|
|
152
|
+
}
|
|
153
|
+
block.push(row);
|
|
154
|
+
}
|
|
155
|
+
flush();
|
|
156
|
+
return tables;
|
|
157
|
+
}
|
|
158
|
+
/** Documento abierto → tablas detectadas. */
|
|
159
|
+
export async function pdfToTables(pdf, options = {}, onProgress) {
|
|
160
|
+
return detectTables(await extractCells(pdf, onProgress), options);
|
|
161
|
+
}
|
|
162
|
+
/** Escapa un valor según RFC 4180. */
|
|
163
|
+
function escapeCsv(value, delimiter) {
|
|
164
|
+
if (value.includes(delimiter) || /["\r\n]/.test(value)) {
|
|
165
|
+
return `"${value.replace(/"/g, '""')}"`;
|
|
166
|
+
}
|
|
167
|
+
return value;
|
|
168
|
+
}
|
|
169
|
+
export function tableToCsv(table, { delimiter = ",", bom = false } = {}) {
|
|
170
|
+
const body = table.rows.map((r) => r.map((c) => escapeCsv(c, delimiter)).join(delimiter)).join("\r\n");
|
|
171
|
+
return (bom ? "" : "") + body;
|
|
172
|
+
}
|
|
173
|
+
/** Tabla en Markdown, usando la primera fila como encabezado. */
|
|
174
|
+
export function tableToMarkdown(table) {
|
|
175
|
+
if (!table.rows.length)
|
|
176
|
+
return "";
|
|
177
|
+
const esc = (s) => s.replace(/\|/g, "\\|");
|
|
178
|
+
const [head, ...body] = table.rows;
|
|
179
|
+
const width = table.columns;
|
|
180
|
+
const pad = (r) => Array.from({ length: width }, (_, i) => esc(r[i] ?? ""));
|
|
181
|
+
return [
|
|
182
|
+
`| ${pad(head).join(" | ")} |`,
|
|
183
|
+
`| ${Array.from({ length: width }, () => "---").join(" | ")} |`,
|
|
184
|
+
...body.map((r) => `| ${pad(r).join(" | ")} |`),
|
|
185
|
+
].join("\n");
|
|
186
|
+
}
|
package/dist/text.d.ts
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Normaliza para comparar: minúsculas, sin acentos, sin puntuación y con los
|
|
3
|
+
* espacios colapsados. Dos textos que solo difieren en formato deben verse
|
|
4
|
+
* iguales, porque para un modelo de lenguaje lo son.
|
|
5
|
+
*/
|
|
6
|
+
export declare function normalizeForCompare(text: string): string;
|
|
7
|
+
export declare function words(text: string): string[];
|
|
8
|
+
/** FNV-1a de 32 bits: rápido, sin dependencias y estable entre ejecuciones. */
|
|
9
|
+
export declare function fnv1a(s: string): number;
|
|
10
|
+
/**
|
|
11
|
+
* Conjunto de n-gramas de palabras, como hashes. Si el texto es más corto que
|
|
12
|
+
* `n`, devuelve un único elemento con todo el texto: un documento de tres
|
|
13
|
+
* palabras sigue pudiendo ser un duplicado.
|
|
14
|
+
*/
|
|
15
|
+
export declare function ngramHashes(text: string, n: number): Set<number>;
|
|
16
|
+
/** Los n-gramas como texto, para poder enseñar cuál coincidió. */
|
|
17
|
+
export declare function ngrams(text: string, n: number): string[];
|
|
18
|
+
/** Jaccard exacto entre dos conjuntos. */
|
|
19
|
+
export declare function jaccard(a: Set<number>, b: Set<number>): number;
|
package/dist/text.js
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
// Utilidades de texto compartidas por la deduplicación y la detección de
|
|
2
|
+
// contaminación. Todo es determinista: la misma entrada da el mismo resultado
|
|
3
|
+
// en cualquier máquina, que es lo que permite auditar un dataset.
|
|
4
|
+
/**
|
|
5
|
+
* Normaliza para comparar: minúsculas, sin acentos, sin puntuación y con los
|
|
6
|
+
* espacios colapsados. Dos textos que solo difieren en formato deben verse
|
|
7
|
+
* iguales, porque para un modelo de lenguaje lo son.
|
|
8
|
+
*/
|
|
9
|
+
export function normalizeForCompare(text) {
|
|
10
|
+
return text
|
|
11
|
+
.toLowerCase()
|
|
12
|
+
.normalize("NFD")
|
|
13
|
+
.replace(/[̀-ͯ]/g, "")
|
|
14
|
+
.replace(/[^\p{L}\p{N}\s]/gu, " ")
|
|
15
|
+
.replace(/\s+/g, " ")
|
|
16
|
+
.trim();
|
|
17
|
+
}
|
|
18
|
+
export function words(text) {
|
|
19
|
+
const n = normalizeForCompare(text);
|
|
20
|
+
return n === "" ? [] : n.split(" ");
|
|
21
|
+
}
|
|
22
|
+
/** FNV-1a de 32 bits: rápido, sin dependencias y estable entre ejecuciones. */
|
|
23
|
+
export function fnv1a(s) {
|
|
24
|
+
let h = 0x811c9dc5;
|
|
25
|
+
for (let i = 0; i < s.length; i++) {
|
|
26
|
+
h ^= s.charCodeAt(i);
|
|
27
|
+
// h *= 16777619 en aritmética de 32 bits sin desbordar el doble de JS.
|
|
28
|
+
h = (h + ((h << 1) + (h << 4) + (h << 7) + (h << 8) + (h << 24))) >>> 0;
|
|
29
|
+
}
|
|
30
|
+
return h >>> 0;
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Conjunto de n-gramas de palabras, como hashes. Si el texto es más corto que
|
|
34
|
+
* `n`, devuelve un único elemento con todo el texto: un documento de tres
|
|
35
|
+
* palabras sigue pudiendo ser un duplicado.
|
|
36
|
+
*/
|
|
37
|
+
export function ngramHashes(text, n) {
|
|
38
|
+
const w = words(text);
|
|
39
|
+
const out = new Set();
|
|
40
|
+
if (w.length === 0)
|
|
41
|
+
return out;
|
|
42
|
+
if (w.length < n) {
|
|
43
|
+
out.add(fnv1a(w.join(" ")));
|
|
44
|
+
return out;
|
|
45
|
+
}
|
|
46
|
+
for (let i = 0; i + n <= w.length; i++)
|
|
47
|
+
out.add(fnv1a(w.slice(i, i + n).join(" ")));
|
|
48
|
+
return out;
|
|
49
|
+
}
|
|
50
|
+
/** Los n-gramas como texto, para poder enseñar cuál coincidió. */
|
|
51
|
+
export function ngrams(text, n) {
|
|
52
|
+
const w = words(text);
|
|
53
|
+
if (w.length === 0)
|
|
54
|
+
return [];
|
|
55
|
+
if (w.length < n)
|
|
56
|
+
return [w.join(" ")];
|
|
57
|
+
const out = [];
|
|
58
|
+
for (let i = 0; i + n <= w.length; i++)
|
|
59
|
+
out.push(w.slice(i, i + n).join(" "));
|
|
60
|
+
return out;
|
|
61
|
+
}
|
|
62
|
+
/** Jaccard exacto entre dos conjuntos. */
|
|
63
|
+
export function jaccard(a, b) {
|
|
64
|
+
if (a.size === 0 && b.size === 0)
|
|
65
|
+
return 1;
|
|
66
|
+
if (a.size === 0 || b.size === 0)
|
|
67
|
+
return 0;
|
|
68
|
+
const [small, large] = a.size <= b.size ? [a, b] : [b, a];
|
|
69
|
+
let shared = 0;
|
|
70
|
+
for (const x of small)
|
|
71
|
+
if (large.has(x))
|
|
72
|
+
shared++;
|
|
73
|
+
return shared / (a.size + b.size - shared);
|
|
74
|
+
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
import type { Table } from "./tables.js";
|
|
2
|
+
export type VerifiableKind = "lookup" | "count" | "extreme";
|
|
3
|
+
export interface VerifiablePair {
|
|
4
|
+
/** La pregunta, con la tabla incluida como contexto. */
|
|
5
|
+
prompt: string;
|
|
6
|
+
/** La respuesta correcta, copiada literalmente de la tabla. */
|
|
7
|
+
answer: string;
|
|
8
|
+
kind: VerifiableKind;
|
|
9
|
+
source: {
|
|
10
|
+
document: string;
|
|
11
|
+
page: number;
|
|
12
|
+
/** Fila y columna de donde salió la respuesta, base 1 sin contar cabecera. */
|
|
13
|
+
row?: number;
|
|
14
|
+
column?: string;
|
|
15
|
+
};
|
|
16
|
+
}
|
|
17
|
+
export interface VerifiableOptions {
|
|
18
|
+
/** Incluir la tabla en el enunciado. Sin ella la tarea no es resoluble. */
|
|
19
|
+
includeTable?: boolean;
|
|
20
|
+
/** Máximo de pares por tabla, para no ahogar el conjunto con una sola. */
|
|
21
|
+
maxPerTable?: number;
|
|
22
|
+
/** Idioma de los enunciados. */
|
|
23
|
+
lang?: "es" | "en";
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Genera pares verificables a partir de una tabla.
|
|
27
|
+
*
|
|
28
|
+
* Solo se generan preguntas cuya respuesta sea inequívoca: si una clave de fila
|
|
29
|
+
* se repite, o si una celda está vacía, esa pregunta no se emite. Un par
|
|
30
|
+
* ambiguo en un conjunto de refuerzo es peor que no tenerlo, porque castiga al
|
|
31
|
+
* modelo por acertar.
|
|
32
|
+
*/
|
|
33
|
+
export declare function tableToVerifiable(table: Table, document: string, options?: VerifiableOptions): VerifiablePair[];
|
|
34
|
+
/**
|
|
35
|
+
* Comprueba una respuesta contra la esperada.
|
|
36
|
+
*
|
|
37
|
+
* Es la función de recompensa: devuelve 1 o 0, nada intermedio. Se comparan
|
|
38
|
+
* números como números cuando ambos lo son, para que «1.120.000» y «1120000»
|
|
39
|
+
* cuenten como la misma respuesta; en otro caso se comparan como texto sin
|
|
40
|
+
* distinguir mayúsculas ni espacios de sobra.
|
|
41
|
+
*/
|
|
42
|
+
export declare function verifyAnswer(expected: string, given: string): boolean;
|
|
43
|
+
/**
|
|
44
|
+
* JSONL de pares verificables, un par por línea.
|
|
45
|
+
*
|
|
46
|
+
* El campo se llama `ground_truth` y no `answer` porque es el nombre que usa el
|
|
47
|
+
* conjunto de referencia de RLVR (allenai/RLVR-GSM-MATH-IF-Mixed-Constraints) y
|
|
48
|
+
* el que espera el ejemplo de función de recompensa de GRPOTrainer. `prompt` es
|
|
49
|
+
* la única columna que el entrenador exige; las demás le llegan a la función de
|
|
50
|
+
* recompensa como argumentos con nombre, así que la procedencia viaja gratis.
|
|
51
|
+
*/
|
|
52
|
+
export declare function verifiableToJsonl(pairs: VerifiablePair[]): string;
|
|
53
|
+
/**
|
|
54
|
+
* La función de recompensa que acompaña a ese JSONL, lista para pegar.
|
|
55
|
+
*
|
|
56
|
+
* Se entrega junto a los datos porque la mitad del trabajo de montar RLVR es
|
|
57
|
+
* escribir el verificador, y aquí ya sabemos exactamente cómo se comparan estas
|
|
58
|
+
* respuestas: es la misma regla que aplica verifyAnswer.
|
|
59
|
+
*/
|
|
60
|
+
export declare function rewardFunctionPython(): string;
|
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
import { tableToMarkdown } from "./tables.js";
|
|
2
|
+
const PHRASES = {
|
|
3
|
+
es: {
|
|
4
|
+
lookup: (col, key) => `¿Cuál es el valor de «${col}» para «${key}»?`,
|
|
5
|
+
count: () => "¿Cuántas filas de datos tiene la tabla, sin contar la cabecera?",
|
|
6
|
+
max: (col) => `¿Cuál es el valor más alto de la columna «${col}»?`,
|
|
7
|
+
min: (col) => `¿Cuál es el valor más bajo de la columna «${col}»?`,
|
|
8
|
+
context: (doc, page) => `Tabla de la página ${page} de ${doc}:`,
|
|
9
|
+
ask: "Responde únicamente con el valor, sin explicación.",
|
|
10
|
+
},
|
|
11
|
+
en: {
|
|
12
|
+
lookup: (col, key) => `What is the value of "${col}" for "${key}"?`,
|
|
13
|
+
count: () => "How many data rows does the table have, not counting the header?",
|
|
14
|
+
max: (col) => `What is the highest value in the "${col}" column?`,
|
|
15
|
+
min: (col) => `What is the lowest value in the "${col}" column?`,
|
|
16
|
+
context: (doc, page) => `Table from page ${page} of ${doc}:`,
|
|
17
|
+
ask: "Answer with the value only, no explanation.",
|
|
18
|
+
},
|
|
19
|
+
};
|
|
20
|
+
/**
|
|
21
|
+
* Lee un número escrito a la colombiana (1.120.000,50) o a la inglesa
|
|
22
|
+
* (1,120,000.50), o devuelve null si no hay número.
|
|
23
|
+
*
|
|
24
|
+
* El caso ambiguo es un único separador seguido de exactamente tres cifras:
|
|
25
|
+
* «1.120» son mil ciento veinte en Bogotá y uno coma doce en Londres. Se
|
|
26
|
+
* resuelve como millar, porque estos pares salen de facturas y extractos donde
|
|
27
|
+
* esa es la lectura correcta. No afecta a comparar una respuesta consigo misma:
|
|
28
|
+
* dos textos idénticos ya coinciden antes de llegar aquí.
|
|
29
|
+
*/
|
|
30
|
+
function parseNumber(raw) {
|
|
31
|
+
const s = raw.replace(/[^\d.,-]/g, "").trim();
|
|
32
|
+
if (s === "" || !/\d/.test(s))
|
|
33
|
+
return null;
|
|
34
|
+
const dots = (s.match(/\./g) ?? []).length;
|
|
35
|
+
const commas = (s.match(/,/g) ?? []).length;
|
|
36
|
+
let normalized;
|
|
37
|
+
if (dots > 0 && commas > 0) {
|
|
38
|
+
// Los dos presentes: el último que aparece es el decimal.
|
|
39
|
+
normalized =
|
|
40
|
+
s.lastIndexOf(",") > s.lastIndexOf(".")
|
|
41
|
+
? s.replace(/\./g, "").replace(",", ".")
|
|
42
|
+
: s.replace(/,/g, "");
|
|
43
|
+
}
|
|
44
|
+
else if (dots > 1 || commas > 1) {
|
|
45
|
+
// Repetido: solo puede ser separador de millares.
|
|
46
|
+
normalized = s.replace(/[.,]/g, "");
|
|
47
|
+
}
|
|
48
|
+
else if (dots === 1 || commas === 1) {
|
|
49
|
+
const sep = dots === 1 ? "." : ",";
|
|
50
|
+
const after = s.slice(s.indexOf(sep) + 1);
|
|
51
|
+
normalized = /^\d{3}$/.test(after)
|
|
52
|
+
? s.replace(/[.,]/g, "") // millares
|
|
53
|
+
: s.replace(",", ".");
|
|
54
|
+
}
|
|
55
|
+
else {
|
|
56
|
+
normalized = s;
|
|
57
|
+
}
|
|
58
|
+
const n = Number(normalized);
|
|
59
|
+
return Number.isFinite(n) ? n : null;
|
|
60
|
+
}
|
|
61
|
+
/** Una columna es numérica si lo son casi todas sus celdas no vacías. */
|
|
62
|
+
function numericColumn(rows, col) {
|
|
63
|
+
const values = rows.map((r) => r[col] ?? "").filter((v) => v.trim() !== "");
|
|
64
|
+
if (values.length < 2)
|
|
65
|
+
return false;
|
|
66
|
+
return values.filter((v) => parseNumber(v) !== null).length / values.length >= 0.8;
|
|
67
|
+
}
|
|
68
|
+
/**
|
|
69
|
+
* Genera pares verificables a partir de una tabla.
|
|
70
|
+
*
|
|
71
|
+
* Solo se generan preguntas cuya respuesta sea inequívoca: si una clave de fila
|
|
72
|
+
* se repite, o si una celda está vacía, esa pregunta no se emite. Un par
|
|
73
|
+
* ambiguo en un conjunto de refuerzo es peor que no tenerlo, porque castiga al
|
|
74
|
+
* modelo por acertar.
|
|
75
|
+
*/
|
|
76
|
+
export function tableToVerifiable(table, document, options = {}) {
|
|
77
|
+
const { includeTable = true, maxPerTable = 10, lang = "es" } = options;
|
|
78
|
+
const p = PHRASES[lang];
|
|
79
|
+
const rows = table.rows;
|
|
80
|
+
if (rows.length < 2)
|
|
81
|
+
return [];
|
|
82
|
+
const header = rows[0];
|
|
83
|
+
const body = rows.slice(1);
|
|
84
|
+
if (header.length < 2)
|
|
85
|
+
return [];
|
|
86
|
+
const context = includeTable
|
|
87
|
+
? `${p.context(document, table.page)}\n\n${tableToMarkdown(table)}\n\n`
|
|
88
|
+
: "";
|
|
89
|
+
const ask = (question) => `${context}${question} ${p.ask}`;
|
|
90
|
+
// Claves de fila que aparecen una sola vez: si «Consulta general» está dos
|
|
91
|
+
// veces, preguntar por ella no tiene respuesta única.
|
|
92
|
+
const keyCount = new Map();
|
|
93
|
+
for (const row of body) {
|
|
94
|
+
const key = (row[0] ?? "").trim();
|
|
95
|
+
if (key)
|
|
96
|
+
keyCount.set(key, (keyCount.get(key) ?? 0) + 1);
|
|
97
|
+
}
|
|
98
|
+
const pairs = [];
|
|
99
|
+
for (let r = 0; r < body.length && pairs.length < maxPerTable; r++) {
|
|
100
|
+
const key = (body[r][0] ?? "").trim();
|
|
101
|
+
if (!key || keyCount.get(key) !== 1)
|
|
102
|
+
continue;
|
|
103
|
+
for (let c = 1; c < header.length && pairs.length < maxPerTable; c++) {
|
|
104
|
+
const col = (header[c] ?? "").trim();
|
|
105
|
+
const value = (body[r][c] ?? "").trim();
|
|
106
|
+
if (!col || !value)
|
|
107
|
+
continue;
|
|
108
|
+
pairs.push({
|
|
109
|
+
prompt: ask(p.lookup(col, key)),
|
|
110
|
+
answer: value,
|
|
111
|
+
kind: "lookup",
|
|
112
|
+
source: { document, page: table.page, row: r + 1, column: col },
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
if (pairs.length < maxPerTable) {
|
|
117
|
+
pairs.push({
|
|
118
|
+
prompt: ask(p.count()),
|
|
119
|
+
answer: String(body.length),
|
|
120
|
+
kind: "count",
|
|
121
|
+
source: { document, page: table.page },
|
|
122
|
+
});
|
|
123
|
+
}
|
|
124
|
+
// Máximo y mínimo por columna numérica: exige leer toda la columna, no una
|
|
125
|
+
// celda, así que es una tarea de otro tipo aunque salga de la misma tabla.
|
|
126
|
+
for (let c = 1; c < header.length && pairs.length < maxPerTable; c++) {
|
|
127
|
+
if (!numericColumn(body, c))
|
|
128
|
+
continue;
|
|
129
|
+
const col = (header[c] ?? "").trim();
|
|
130
|
+
const cells = body
|
|
131
|
+
.map((row) => ({ raw: (row[c] ?? "").trim(), n: parseNumber(row[c] ?? "") }))
|
|
132
|
+
.filter((x) => x.n !== null);
|
|
133
|
+
if (cells.length < 2)
|
|
134
|
+
continue;
|
|
135
|
+
const top = cells.reduce((a, b) => (b.n > a.n ? b : a));
|
|
136
|
+
const bottom = cells.reduce((a, b) => (b.n < a.n ? b : a));
|
|
137
|
+
// Si hay empate en el extremo, la respuesta no es única: se descarta.
|
|
138
|
+
if (cells.filter((x) => x.n === top.n).length === 1) {
|
|
139
|
+
pairs.push({
|
|
140
|
+
prompt: ask(p.max(col)),
|
|
141
|
+
answer: top.raw,
|
|
142
|
+
kind: "extreme",
|
|
143
|
+
source: { document, page: table.page, column: col },
|
|
144
|
+
});
|
|
145
|
+
}
|
|
146
|
+
if (pairs.length < maxPerTable && cells.filter((x) => x.n === bottom.n).length === 1) {
|
|
147
|
+
pairs.push({
|
|
148
|
+
prompt: ask(p.min(col)),
|
|
149
|
+
answer: bottom.raw,
|
|
150
|
+
kind: "extreme",
|
|
151
|
+
source: { document, page: table.page, column: col },
|
|
152
|
+
});
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
return pairs;
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Comprueba una respuesta contra la esperada.
|
|
159
|
+
*
|
|
160
|
+
* Es la función de recompensa: devuelve 1 o 0, nada intermedio. Se comparan
|
|
161
|
+
* números como números cuando ambos lo son, para que «1.120.000» y «1120000»
|
|
162
|
+
* cuenten como la misma respuesta; en otro caso se comparan como texto sin
|
|
163
|
+
* distinguir mayúsculas ni espacios de sobra.
|
|
164
|
+
*/
|
|
165
|
+
export function verifyAnswer(expected, given) {
|
|
166
|
+
const a = expected.trim();
|
|
167
|
+
const b = given.trim();
|
|
168
|
+
if (a === b)
|
|
169
|
+
return true;
|
|
170
|
+
const na = parseNumber(a);
|
|
171
|
+
const nb = parseNumber(b);
|
|
172
|
+
if (na !== null && nb !== null)
|
|
173
|
+
return na === nb;
|
|
174
|
+
return a.toLowerCase().replace(/\s+/g, " ") === b.toLowerCase().replace(/\s+/g, " ");
|
|
175
|
+
}
|
|
176
|
+
/**
|
|
177
|
+
* JSONL de pares verificables, un par por línea.
|
|
178
|
+
*
|
|
179
|
+
* El campo se llama `ground_truth` y no `answer` porque es el nombre que usa el
|
|
180
|
+
* conjunto de referencia de RLVR (allenai/RLVR-GSM-MATH-IF-Mixed-Constraints) y
|
|
181
|
+
* el que espera el ejemplo de función de recompensa de GRPOTrainer. `prompt` es
|
|
182
|
+
* la única columna que el entrenador exige; las demás le llegan a la función de
|
|
183
|
+
* recompensa como argumentos con nombre, así que la procedencia viaja gratis.
|
|
184
|
+
*/
|
|
185
|
+
export function verifiableToJsonl(pairs) {
|
|
186
|
+
return pairs
|
|
187
|
+
.map((p) => JSON.stringify({ prompt: p.prompt, ground_truth: p.answer, kind: p.kind, sirdas: p.source }))
|
|
188
|
+
.join("\n");
|
|
189
|
+
}
|
|
190
|
+
/**
|
|
191
|
+
* La función de recompensa que acompaña a ese JSONL, lista para pegar.
|
|
192
|
+
*
|
|
193
|
+
* Se entrega junto a los datos porque la mitad del trabajo de montar RLVR es
|
|
194
|
+
* escribir el verificador, y aquí ya sabemos exactamente cómo se comparan estas
|
|
195
|
+
* respuestas: es la misma regla que aplica verifyAnswer.
|
|
196
|
+
*/
|
|
197
|
+
export function rewardFunctionPython() {
|
|
198
|
+
return `# Función de recompensa para los pares generados por \`sirdas rl\`.
|
|
199
|
+
# Uso con TRL: el dataset necesita la columna "prompt"; las demás columnas
|
|
200
|
+
# llegan aquí como argumentos con nombre.
|
|
201
|
+
#
|
|
202
|
+
# from trl import GRPOConfig, GRPOTrainer
|
|
203
|
+
# trainer = GRPOTrainer(model=..., reward_funcs=sirdas_reward, args=..., train_dataset=ds)
|
|
204
|
+
import re
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _to_number(raw):
|
|
208
|
+
"""Lee 1.120.000 (es) y 1,120,000.50 (en). Devuelve None si no hay número."""
|
|
209
|
+
s = re.sub(r"[^\\d.,-]", "", str(raw)).strip()
|
|
210
|
+
if not s or not re.search(r"\\d", s):
|
|
211
|
+
return None
|
|
212
|
+
dots, commas = s.count("."), s.count(",")
|
|
213
|
+
if dots and commas:
|
|
214
|
+
s = s.replace(".", "").replace(",", ".") if s.rfind(",") > s.rfind(".") else s.replace(",", "")
|
|
215
|
+
elif dots > 1 or commas > 1:
|
|
216
|
+
s = re.sub(r"[.,]", "", s)
|
|
217
|
+
elif dots == 1 or commas == 1:
|
|
218
|
+
sep = "." if dots == 1 else ","
|
|
219
|
+
after = s.split(sep, 1)[1]
|
|
220
|
+
s = re.sub(r"[.,]", "", s) if re.fullmatch(r"\\d{3}", after) else s.replace(",", ".")
|
|
221
|
+
try:
|
|
222
|
+
return float(s)
|
|
223
|
+
except ValueError:
|
|
224
|
+
return None
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def sirdas_reward(completions, ground_truth, **kwargs):
|
|
228
|
+
"""1.0 si la respuesta coincide con el valor impreso en el documento, 0.0 si no."""
|
|
229
|
+
rewards = []
|
|
230
|
+
for completion, truth in zip(completions, ground_truth):
|
|
231
|
+
given = completion if isinstance(completion, str) else completion[-1]["content"]
|
|
232
|
+
given, truth = given.strip(), str(truth).strip()
|
|
233
|
+
if given == truth:
|
|
234
|
+
rewards.append(1.0)
|
|
235
|
+
continue
|
|
236
|
+
a, b = _to_number(truth), _to_number(given)
|
|
237
|
+
if a is not None and b is not None:
|
|
238
|
+
rewards.append(1.0 if a == b else 0.0)
|
|
239
|
+
else:
|
|
240
|
+
norm = lambda x: " ".join(x.lower().split())
|
|
241
|
+
rewards.append(1.0 if norm(a if a else truth) == norm(given) else 0.0)
|
|
242
|
+
return rewards
|
|
243
|
+
`;
|
|
244
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@sirdaspdf/core",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Sırdaş core: PDF operations, PDF→Markdown for AI and local PII anonymization. Runs in the browser and in Node; files never leave the machine.",
|
|
5
|
+
"keywords": [
|
|
6
|
+
"pdf",
|
|
7
|
+
"markdown",
|
|
8
|
+
"llm",
|
|
9
|
+
"anonymization",
|
|
10
|
+
"pii",
|
|
11
|
+
"privacy",
|
|
12
|
+
"local-first",
|
|
13
|
+
"dataset",
|
|
14
|
+
"fine-tuning",
|
|
15
|
+
"deduplication"
|
|
16
|
+
],
|
|
17
|
+
"homepage": "https://sirdas.app",
|
|
18
|
+
"bugs": {
|
|
19
|
+
"url": "https://github.com/Brayan15p/SIRDAS-APP-PDF/issues"
|
|
20
|
+
},
|
|
21
|
+
"repository": {
|
|
22
|
+
"type": "git",
|
|
23
|
+
"url": "git+https://github.com/Brayan15p/SIRDAS-APP-PDF.git",
|
|
24
|
+
"directory": "packages/core"
|
|
25
|
+
},
|
|
26
|
+
"license": "MIT",
|
|
27
|
+
"author": "Sırdaş",
|
|
28
|
+
"type": "module",
|
|
29
|
+
"main": "./dist/index.js",
|
|
30
|
+
"types": "./dist/index.d.ts",
|
|
31
|
+
"exports": {
|
|
32
|
+
".": {
|
|
33
|
+
"types": "./dist/index.d.ts",
|
|
34
|
+
"default": "./dist/index.js"
|
|
35
|
+
},
|
|
36
|
+
"./*": {
|
|
37
|
+
"types": "./dist/*.d.ts",
|
|
38
|
+
"default": "./dist/*.js"
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
"files": [
|
|
42
|
+
"dist",
|
|
43
|
+
"LICENSE"
|
|
44
|
+
],
|
|
45
|
+
"engines": {
|
|
46
|
+
"node": ">=20"
|
|
47
|
+
},
|
|
48
|
+
"publishConfig": {
|
|
49
|
+
"access": "public"
|
|
50
|
+
},
|
|
51
|
+
"sideEffects": false,
|
|
52
|
+
"scripts": {
|
|
53
|
+
"build": "tsc -p tsconfig.json",
|
|
54
|
+
"test": "npm run build && node --test test/*.test.ts",
|
|
55
|
+
"prepublishOnly": "npm run test"
|
|
56
|
+
},
|
|
57
|
+
"dependencies": {
|
|
58
|
+
"pdf-lib": "1.17.1"
|
|
59
|
+
},
|
|
60
|
+
"peerDependencies": {
|
|
61
|
+
"pdfjs-dist": ">=4"
|
|
62
|
+
},
|
|
63
|
+
"peerDependenciesMeta": {
|
|
64
|
+
"pdfjs-dist": {
|
|
65
|
+
"optional": true
|
|
66
|
+
}
|
|
67
|
+
},
|
|
68
|
+
"devDependencies": {
|
|
69
|
+
"pdfjs-dist": "4.10.38",
|
|
70
|
+
"typescript": "5.9.2"
|
|
71
|
+
}
|
|
72
|
+
}
|