@sirdaspdf/core 0.3.0 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # @sirdaspdf/core
2
2
 
3
- The engine behind [Sırdaş](https://sirdas.app): PDF operations, PDF→Markdown for
3
+ The engine behind [Sırdaş](https://www.sirdaspdf.com): PDF operations, PDF→Markdown for
4
4
  LLMs, local anonymization of personal data, and the pieces needed to turn
5
5
  documents into a training set. Runs in the browser and in Node, with the same
6
6
  code path.
@@ -52,5 +52,5 @@ import { dedupe } from "@sirdaspdf/core/dedupe";
52
52
  Anonymization detection is rule-based and can miss unusual formats. Review
53
53
  before sharing a sensitive document.
54
54
 
55
- Full documentation: https://sirdas.app/docs/agents/core.md ·
55
+ Full documentation: https://www.sirdaspdf.com/docs/agents/core.md ·
56
56
  Source: https://github.com/Brayan15p/SIRDAS-APP-PDF · MIT
@@ -0,0 +1,8 @@
1
+ export interface TextCheckbox {
2
+ label: string;
3
+ checked: boolean;
4
+ /** La línea donde apareció, para citarla. */
5
+ line: string;
6
+ }
7
+ /** Casillas escritas como caracteres en el texto de cualquier documento. */
8
+ export declare function findTextCheckboxes(text: string): TextCheckbox[];
@@ -0,0 +1,23 @@
1
+ const MARK_ON = "☑☒✅✔✓■▣⊠";
2
+ const MARK_OFF = "☐□▢";
3
+ /** Una casilla y su etiqueta: «☒ Sí», «[x] Hipertensión», «( ) No». */
4
+ const BOX = new RegExp(String.raw `(?:([${MARK_ON}${MARK_OFF}])|\[\s*([xX✓✔*]?)\s*\]|\(\s*([xX✓✔*•]?)\s*\))\s*([^${MARK_ON}${MARK_OFF}\[\(\n]{1,60}?)(?=\s{2,}|\s*[${MARK_ON}${MARK_OFF}\[\(]|$)`, "gmu");
5
+ /** Casillas escritas como caracteres en el texto de cualquier documento. */
6
+ export function findTextCheckboxes(text) {
7
+ const out = [];
8
+ for (const line of text.split("\n")) {
9
+ // Sin al menos una marca de casilla, «(ver anexo)» o «[1]» no son casillas.
10
+ if (!new RegExp(`[${MARK_ON}${MARK_OFF}]|\\[\\s*[xX✓✔*]?\\s*\\]\\s*\\p{L}|\\(\\s*[xX✓✔*•]?\\s*\\)\\s*\\p{L}`, "u").test(line))
11
+ continue;
12
+ for (const m of line.matchAll(BOX)) {
13
+ const label = (m[4] ?? "").trim().replace(/[:;,.]$/, "");
14
+ if (!label || !/\p{L}/u.test(label))
15
+ continue;
16
+ const glyph = m[1];
17
+ const inner = m[2] ?? m[3];
18
+ const checked = glyph ? MARK_ON.includes(glyph) : !!inner;
19
+ out.push({ label, checked, line: line.trim().slice(0, 160) });
20
+ }
21
+ }
22
+ return out;
23
+ }
package/dist/dataset.js CHANGED
@@ -116,7 +116,7 @@ export function datasetCard(stats, opts) {
116
116
  "",
117
117
  `# ${opts.name ?? "Dataset"}`,
118
118
  "",
119
- "Generado con [Sırdaş](https://sirdas.app) en la máquina de quien lo creó: los documentos originales no se subieron a ningún servicio.",
119
+ "Generado con [Sırdaş](https://www.sirdaspdf.com) en la máquina de quien lo creó: los documentos originales no se subieron a ningún servicio.",
120
120
  "",
121
121
  "| | |",
122
122
  "|---|---|",
package/dist/forms.d.ts CHANGED
@@ -9,11 +9,4 @@ export interface FormField {
9
9
  page: number | null;
10
10
  }
11
11
  export declare function readFormFields(buf: ArrayBuffer | Uint8Array): Promise<FormField[]>;
12
- export interface TextCheckbox {
13
- label: string;
14
- checked: boolean;
15
- /** La línea donde apareció, para citarla. */
16
- line: string;
17
- }
18
- /** Casillas escritas como caracteres en el texto de cualquier documento. */
19
- export declare function findTextCheckboxes(text: string): TextCheckbox[];
12
+ export { findTextCheckboxes, type TextCheckbox } from "./checkboxes.js";
package/dist/forms.js CHANGED
@@ -32,26 +32,6 @@ export async function readFormFields(buf) {
32
32
  }
33
33
  return out;
34
34
  }
35
- const MARK_ON = "☑☒✅✔✓■▣⊠";
36
- const MARK_OFF = "☐□▢";
37
- /** Una casilla y su etiqueta: «☒ Sí», «[x] Hipertensión», «( ) No». */
38
- const BOX = new RegExp(String.raw `(?:([${MARK_ON}${MARK_OFF}])|\[\s*([xX✓✔*]?)\s*\]|\(\s*([xX✓✔*•]?)\s*\))\s*([^${MARK_ON}${MARK_OFF}\[\(\n]{1,60}?)(?=\s{2,}|\s*[${MARK_ON}${MARK_OFF}\[\(]|$)`, "gmu");
39
- /** Casillas escritas como caracteres en el texto de cualquier documento. */
40
- export function findTextCheckboxes(text) {
41
- const out = [];
42
- for (const line of text.split("\n")) {
43
- // Sin al menos una marca de casilla, «(ver anexo)» o «[1]» no son casillas.
44
- if (!new RegExp(`[${MARK_ON}${MARK_OFF}]|\\[\\s*[xX✓✔*]?\\s*\\]\\s*\\p{L}|\\(\\s*[xX✓✔*•]?\\s*\\)\\s*\\p{L}`, "u").test(line))
45
- continue;
46
- for (const m of line.matchAll(BOX)) {
47
- const label = (m[4] ?? "").trim().replace(/[:;,.]$/, "");
48
- if (!label || !/\p{L}/u.test(label))
49
- continue;
50
- const glyph = m[1];
51
- const inner = m[2] ?? m[3];
52
- const checked = glyph ? MARK_ON.includes(glyph) : !!inner;
53
- out.push({ label, checked, line: line.trim().slice(0, 160) });
54
- }
55
- }
56
- return out;
57
- }
35
+ // Las casillas escritas como texto viven aparte, sin pdf-lib: la web las usa en
36
+ // la primera carga y arrastrar pdf-lib (173 KB comprimido) la volvía lenta.
37
+ export { findTextCheckboxes } from "./checkboxes.js";
package/dist/index.d.ts CHANGED
@@ -26,3 +26,4 @@ export * from "./locate.js";
26
26
  export * from "./forms.js";
27
27
  export * from "./segment.js";
28
28
  export * from "./formats.js";
29
+ export * from "./xlsx.js";
package/dist/index.js CHANGED
@@ -28,3 +28,4 @@ export * from "./locate.js";
28
28
  export * from "./forms.js";
29
29
  export * from "./segment.js";
30
30
  export * from "./formats.js";
31
+ export * from "./xlsx.js";
package/dist/xlsx.d.ts ADDED
@@ -0,0 +1,18 @@
1
+ import type { Table } from "./tables.js";
2
+ export interface XlsxOptions {
3
+ /**
4
+ * Idioma del documento, para desempatar «1.250»: en español son mil
5
+ * doscientos cincuenta; en inglés, uno coma veinticinco.
6
+ */
7
+ lang?: "es" | "en";
8
+ /** Nombre de cada hoja; por defecto «Tabla 1 (pág. 3)» o «Table 1 (p. 3)». */
9
+ sheetNames?: string[];
10
+ }
11
+ /**
12
+ * Convierte un importe escrito como texto en número, o null si no lo es sin
13
+ * ambigüedad. Acepta símbolo de moneda, signo, paréntesis contables y
14
+ * separadores de miles de los dos estilos.
15
+ */
16
+ export declare function parseAmount(raw: string, lang?: "es" | "en"): number | null;
17
+ /** Un libro con una hoja por tabla. */
18
+ export declare function tablesToXlsx(tables: Table[], { lang, sheetNames: given }?: XlsxOptions): Uint8Array;
package/dist/xlsx.js ADDED
@@ -0,0 +1,161 @@
1
+ // Tablas a un libro de Excel (.xlsx) de verdad, sin dependencias nuevas.
2
+ //
3
+ // Un .xlsx es un ZIP con XML dentro (Office Open XML, ECMA-376). Se escribe lo
4
+ // mínimo que Excel, Google Sheets, Numbers y LibreOffice abren sin quejarse:
5
+ // cadenas en línea (sin tabla de cadenas compartidas), un estilo en negrita
6
+ // para la primera fila y un ancho de columna aproximado.
7
+ //
8
+ // La razón de ser frente al CSV: con CSV, Excel en español mete todo en una
9
+ // columna si el separador no coincide y deja los importes como texto. Aquí los
10
+ // importes llegan como números y se pueden sumar.
11
+ import { strToU8, zipSync } from "fflate";
12
+ const esc = (s) => s
13
+ .replace(/&/g, "&amp;")
14
+ .replace(/</g, "&lt;")
15
+ .replace(/>/g, "&gt;")
16
+ .replace(/"/g, "&quot;")
17
+ // XML 1.0 no admite caracteres de control; un PDF a veces los trae.
18
+ .replace(/[\u0000-\u0008\u000B\u000C\u000E-\u001F]/g, "");
19
+ /**
20
+ * Convierte un importe escrito como texto en número, o null si no lo es sin
21
+ * ambigüedad. Acepta símbolo de moneda, signo, paréntesis contables y
22
+ * separadores de miles de los dos estilos.
23
+ */
24
+ export function parseAmount(raw, lang = "es") {
25
+ let s = raw.trim().replace(/ /g, " ");
26
+ if (!s || s.length > 30)
27
+ return null;
28
+ let negative = false;
29
+ if (/^\(.*\)$/.test(s)) {
30
+ negative = true;
31
+ s = s.slice(1, -1).trim();
32
+ }
33
+ s = s.replace(/^(-|−)\s*/, () => ((negative = !negative), ""));
34
+ s = s.replace(/^(COP|USD|EUR|US\$|\$|€)\s*/i, "").replace(/\s*(COP|USD|EUR|%)$/i, "");
35
+ s = s.replace(/^(-|−)\s*/, () => ((negative = !negative), ""));
36
+ if (!/^\d[\d.,' ]*$/.test(s))
37
+ return null;
38
+ s = s.replace(/[' ]/g, "");
39
+ const dots = (s.match(/\./g) ?? []).length;
40
+ const commas = (s.match(/,/g) ?? []).length;
41
+ let normalized;
42
+ if (dots && commas) {
43
+ // El último separador es el decimal: 1.250.000,50 o 1,250,000.50.
44
+ const decimal = s.lastIndexOf(",") > s.lastIndexOf(".") ? "," : ".";
45
+ const thousands = decimal === "," ? "." : ",";
46
+ const [int, dec] = s.split(decimal);
47
+ if (!new RegExp(`^\\d{1,3}(\\${thousands}\\d{3})*$`).test(int) || dec === undefined || !/^\d+$/.test(dec))
48
+ return null;
49
+ normalized = `${int.split(thousands).join("")}.${dec}`;
50
+ }
51
+ else if (dots > 1 || commas > 1) {
52
+ const sep = dots > 1 ? "." : ",";
53
+ if (!new RegExp(`^\\d{1,3}(\\${sep}\\d{3})+$`).test(s))
54
+ return null;
55
+ normalized = s.split(sep).join("");
56
+ }
57
+ else if (dots === 1 || commas === 1) {
58
+ const sep = dots ? "." : ",";
59
+ const [int, dec] = s.split(sep);
60
+ // Tres cifras detrás de un único separador: miles en el estilo del idioma
61
+ // donde ese separador marca miles (el punto en español, la coma en inglés).
62
+ const isThousands = dec.length === 3 && int.length <= 3 && (lang === "es" ? sep === "." : sep === ",");
63
+ normalized = isThousands ? int + dec : `${int}.${dec}`;
64
+ }
65
+ else {
66
+ // Un número largo sin separadores es casi siempre un identificador (NIT,
67
+ // cédula, cuenta): como número, Excel lo muestra en notación científica.
68
+ if (s.length > 12 || (s.length > 1 && s.startsWith("0")))
69
+ return null;
70
+ normalized = s;
71
+ }
72
+ const n = Number(normalized);
73
+ return Number.isFinite(n) ? (negative ? -n : n) : null;
74
+ }
75
+ function colName(i) {
76
+ let s = "";
77
+ for (let n = i + 1; n > 0; n = Math.floor((n - 1) / 26))
78
+ s = String.fromCharCode(65 + ((n - 1) % 26)) + s;
79
+ return s;
80
+ }
81
+ function sheetXml(table, lang) {
82
+ const width = Math.max(table.columns, ...table.rows.map((r) => r.length), 1);
83
+ const widths = Array.from({ length: width }, (_, c) => Math.min(60, Math.max(8, ...table.rows.map((r) => (r[c] ?? "").length + 2))));
84
+ const cols = widths.map((w, i) => `<col min="${i + 1}" max="${i + 1}" width="${w}" customWidth="1"/>`).join("");
85
+ const rows = table.rows
86
+ .map((row, r) => {
87
+ const cells = row
88
+ .map((value, c) => {
89
+ const ref = `${colName(c)}${r + 1}`;
90
+ const header = r === 0 ? ' s="1"' : "";
91
+ const n = r === 0 ? null : parseAmount(value, lang);
92
+ if (n !== null)
93
+ return `<c r="${ref}"${header}><v>${n}</v></c>`;
94
+ if (!value)
95
+ return "";
96
+ return `<c r="${ref}"${header} t="inlineStr"><is><t xml:space="preserve">${esc(value)}</t></is></c>`;
97
+ })
98
+ .join("");
99
+ return `<row r="${r + 1}">${cells}</row>`;
100
+ })
101
+ .join("");
102
+ return (`<?xml version="1.0" encoding="UTF-8" standalone="yes"?>` +
103
+ `<worksheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">` +
104
+ `<sheetViews><sheetView workbookViewId="0"><pane ySplit="1" topLeftCell="A2" activePane="bottomLeft" state="frozen"/></sheetView></sheetViews>` +
105
+ `<cols>${cols}</cols><sheetData>${rows}</sheetData></worksheet>`);
106
+ }
107
+ /** Nombres de hoja válidos: máximo 31 caracteres, sin []:*?/\ y sin repetir. */
108
+ function sheetNames(tables, lang, given) {
109
+ const used = new Set();
110
+ return tables.map((t, i) => {
111
+ const base = (given?.[i] ?? (lang === "es" ? `Tabla ${i + 1} (pág. ${t.page})` : `Table ${i + 1} (p. ${t.page})`))
112
+ .replace(/[[\]:*?/\\]/g, " ")
113
+ .trim()
114
+ .slice(0, 31) || `${i + 1}`;
115
+ let name = base;
116
+ for (let n = 2; used.has(name.toLowerCase()); n++)
117
+ name = `${base.slice(0, 31 - String(n).length - 1)} ${n}`;
118
+ used.add(name.toLowerCase());
119
+ return name;
120
+ });
121
+ }
122
+ /** Un libro con una hoja por tabla. */
123
+ export function tablesToXlsx(tables, { lang = "es", sheetNames: given } = {}) {
124
+ if (!tables.length)
125
+ throw new Error("No tables to write.");
126
+ const names = sheetNames(tables, lang, given);
127
+ const files = {
128
+ "[Content_Types].xml": strToU8(`<?xml version="1.0" encoding="UTF-8" standalone="yes"?>` +
129
+ `<Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types">` +
130
+ `<Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/>` +
131
+ `<Default Extension="xml" ContentType="application/xml"/>` +
132
+ `<Override PartName="/xl/workbook.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.sheet.main+xml"/>` +
133
+ `<Override PartName="/xl/styles.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.styles+xml"/>` +
134
+ names.map((_, i) => `<Override PartName="/xl/worksheets/sheet${i + 1}.xml" ContentType="application/vnd.openxmlformats-officedocument.spreadsheetml.worksheet+xml"/>`).join("") +
135
+ `</Types>`),
136
+ "_rels/.rels": strToU8(`<?xml version="1.0" encoding="UTF-8" standalone="yes"?>` +
137
+ `<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">` +
138
+ `<Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="xl/workbook.xml"/>` +
139
+ `</Relationships>`),
140
+ "xl/workbook.xml": strToU8(`<?xml version="1.0" encoding="UTF-8" standalone="yes"?>` +
141
+ `<workbook xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main" xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships"><sheets>` +
142
+ names.map((n, i) => `<sheet name="${esc(n)}" sheetId="${i + 1}" r:id="rId${i + 1}"/>`).join("") +
143
+ `</sheets></workbook>`),
144
+ "xl/_rels/workbook.xml.rels": strToU8(`<?xml version="1.0" encoding="UTF-8" standalone="yes"?>` +
145
+ `<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">` +
146
+ names.map((_, i) => `<Relationship Id="rId${i + 1}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/worksheet" Target="worksheets/sheet${i + 1}.xml"/>`).join("") +
147
+ `<Relationship Id="rId${names.length + 1}" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/styles" Target="styles.xml"/>` +
148
+ `</Relationships>`),
149
+ "xl/styles.xml": strToU8(`<?xml version="1.0" encoding="UTF-8" standalone="yes"?>` +
150
+ `<styleSheet xmlns="http://schemas.openxmlformats.org/spreadsheetml/2006/main">` +
151
+ `<fonts count="2"><font><sz val="11"/><name val="Calibri"/></font><font><b/><sz val="11"/><name val="Calibri"/></font></fonts>` +
152
+ `<fills count="2"><fill><patternFill patternType="none"/></fill><fill><patternFill patternType="gray125"/></fill></fills>` +
153
+ `<borders count="1"><border><left/><right/><top/><bottom/><diagonal/></border></borders>` +
154
+ `<cellStyleXfs count="1"><xf numFmtId="0" fontId="0" fillId="0" borderId="0"/></cellStyleXfs>` +
155
+ `<cellXfs count="2"><xf numFmtId="0" fontId="0" fillId="0" borderId="0" xfId="0"/><xf numFmtId="0" fontId="1" fillId="0" borderId="0" xfId="0" applyFont="1"/></cellXfs>` +
156
+ `<cellStyles count="1"><cellStyle name="Normal" xfId="0" builtinId="0"/></cellStyles>` +
157
+ `</styleSheet>`),
158
+ };
159
+ tables.forEach((t, i) => (files[`xl/worksheets/sheet${i + 1}.xml`] = strToU8(sheetXml(t, lang))));
160
+ return zipSync(files, { level: 6 });
161
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@sirdaspdf/core",
3
- "version": "0.3.0",
3
+ "version": "0.3.2",
4
4
  "description": "Sırdaş core: PDF operations, PDF→Markdown for AI and local PII anonymization. Runs in the browser and in Node; files never leave the machine.",
5
5
  "keywords": [
6
6
  "pdf",
@@ -14,7 +14,7 @@
14
14
  "fine-tuning",
15
15
  "deduplication"
16
16
  ],
17
- "homepage": "https://sirdas.app",
17
+ "homepage": "https://www.sirdaspdf.com",
18
18
  "bugs": {
19
19
  "url": "https://github.com/Brayan15p/SIRDAS-APP-PDF/issues"
20
20
  },