@sirdaspdf/core 0.2.0 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/dataset.d.ts +25 -1
- package/dist/dataset.js +68 -2
- package/dist/formats.d.ts +30 -0
- package/dist/formats.js +521 -0
- package/dist/forms.d.ts +19 -0
- package/dist/forms.js +57 -0
- package/dist/index.d.ts +4 -0
- package/dist/index.js +4 -0
- package/dist/locate.d.ts +13 -0
- package/dist/locate.js +35 -0
- package/dist/markdown.d.ts +6 -0
- package/dist/markdown.js +14 -2
- package/dist/segment.d.ts +10 -0
- package/dist/segment.js +46 -0
- package/package.json +3 -2
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# @sirdaspdf/core
|
|
2
2
|
|
|
3
|
-
The engine behind [Sırdaş](https://
|
|
3
|
+
The engine behind [Sırdaş](https://www.sirdaspdf.com): PDF operations, PDF→Markdown for
|
|
4
4
|
LLMs, local anonymization of personal data, and the pieces needed to turn
|
|
5
5
|
documents into a training set. Runs in the browser and in Node, with the same
|
|
6
6
|
code path.
|
|
@@ -52,5 +52,5 @@ import { dedupe } from "@sirdaspdf/core/dedupe";
|
|
|
52
52
|
Anonymization detection is rule-based and can miss unusual formats. Review
|
|
53
53
|
before sharing a sensitive document.
|
|
54
54
|
|
|
55
|
-
Full documentation: https://
|
|
55
|
+
Full documentation: https://www.sirdaspdf.com/docs/agents/core.md ·
|
|
56
56
|
Source: https://github.com/Brayan15p/SIRDAS-APP-PDF · MIT
|
package/dist/dataset.d.ts
CHANGED
|
@@ -2,7 +2,18 @@
|
|
|
2
2
|
* Formatos de salida. Los nombres de campo son los que consumen los
|
|
3
3
|
* entrenadores; cambiarlos rompe la carga del dataset, así que no se tocan.
|
|
4
4
|
*/
|
|
5
|
-
export type DatasetFormat = "text" | "chat" | "prompt-completion"
|
|
5
|
+
export type DatasetFormat = "text" | "chat" | "prompt-completion"
|
|
6
|
+
/** Stanford Alpaca: instruction / input / output. Lo usan Axolotl, LLaMA-Factory y Unsloth. */
|
|
7
|
+
| "alpaca"
|
|
8
|
+
/** ShareGPT: conversations con from human / gpt. */
|
|
9
|
+
| "sharegpt"
|
|
10
|
+
/** Documentos de LangChain: page_content + metadata. */
|
|
11
|
+
| "langchain"
|
|
12
|
+
/** TextNode de LlamaIndex: id_ + text + metadata. */
|
|
13
|
+
| "llamaindex"
|
|
14
|
+
/** Para una base vectorial: id estable + text + metadata plana. */
|
|
15
|
+
| "embeddings";
|
|
16
|
+
export declare const DATASET_FORMATS: DatasetFormat[];
|
|
6
17
|
export interface SourceDocument {
|
|
7
18
|
/** Nombre del archivo: es la clave del reparto y de la trazabilidad. */
|
|
8
19
|
name: string;
|
|
@@ -54,6 +65,19 @@ export declare function buildRecords(docs: SourceDocument[], options?: BuildOpti
|
|
|
54
65
|
* campos que no conocen, así que dejarla puesta sale gratis y permite rastrear
|
|
55
66
|
* de qué página salió cada ejemplo cuando el modelo diga algo raro.
|
|
56
67
|
*/
|
|
68
|
+
/**
|
|
69
|
+
* Tarjeta de dataset para Hugging Face (README.md con metadatos YAML). Un
|
|
70
|
+
* dataset sin tarjeta no dice de dónde salió ni qué se le quitó, y eso es justo
|
|
71
|
+
* lo que un equipo de ML necesita para confiar en él.
|
|
72
|
+
*/
|
|
73
|
+
export declare function datasetCard(stats: DatasetStats, opts: {
|
|
74
|
+
name?: string;
|
|
75
|
+
format: DatasetFormat;
|
|
76
|
+
language?: string;
|
|
77
|
+
splits?: string[];
|
|
78
|
+
anonymized: boolean;
|
|
79
|
+
deduplicated?: boolean;
|
|
80
|
+
}): string;
|
|
57
81
|
export declare function toJsonl(records: DatasetRecord[], withMeta?: boolean): string;
|
|
58
82
|
export interface DatasetStats {
|
|
59
83
|
records: number;
|
package/dist/dataset.js
CHANGED
|
@@ -5,9 +5,11 @@
|
|
|
5
5
|
// Eso exigiría mandar tus documentos a una API, que es exactamente lo que este
|
|
6
6
|
// producto existe para evitar. Aquí todo sale del documento por reglas, y cada
|
|
7
7
|
// registro dice de dónde salió.
|
|
8
|
+
import { contentId } from "./agentchunks.js";
|
|
8
9
|
import { chunkMarkdown } from "./chunk.js";
|
|
9
10
|
import { anonymize as redactPii } from "./anonymize.js";
|
|
10
11
|
import { estimateTokens } from "./markdown.js";
|
|
12
|
+
export const DATASET_FORMATS = ["text", "chat", "prompt-completion", "alpaca", "sharegpt", "langchain", "llamaindex", "embeddings"];
|
|
11
13
|
const DEFAULT_TEMPLATE = "Resume el contenido de «{heading}» del documento {document}.";
|
|
12
14
|
/** Un trozo sin encabezado no puede pedir «resume la sección “archivo.pdf”». */
|
|
13
15
|
const DEFAULT_TEMPLATE_NO_HEADING = "Resume el contenido del documento {document}.";
|
|
@@ -46,13 +48,30 @@ export function buildRecords(docs, options = {}) {
|
|
|
46
48
|
const hasHeading = (chunk.heading ?? "").trim() !== "";
|
|
47
49
|
const tpl = options.template ?? (hasHeading ? DEFAULT_TEMPLATE : DEFAULT_TEMPLATE_NO_HEADING);
|
|
48
50
|
const instruction = fillTemplate(tpl, { heading: chunk.heading ?? "", document: doc.name });
|
|
49
|
-
records.push({ data: shape(format, text, instruction, system), meta });
|
|
51
|
+
records.push({ data: shape(format, text, instruction, system, meta), meta });
|
|
50
52
|
});
|
|
51
53
|
}
|
|
52
54
|
return records;
|
|
53
55
|
}
|
|
54
|
-
function shape(format, text, instruction, system) {
|
|
56
|
+
function shape(format, text, instruction, system, meta) {
|
|
57
|
+
const metadata = meta ? { source: meta.document, chunk: meta.chunk, heading: meta.heading, tokens: meta.tokens } : {};
|
|
55
58
|
switch (format) {
|
|
59
|
+
case "alpaca":
|
|
60
|
+
return { instruction, input: "", output: text, ...(system ? { system } : {}) };
|
|
61
|
+
case "sharegpt":
|
|
62
|
+
return {
|
|
63
|
+
conversations: [
|
|
64
|
+
...(system ? [{ from: "system", value: system }] : []),
|
|
65
|
+
{ from: "human", value: instruction },
|
|
66
|
+
{ from: "gpt", value: text },
|
|
67
|
+
],
|
|
68
|
+
};
|
|
69
|
+
case "langchain":
|
|
70
|
+
return { page_content: text, metadata, type: "Document" };
|
|
71
|
+
case "llamaindex":
|
|
72
|
+
return { id_: contentId(text), text, metadata, class_name: "TextNode" };
|
|
73
|
+
case "embeddings":
|
|
74
|
+
return { id: contentId(text), text, metadata };
|
|
56
75
|
case "text":
|
|
57
76
|
return { text };
|
|
58
77
|
case "prompt-completion":
|
|
@@ -75,6 +94,53 @@ function shape(format, text, instruction, system) {
|
|
|
75
94
|
* campos que no conocen, así que dejarla puesta sale gratis y permite rastrear
|
|
76
95
|
* de qué página salió cada ejemplo cuando el modelo diga algo raro.
|
|
77
96
|
*/
|
|
97
|
+
/**
|
|
98
|
+
* Tarjeta de dataset para Hugging Face (README.md con metadatos YAML). Un
|
|
99
|
+
* dataset sin tarjeta no dice de dónde salió ni qué se le quitó, y eso es justo
|
|
100
|
+
* lo que un equipo de ML necesita para confiar en él.
|
|
101
|
+
*/
|
|
102
|
+
export function datasetCard(stats, opts) {
|
|
103
|
+
const size = stats.records < 1000 ? "n<1K" : stats.records < 10000 ? "1K<n<10K" : stats.records < 100000 ? "10K<n<100K" : "100K<n<1M";
|
|
104
|
+
const splits = opts.splits ?? ["train"];
|
|
105
|
+
return [
|
|
106
|
+
"---",
|
|
107
|
+
`language: [${opts.language ?? "es"}]`,
|
|
108
|
+
`size_categories: [${size}]`,
|
|
109
|
+
"tags: [sirdas, pdf, local-processing]",
|
|
110
|
+
"configs:",
|
|
111
|
+
" - config_name: default",
|
|
112
|
+
" data_files:",
|
|
113
|
+
...splits.map((sp) => ` - split: ${sp}
|
|
114
|
+
path: ${sp}.jsonl`),
|
|
115
|
+
"---",
|
|
116
|
+
"",
|
|
117
|
+
`# ${opts.name ?? "Dataset"}`,
|
|
118
|
+
"",
|
|
119
|
+
"Generado con [Sırdaş](https://www.sirdaspdf.com) en la máquina de quien lo creó: los documentos originales no se subieron a ningún servicio.",
|
|
120
|
+
"",
|
|
121
|
+
"| | |",
|
|
122
|
+
"|---|---|",
|
|
123
|
+
`| Formato | \`${opts.format}\` |`,
|
|
124
|
+
`| Registros | ${stats.records} |`,
|
|
125
|
+
`| Documentos de origen | ${stats.documents} |`,
|
|
126
|
+
`| Tokens (estimados) | ${stats.tokens} (mediana ${stats.medianTokens}, máx. ${stats.maxTokens}) |`,
|
|
127
|
+
`| Anonimizado | ${opts.anonymized ? `sí: ${stats.redacted} datos personales sustituidos en ${stats.recordsWithPii} registros` : "no"} |`,
|
|
128
|
+
`| Deduplicado | ${opts.deduplicated ? "sí (MinHash)" : "no"} |`,
|
|
129
|
+
`| Particiones | ${splits.join(", ")}${splits.length > 1 ? " — repartidas por documento, nunca por trozo, para que no se filtren a evaluación" : ""} |`,
|
|
130
|
+
"",
|
|
131
|
+
"## Uso",
|
|
132
|
+
"",
|
|
133
|
+
"```python",
|
|
134
|
+
"from datasets import load_dataset",
|
|
135
|
+
'ds = load_dataset("json", data_files={' + splits.map((sp) => `"${sp}": "${sp}.jsonl"`).join(", ") + "})",
|
|
136
|
+
"```",
|
|
137
|
+
"",
|
|
138
|
+
"## Límites",
|
|
139
|
+
"",
|
|
140
|
+
"La anonimización es por reglas (identificaciones, NIT, correos, teléfonos, direcciones, nombres con etiqueta) y puede dejar pasar formatos poco comunes. Revisa una muestra antes de publicar.",
|
|
141
|
+
"",
|
|
142
|
+
].join("\n");
|
|
143
|
+
}
|
|
78
144
|
export function toJsonl(records, withMeta = true) {
|
|
79
145
|
return records
|
|
80
146
|
.map((r) => JSON.stringify(withMeta ? { ...r.data, sirdas: r.meta } : r.data))
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
export type SourceFormat = "pdf" | "docx" | "xlsx" | "pptx" | "odt" | "ods" | "odp" | "epub" | "html" | "csv" | "tsv" | "md" | "txt" | "json" | "eml" | "rtf" | "image" | "unknown";
|
|
2
|
+
export interface ConvertedDocument {
|
|
3
|
+
format: SourceFormat;
|
|
4
|
+
markdown: string;
|
|
5
|
+
/** Hojas, diapositivas o capítulos: la unidad natural de cada formato. */
|
|
6
|
+
parts: {
|
|
7
|
+
title: string;
|
|
8
|
+
markdown: string;
|
|
9
|
+
}[];
|
|
10
|
+
/** Tablas ya estructuradas (Excel, CSV, tablas de Word y HTML). */
|
|
11
|
+
tables: {
|
|
12
|
+
title: string;
|
|
13
|
+
rows: string[][];
|
|
14
|
+
}[];
|
|
15
|
+
warnings: string[];
|
|
16
|
+
}
|
|
17
|
+
/** Por contenido primero (la extensión miente a menudo), por nombre después. */
|
|
18
|
+
export declare function detectFormat(bytes: Uint8Array, filename?: string): SourceFormat;
|
|
19
|
+
export declare function rowsToMarkdown(rows: string[][]): string;
|
|
20
|
+
export declare function htmlToMarkdown(html: string): {
|
|
21
|
+
markdown: string;
|
|
22
|
+
tables: string[][][];
|
|
23
|
+
};
|
|
24
|
+
/** CSV con comillas, separador autodetectado (coma, punto y coma o tabulador). */
|
|
25
|
+
export declare function parseDelimited(text: string, delimiter?: string): string[][];
|
|
26
|
+
/**
|
|
27
|
+
* Convierte cualquier formato soportado a Markdown. PDF e imágenes no pasan
|
|
28
|
+
* por aquí: necesitan PDF.js y OCR, y quien llama ya los tiene.
|
|
29
|
+
*/
|
|
30
|
+
export declare function convertToMarkdown(bytes: Uint8Array, filename?: string): ConvertedDocument;
|
package/dist/formats.js
ADDED
|
@@ -0,0 +1,521 @@
|
|
|
1
|
+
// Más allá del PDF: Word, Excel, PowerPoint, OpenDocument, HTML, EPUB, CSV,
|
|
2
|
+
// correos y texto a Markdown, con las mismas reglas que un PDF para que todo lo
|
|
3
|
+
// de después (clasificar, extraer, anonimizar, trocear) funcione igual.
|
|
4
|
+
// Los formatos de Office son ZIP con XML dentro: se leen con fflate, sin
|
|
5
|
+
// LibreOffice ni servidor. Se lee el contenido, no se reproduce la maquetación.
|
|
6
|
+
import { strFromU8, unzipSync } from "fflate";
|
|
7
|
+
const EXT = {
|
|
8
|
+
pdf: "pdf", docx: "docx", docm: "docx", xlsx: "xlsx", xlsm: "xlsx", pptx: "pptx", odt: "odt", ods: "ods", odp: "odp",
|
|
9
|
+
epub: "epub", html: "html", htm: "html", xhtml: "html", csv: "csv", tsv: "tsv", md: "md", markdown: "md", txt: "txt",
|
|
10
|
+
text: "txt", log: "txt", json: "json", jsonl: "txt", eml: "eml", rtf: "rtf", xml: "txt",
|
|
11
|
+
png: "image", jpg: "image", jpeg: "image", webp: "image", gif: "image", bmp: "image", tif: "image", tiff: "image", heic: "image",
|
|
12
|
+
};
|
|
13
|
+
/** Por contenido primero (la extensión miente a menudo), por nombre después. */
|
|
14
|
+
export function detectFormat(bytes, filename = "") {
|
|
15
|
+
const head = strFromU8(bytes.subarray(0, 8), true);
|
|
16
|
+
if (head.startsWith("%PDF"))
|
|
17
|
+
return "pdf";
|
|
18
|
+
if (bytes[0] === 0x89 && head.slice(1, 4) === "PNG")
|
|
19
|
+
return "image";
|
|
20
|
+
if (bytes[0] === 0xff && bytes[1] === 0xd8)
|
|
21
|
+
return "image";
|
|
22
|
+
if (head.startsWith("{\\rtf"))
|
|
23
|
+
return "rtf";
|
|
24
|
+
if (bytes[0] === 0x50 && bytes[1] === 0x4b) {
|
|
25
|
+
try {
|
|
26
|
+
const names = Object.keys(unzipSync(bytes, { filter: (f) => f.name === "mimetype" || f.name === "[Content_Types].xml" || /^(word|xl|ppt)\//.test(f.name) }));
|
|
27
|
+
if (names.some((n) => n.startsWith("word/")))
|
|
28
|
+
return "docx";
|
|
29
|
+
if (names.some((n) => n.startsWith("xl/")))
|
|
30
|
+
return "xlsx";
|
|
31
|
+
if (names.some((n) => n.startsWith("ppt/")))
|
|
32
|
+
return "pptx";
|
|
33
|
+
const mime = unzipSync(bytes, { filter: (f) => f.name === "mimetype" }).mimetype;
|
|
34
|
+
const m = mime ? strFromU8(mime) : "";
|
|
35
|
+
if (m.includes("opendocument.text"))
|
|
36
|
+
return "odt";
|
|
37
|
+
if (m.includes("opendocument.spreadsheet"))
|
|
38
|
+
return "ods";
|
|
39
|
+
if (m.includes("opendocument.presentation"))
|
|
40
|
+
return "odp";
|
|
41
|
+
if (m.includes("epub"))
|
|
42
|
+
return "epub";
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
/* ZIP dañado: se decide por extensión */
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
const ext = filename.toLowerCase().split(".").pop() ?? "";
|
|
49
|
+
return EXT[ext] ?? "unknown";
|
|
50
|
+
}
|
|
51
|
+
/* ---------- XML mínimo ---------- */
|
|
52
|
+
const decodeEntities = (s) => s
|
|
53
|
+
.replace(/</g, "<").replace(/>/g, ">").replace(/"/g, '"').replace(/'/g, "'").replace(/ /g, " ")
|
|
54
|
+
.replace(/&#x([0-9a-f]+);/gi, (_, h) => String.fromCodePoint(parseInt(h, 16)))
|
|
55
|
+
.replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(Number(d)))
|
|
56
|
+
.replace(/&/g, "&");
|
|
57
|
+
/** Texto de todas las etiquetas `tag` dentro de un fragmento XML, en orden. */
|
|
58
|
+
const texts = (xml, tag) => [...xml.matchAll(new RegExp(`<${tag}(?:\\s[^>]*)?>([\\s\\S]*?)</${tag}>`, "g"))].map((m) => decodeEntities(m[1]));
|
|
59
|
+
/** Atributo de una etiqueta, sin depender del orden en que el programa los escribió. */
|
|
60
|
+
const attr = (tag, name) => {
|
|
61
|
+
const m = tag.match(new RegExp(`\\s${name.replace(":", "\\:")}="([^"]*)"`));
|
|
62
|
+
return m ? decodeEntities(m[1]) : undefined;
|
|
63
|
+
};
|
|
64
|
+
/** Nivel de título por tamaño de letra frente al cuerpo, como con un PDF. */
|
|
65
|
+
function headingBySize(size, body, bold, text) {
|
|
66
|
+
if (!size || !body)
|
|
67
|
+
return 0;
|
|
68
|
+
const r = size / body;
|
|
69
|
+
const short = text.length < 90 && !/[.,;:]$/.test(text);
|
|
70
|
+
if (r >= 1.6 && short)
|
|
71
|
+
return 1;
|
|
72
|
+
if (r >= 1.3 && short)
|
|
73
|
+
return 2;
|
|
74
|
+
if (((r >= 1.12 && short) || (bold && text.length < 70 && short)) && !/^[-•]/.test(text))
|
|
75
|
+
return 3;
|
|
76
|
+
return 0;
|
|
77
|
+
}
|
|
78
|
+
const modeOf = (values) => {
|
|
79
|
+
const c = new Map();
|
|
80
|
+
for (const v of values)
|
|
81
|
+
if (v)
|
|
82
|
+
c.set(v, (c.get(v) ?? 0) + 1);
|
|
83
|
+
return [...c.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
84
|
+
};
|
|
85
|
+
const cellMd = (s) => s.replace(/\|/g, "\\|").replace(/\s*\n\s*/g, " ").trim();
|
|
86
|
+
export function rowsToMarkdown(rows) {
|
|
87
|
+
const width = Math.max(0, ...rows.map((r) => r.length));
|
|
88
|
+
if (!width)
|
|
89
|
+
return "";
|
|
90
|
+
const pad = (r) => [...r, ...Array(width - r.length).fill("")].map(cellMd);
|
|
91
|
+
const [head, ...body] = rows;
|
|
92
|
+
return [`| ${pad(head).join(" | ")} |`, `| ${Array(width).fill("---").join(" | ")} |`, ...body.map((r) => `| ${pad(r).join(" | ")} |`)].join("\n");
|
|
93
|
+
}
|
|
94
|
+
/* ---------- Word (DOCX) ---------- */
|
|
95
|
+
function docxToMarkdown(files) {
|
|
96
|
+
const xml = strFromU8(files["word/document.xml"] ?? new Uint8Array());
|
|
97
|
+
const styles = strFromU8(files["word/styles.xml"] ?? new Uint8Array());
|
|
98
|
+
// Nivel de título por estilo: «Heading1», «Ttulo1» (Word en español) o outlineLvl.
|
|
99
|
+
const headingOf = new Map();
|
|
100
|
+
for (const m of styles.matchAll(/<w:style\b[^>]*w:styleId="([^"]+)"[^>]*>([\s\S]*?)<\/w:style>/g)) {
|
|
101
|
+
const lvl = m[2].match(/<w:outlineLvl w:val="(\d)"/)?.[1] ?? m[1].match(/(?:heading|t[ií]?tulo|ttulo)\s*(\d)/i)?.[1];
|
|
102
|
+
if (lvl !== undefined)
|
|
103
|
+
headingOf.set(m[1], Math.min(6, Number(lvl) + (m[2].includes("outlineLvl") ? 1 : 0)));
|
|
104
|
+
else if (/^(title|t[ií]tulo)$/i.test(m[1]))
|
|
105
|
+
headingOf.set(m[1], 1);
|
|
106
|
+
}
|
|
107
|
+
const out = [];
|
|
108
|
+
const tables = [];
|
|
109
|
+
const body = xml.match(/<w:body>([\s\S]*)<\/w:body>/)?.[1] ?? xml;
|
|
110
|
+
const blocks = [...body.matchAll(/<w:tbl>[\s\S]*?<\/w:tbl>|<w:p[ >][\s\S]*?<\/w:p>|<w:p\/>/g)].map((m) => m[0]);
|
|
111
|
+
const sizeOf = (b) => Math.max(0, ...[...b.matchAll(/<w:sz w:val="(\d+)"/g)].map((m) => Number(m[1])));
|
|
112
|
+
// El cuerpo es el tamaño más frecuente, ponderado por cuánto texto lo usa.
|
|
113
|
+
const bodySize = modeOf(blocks.filter((b) => !b.startsWith("<w:tbl")).flatMap((b) => Array(Math.min(20, texts(b, "w:t").join("").length)).fill(sizeOf(b))));
|
|
114
|
+
for (const b of blocks) {
|
|
115
|
+
if (b.startsWith("<w:tbl")) {
|
|
116
|
+
const rows = [...b.matchAll(/<w:tr[ >][\s\S]*?<\/w:tr>/g)].map((r) => [...r[0].matchAll(/<w:tc>[\s\S]*?<\/w:tc>/g)].map((c) => texts(c[0], "w:t").join("")));
|
|
117
|
+
if (rows.length) {
|
|
118
|
+
tables.push({ title: `Tabla ${tables.length + 1}`, rows });
|
|
119
|
+
out.push(rowsToMarkdown(rows), "");
|
|
120
|
+
}
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
const text = b.replace(/<w:tab\/>/g, "<w:t>\t</w:t>").replace(/<w:br\/>/g, "<w:t>\n</w:t>");
|
|
124
|
+
let line = texts(text, "w:t").join("").replace(/[ \t]+/g, " ").trim();
|
|
125
|
+
if (!line)
|
|
126
|
+
continue;
|
|
127
|
+
const bullet = /^[•●▪◦‣∙·\-–*]\s*/.test(line) || /<w:numPr>/.test(b);
|
|
128
|
+
line = line.replace(/^[•●▪◦‣∙·–*]\s*/, "");
|
|
129
|
+
const style = b.match(/<w:pStyle w:val="([^"]+)"/)?.[1] ?? "";
|
|
130
|
+
const runs = [...b.matchAll(/<w:r>[\s\S]*?<\/w:r>/g)].map((r) => r[0]).filter((r) => texts(r, "w:t").join("").trim());
|
|
131
|
+
const bold = runs.length > 0 && runs.every((r) => /<w:b\/>|<w:b w:val="(?:1|true)"\/>/.test(r));
|
|
132
|
+
const level = headingOf.get(style) ?? (bullet ? 0 : headingBySize(sizeOf(b), bodySize, bold, line));
|
|
133
|
+
if (level)
|
|
134
|
+
out.push(`${"#".repeat(level)} ${line}`, "");
|
|
135
|
+
else if (bullet)
|
|
136
|
+
out.push(`- ${line}`);
|
|
137
|
+
else
|
|
138
|
+
out.push(line, "");
|
|
139
|
+
}
|
|
140
|
+
const markdown = out.join("\n").replace(/\n{3,}/g, "\n\n").trim();
|
|
141
|
+
return { format: "docx", markdown, parts: [{ title: "Documento", markdown }], tables, warnings: [] };
|
|
142
|
+
}
|
|
143
|
+
/* ---------- Excel (XLSX) ---------- */
|
|
144
|
+
function colIndex(ref) {
|
|
145
|
+
const letters = ref.match(/^[A-Z]+/)?.[0] ?? "A";
|
|
146
|
+
return [...letters].reduce((n, ch) => n * 26 + ch.charCodeAt(0) - 64, 0) - 1;
|
|
147
|
+
}
|
|
148
|
+
function xlsxToMarkdown(files) {
|
|
149
|
+
const read = (p) => (files[p] ? strFromU8(files[p]) : "");
|
|
150
|
+
const shared = [...read("xl/sharedStrings.xml").matchAll(/<si>([\s\S]*?)<\/si>/g)].map((m) => texts(m[1], "t").join(""));
|
|
151
|
+
const workbook = read("xl/workbook.xml");
|
|
152
|
+
const rels = read("xl/_rels/workbook.xml.rels");
|
|
153
|
+
const target = new Map([...rels.matchAll(/<Relationship\b[^>]*>/g)].map((m) => [attr(m[0], "Id") ?? "", (attr(m[0], "Target") ?? "").replace(/^\/?xl\//, "")]));
|
|
154
|
+
const sheets = [...workbook.matchAll(/<sheet\b[^>]*>/g)].map((m) => ({ name: attr(m[0], "name") ?? "Hoja", path: `xl/${target.get(attr(m[0], "r:id") ?? "") ?? ""}` }));
|
|
155
|
+
const parts = [];
|
|
156
|
+
const tables = [];
|
|
157
|
+
const warnings = [];
|
|
158
|
+
for (const sheet of sheets) {
|
|
159
|
+
const xml = read(sheet.path);
|
|
160
|
+
const rows = [];
|
|
161
|
+
for (const r of xml.matchAll(/<row\b[^>]*>([\s\S]*?)<\/row>|<row\b[^>]*\/>/g)) {
|
|
162
|
+
if (!r[1])
|
|
163
|
+
continue;
|
|
164
|
+
const row = [];
|
|
165
|
+
for (const c of r[1].matchAll(/<c\b([^>]*?)(?:\/>|>([\s\S]*?)<\/c>)/g)) {
|
|
166
|
+
const attrs = c[1];
|
|
167
|
+
const ref = attrs.match(/r="([A-Z]+)\d+"/)?.[1] ?? "";
|
|
168
|
+
const type = attrs.match(/t="([^"]+)"/)?.[1];
|
|
169
|
+
const inner = c[2] ?? "";
|
|
170
|
+
const v = inner.match(/<v>([\s\S]*?)<\/v>/)?.[1];
|
|
171
|
+
const value = type === "s" ? shared[Number(v)] ?? "" : type === "inlineStr" ? texts(inner, "t").join("") : decodeEntities(v ?? "");
|
|
172
|
+
row[ref ? colIndex(ref) : row.length] = value;
|
|
173
|
+
}
|
|
174
|
+
if (row.some((x) => x?.trim()))
|
|
175
|
+
rows.push(Array.from(row, (x) => x ?? ""));
|
|
176
|
+
}
|
|
177
|
+
if (!rows.length)
|
|
178
|
+
continue;
|
|
179
|
+
if (xml.includes("<f>"))
|
|
180
|
+
warnings.push(`«${sheet.name}» tiene fórmulas: se usa el último valor calculado que guardó Excel.`);
|
|
181
|
+
tables.push({ title: sheet.name, rows });
|
|
182
|
+
parts.push({ title: sheet.name, markdown: `## ${sheet.name}\n\n${rowsToMarkdown(rows)}` });
|
|
183
|
+
}
|
|
184
|
+
return { format: "xlsx", markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables, warnings };
|
|
185
|
+
}
|
|
186
|
+
/* ---------- PowerPoint (PPTX) ---------- */
|
|
187
|
+
function pptxToMarkdown(files) {
|
|
188
|
+
const slides = Object.keys(files)
|
|
189
|
+
.filter((n) => /^ppt\/slides\/slide\d+\.xml$/.test(n))
|
|
190
|
+
.sort((a, b) => Number(a.match(/\d+/g).pop()) - Number(b.match(/\d+/g).pop()));
|
|
191
|
+
const parts = slides.map((name, i) => {
|
|
192
|
+
const xml = strFromU8(files[name]);
|
|
193
|
+
const paragraphs = [...xml.matchAll(/<a:p>([\s\S]*?)<\/a:p>/g)].map((m) => texts(m[1], "a:t").join("").trim()).filter(Boolean);
|
|
194
|
+
const notesName = `ppt/notesSlides/notesSlide${name.match(/\d+/g).pop()}.xml`;
|
|
195
|
+
const notes = files[notesName]
|
|
196
|
+
? [...strFromU8(files[notesName]).matchAll(/<a:p>([\s\S]*?)<\/a:p>/g)].map((m) => texts(m[1], "a:t").join("").trim()).filter((t) => t && !/^\d+$/.test(t))
|
|
197
|
+
: [];
|
|
198
|
+
const [title, ...rest] = paragraphs;
|
|
199
|
+
const md = [`## Diapositiva ${i + 1}${title ? `: ${title}` : ""}`, "", ...rest.map((p) => `- ${p}`), ...(notes.length ? ["", `> Notas: ${notes.join(" ")}`] : [])].join("\n");
|
|
200
|
+
return { title: title ?? `Diapositiva ${i + 1}`, markdown: md };
|
|
201
|
+
});
|
|
202
|
+
return { format: "pptx", markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables: [], warnings: [] };
|
|
203
|
+
}
|
|
204
|
+
/* ---------- OpenDocument ---------- */
|
|
205
|
+
function odfToMarkdown(files, format) {
|
|
206
|
+
const xml = strFromU8(files["content.xml"] ?? new Uint8Array());
|
|
207
|
+
const plain = (s) => decodeEntities(s.replace(/<text:(?:s|tab)\b[^>]*\/>/g, " ").replace(/<text:line-break\/>/g, "\n").replace(/<[^>]+>/g, "")).trim();
|
|
208
|
+
if (format === "ods") {
|
|
209
|
+
const tables = [];
|
|
210
|
+
for (const t of xml.matchAll(/<table:table\b[^>]*table:name="([^"]*)"[^>]*>([\s\S]*?)<\/table:table>/g)) {
|
|
211
|
+
const rows = [...t[2].matchAll(/<table:table-row\b[^>]*>([\s\S]*?)<\/table:table-row>/g)]
|
|
212
|
+
.map((r) => [...r[1].matchAll(/<table:table-cell\b([^>]*?)(?:\/>|>([\s\S]*?)<\/table:table-cell>)/g)].flatMap((c) => {
|
|
213
|
+
const repeat = Math.min(50, Number(c[1].match(/number-columns-repeated="(\d+)"/)?.[1] ?? 1));
|
|
214
|
+
return Array(repeat).fill(plain(c[2] ?? ""));
|
|
215
|
+
}))
|
|
216
|
+
.map((r) => { while (r.length && !r[r.length - 1])
|
|
217
|
+
r.pop(); return r; })
|
|
218
|
+
.filter((r) => r.length);
|
|
219
|
+
if (rows.length)
|
|
220
|
+
tables.push({ title: decodeEntities(t[1]), rows });
|
|
221
|
+
}
|
|
222
|
+
const parts = tables.map((t) => ({ title: t.title, markdown: `## ${t.title}\n\n${rowsToMarkdown(t.rows)}` }));
|
|
223
|
+
return { format, markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables, warnings: [] };
|
|
224
|
+
}
|
|
225
|
+
// Estilos automáticos: tamaño y negrita de cada estilo de párrafo y de texto.
|
|
226
|
+
const styleInfo = new Map();
|
|
227
|
+
for (const m of xml.matchAll(/<style:style\b([^>]*)>([\s\S]*?)<\/style:style>/g)) {
|
|
228
|
+
const size = Number(m[2].match(/fo:font-size="([\d.]+)pt"/)?.[1] ?? 0);
|
|
229
|
+
styleInfo.set(attr(m[0], "style:name") ?? "", { size, bold: /fo:font-weight="bold"/.test(m[2]) });
|
|
230
|
+
}
|
|
231
|
+
const body = xml.match(/<office:(?:text|presentation)>([\s\S]*)<\/office:(?:text|presentation)>/)?.[1] ?? xml;
|
|
232
|
+
const blocks = [...body.matchAll(/<table:table\b[\s\S]*?<\/table:table>|<text:list\b[\s\S]*?<\/text:list>|<text:(h|p)\b[^>]*>[\s\S]*?<\/text:\1>|<draw:page\b[^>]*>/g)].map((m) => m[0]);
|
|
233
|
+
const paraSize = (b) => styleInfo.get(attr(b.match(/^<[^>]+>/)[0], "text:style-name") ?? "")?.size ?? 0;
|
|
234
|
+
const bodySize = modeOf(blocks.filter((b) => b.startsWith("<text:p")).map(paraSize));
|
|
235
|
+
const out = [];
|
|
236
|
+
const tables = [];
|
|
237
|
+
let slide = 0;
|
|
238
|
+
for (const b of blocks) {
|
|
239
|
+
if (b.startsWith("<draw:page")) {
|
|
240
|
+
out.push(`## Diapositiva ${++slide}`, "");
|
|
241
|
+
}
|
|
242
|
+
else if (b.startsWith("<table:table")) {
|
|
243
|
+
const rows = [...b.matchAll(/<table:table-row\b[^>]*>([\s\S]*?)<\/table:table-row>/g)].map((r) => [...r[1].matchAll(/<table:table-cell\b[^>]*?(?:\/>|>([\s\S]*?)<\/table:table-cell>)/g)].map((c) => plain(c[1] ?? "")));
|
|
244
|
+
if (rows.length) {
|
|
245
|
+
tables.push({ title: attr(b.match(/^<[^>]+>/)[0], "table:name") ?? `Tabla ${tables.length + 1}`, rows });
|
|
246
|
+
out.push(rowsToMarkdown(rows), "");
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
else if (b.startsWith("<text:list")) {
|
|
250
|
+
for (const it of b.matchAll(/<text:list-item\b[^>]*>([\s\S]*?)<\/text:list-item>/g)) {
|
|
251
|
+
const t = plain(it[1]);
|
|
252
|
+
if (t)
|
|
253
|
+
out.push(`- ${t}`);
|
|
254
|
+
}
|
|
255
|
+
out.push("");
|
|
256
|
+
}
|
|
257
|
+
else {
|
|
258
|
+
const text = plain(b);
|
|
259
|
+
if (!text)
|
|
260
|
+
continue;
|
|
261
|
+
const open = b.match(/^<[^>]+>/)[0];
|
|
262
|
+
const spans = [...b.matchAll(/<text:span\b[^>]*>/g)].map((m) => styleInfo.get(attr(m[0], "text:style-name") ?? ""));
|
|
263
|
+
const bold = styleInfo.get(attr(open, "text:style-name") ?? "")?.bold || (spans.length > 0 && spans.every((x) => x?.bold));
|
|
264
|
+
const level = b.startsWith("<text:h") ? Number(attr(open, "text:outline-level") ?? 2) : headingBySize(paraSize(b), bodySize, !!bold, text);
|
|
265
|
+
out.push(level ? `${"#".repeat(Math.min(6, level))} ${text}` : text, "");
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
const markdown = out.join("\n").replace(/\n{3,}/g, "\n\n").trim();
|
|
269
|
+
return { format, markdown, parts: [{ title: "Documento", markdown }], tables, warnings: [] };
|
|
270
|
+
}
|
|
271
|
+
/* ---------- HTML / EPUB ---------- */
|
|
272
|
+
export function htmlToMarkdown(html) {
|
|
273
|
+
const tables = [];
|
|
274
|
+
let s = html
|
|
275
|
+
.replace(/<(script|style|noscript|svg|head|nav|footer)\b[\s\S]*?<\/\1>/gi, "")
|
|
276
|
+
.replace(/<!--[\s\S]*?-->/g, "");
|
|
277
|
+
s = s.replace(/<table\b[\s\S]*?<\/table>/gi, (t) => {
|
|
278
|
+
const rows = [...t.matchAll(/<tr\b[\s\S]*?<\/tr>/gi)].map((r) => [...r[0].matchAll(/<t[hd]\b[^>]*>([\s\S]*?)<\/t[hd]>/gi)].map((c) => decodeEntities(c[1].replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim())).filter((r) => r.length);
|
|
279
|
+
if (!rows.length)
|
|
280
|
+
return "";
|
|
281
|
+
tables.push(rows);
|
|
282
|
+
return `\n\n${rowsToMarkdown(rows)}\n\n`;
|
|
283
|
+
});
|
|
284
|
+
s = s
|
|
285
|
+
.replace(/<h([1-6])\b[^>]*>([\s\S]*?)<\/h\1>/gi, (_, n, t) => `\n\n${"#".repeat(Number(n))} ${t.replace(/<[^>]+>/g, "").trim()}\n\n`)
|
|
286
|
+
.replace(/<li\b[^>]*>/gi, "\n- ")
|
|
287
|
+
.replace(/<br\s*\/?>/gi, "\n")
|
|
288
|
+
.replace(/<\/(p|div|section|article|ul|ol|blockquote|tr)>/gi, "\n\n")
|
|
289
|
+
.replace(/<(strong|b)\b[^>]*>([\s\S]*?)<\/\1>/gi, "**$2**")
|
|
290
|
+
.replace(/<a\b[^>]*href="(https?:[^"]+)"[^>]*>([\s\S]*?)<\/a>/gi, "[$2]($1)")
|
|
291
|
+
.replace(/<[^>]+>/g, "");
|
|
292
|
+
const markdown = decodeEntities(s)
|
|
293
|
+
.split("\n")
|
|
294
|
+
.map((l) => l.replace(/[ \t]+/g, " ").trim())
|
|
295
|
+
.join("\n")
|
|
296
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
297
|
+
.trim();
|
|
298
|
+
return { markdown, tables };
|
|
299
|
+
}
|
|
300
|
+
function epubToMarkdown(files) {
|
|
301
|
+
const container = strFromU8(files["META-INF/container.xml"] ?? new Uint8Array());
|
|
302
|
+
const opfPath = container.match(/full-path="([^"]+)"/)?.[1] ?? Object.keys(files).find((n) => n.endsWith(".opf")) ?? "";
|
|
303
|
+
const opf = strFromU8(files[opfPath] ?? new Uint8Array());
|
|
304
|
+
const base = opfPath.includes("/") ? opfPath.slice(0, opfPath.lastIndexOf("/") + 1) : "";
|
|
305
|
+
const manifest = new Map([...opf.matchAll(/<item\b[^>]*id="([^"]+)"[^>]*href="([^"]+)"/g)].map((m) => [m[1], m[2]]));
|
|
306
|
+
const spine = [...opf.matchAll(/<itemref\b[^>]*idref="([^"]+)"/g)].map((m) => base + decodeURIComponent(manifest.get(m[1]) ?? ""));
|
|
307
|
+
const parts = spine
|
|
308
|
+
.filter((p) => files[p])
|
|
309
|
+
.map((p, i) => {
|
|
310
|
+
const { markdown } = htmlToMarkdown(strFromU8(files[p]));
|
|
311
|
+
return { title: markdown.match(/^#+\s+(.+)$/m)?.[1] ?? `Capítulo ${i + 1}`, markdown };
|
|
312
|
+
})
|
|
313
|
+
.filter((p) => p.markdown);
|
|
314
|
+
return { format: "epub", markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables: [], warnings: [] };
|
|
315
|
+
}
|
|
316
|
+
/* ---------- texto ---------- */
|
|
317
|
+
/** CSV con comillas, separador autodetectado (coma, punto y coma o tabulador). */
|
|
318
|
+
export function parseDelimited(text, delimiter) {
|
|
319
|
+
const first = text.split("\n", 1)[0];
|
|
320
|
+
const sep = delimiter ?? [";", "\t", ","].reduce((best, d) => (first.split(d).length > first.split(best).length ? d : best), ",");
|
|
321
|
+
const rows = [];
|
|
322
|
+
let row = [];
|
|
323
|
+
let cell = "";
|
|
324
|
+
let quoted = false;
|
|
325
|
+
for (let i = 0; i < text.length; i++) {
|
|
326
|
+
const ch = text[i];
|
|
327
|
+
if (quoted) {
|
|
328
|
+
if (ch === '"' && text[i + 1] === '"') {
|
|
329
|
+
cell += '"';
|
|
330
|
+
i++;
|
|
331
|
+
}
|
|
332
|
+
else if (ch === '"')
|
|
333
|
+
quoted = false;
|
|
334
|
+
else
|
|
335
|
+
cell += ch;
|
|
336
|
+
}
|
|
337
|
+
else if (ch === '"')
|
|
338
|
+
quoted = true;
|
|
339
|
+
else if (ch === sep) {
|
|
340
|
+
row.push(cell);
|
|
341
|
+
cell = "";
|
|
342
|
+
}
|
|
343
|
+
else if (ch === "\n" || ch === "\r") {
|
|
344
|
+
if (ch === "\r" && text[i + 1] === "\n")
|
|
345
|
+
i++;
|
|
346
|
+
row.push(cell);
|
|
347
|
+
cell = "";
|
|
348
|
+
if (row.some((c) => c.trim()))
|
|
349
|
+
rows.push(row);
|
|
350
|
+
row = [];
|
|
351
|
+
}
|
|
352
|
+
else
|
|
353
|
+
cell += ch;
|
|
354
|
+
}
|
|
355
|
+
row.push(cell);
|
|
356
|
+
if (row.some((c) => c.trim()))
|
|
357
|
+
rows.push(row);
|
|
358
|
+
return rows;
|
|
359
|
+
}
|
|
360
|
+
function emlToMarkdown(raw) {
|
|
361
|
+
const [headerBlock, ...bodyParts] = raw.split(/\r?\n\r?\n/);
|
|
362
|
+
const header = (name) => headerBlock.match(new RegExp(`^${name}:\\s*(.+(?:\\r?\\n[ \\t].+)*)`, "im"))?.[1].replace(/\r?\n[ \t]+/g, " ").trim();
|
|
363
|
+
let body = bodyParts.join("\n\n");
|
|
364
|
+
const htmlPart = body.match(/Content-Type:\s*text\/html[\s\S]*?\r?\n\r?\n([\s\S]*?)(?:\r?\n--|$)/i)?.[1];
|
|
365
|
+
const textPart = body.match(/Content-Type:\s*text\/plain[\s\S]*?\r?\n\r?\n([\s\S]*?)(?:\r?\n--|$)/i)?.[1];
|
|
366
|
+
body = textPart ?? (htmlPart ? htmlToMarkdown(htmlPart).markdown : body);
|
|
367
|
+
// quoted-printable, lo más común en correos en español.
|
|
368
|
+
body = body.replace(/=\r?\n/g, "").replace(/(?:=[0-9A-F]{2})+/g, (m) => {
|
|
369
|
+
try {
|
|
370
|
+
return new TextDecoder().decode(new Uint8Array(m.match(/[0-9A-F]{2}/g).map((h) => parseInt(h, 16))));
|
|
371
|
+
}
|
|
372
|
+
catch {
|
|
373
|
+
return m;
|
|
374
|
+
}
|
|
375
|
+
});
|
|
376
|
+
const meta = [["De", header("From")], ["Para", header("To")], ["Fecha", header("Date")]].filter(([, v]) => v).map(([k, v]) => `${k}: ${v}`);
|
|
377
|
+
const markdown = [`# ${header("Subject") ?? "(sin asunto)"}`, "", ...meta, "", body.trim()].join("\n");
|
|
378
|
+
const warnings = /Content-Disposition:\s*attachment/i.test(raw) ? ["El correo trae adjuntos: se leyó solo el cuerpo."] : [];
|
|
379
|
+
return { format: "eml", markdown, parts: [{ title: header("Subject") ?? "Correo", markdown }], tables: [], warnings };
|
|
380
|
+
}
|
|
381
|
+
/** RTF por grupos: se saltan tablas de fuentes, colores, estilos e info, y se decodifica cp1252. */
|
|
382
|
+
function rtfToText(rtf) {
|
|
383
|
+
const cp1252 = new TextDecoder("windows-1252");
|
|
384
|
+
const SKIP = new Set(["fonttbl", "colortbl", "stylesheet", "info", "pict", "header", "footer", "listtable", "listoverridetable", "generator", "expandedcolortbl", "themedata", "datastore", "xmlnstbl", "latentstyles"]);
|
|
385
|
+
let out = "";
|
|
386
|
+
const stack = [];
|
|
387
|
+
let skipping = false;
|
|
388
|
+
let i = 0;
|
|
389
|
+
while (i < rtf.length) {
|
|
390
|
+
const ch = rtf[i];
|
|
391
|
+
if (ch === "{") {
|
|
392
|
+
stack.push(skipping);
|
|
393
|
+
i++;
|
|
394
|
+
if (rtf.startsWith("\\*", i))
|
|
395
|
+
skipping = true;
|
|
396
|
+
continue;
|
|
397
|
+
}
|
|
398
|
+
if (ch === "}") {
|
|
399
|
+
skipping = stack.pop() ?? false;
|
|
400
|
+
i++;
|
|
401
|
+
continue;
|
|
402
|
+
}
|
|
403
|
+
if (ch === "\\") {
|
|
404
|
+
const next = rtf[i + 1];
|
|
405
|
+
if (next === "\\" || next === "{" || next === "}") {
|
|
406
|
+
if (!skipping)
|
|
407
|
+
out += next;
|
|
408
|
+
i += 2;
|
|
409
|
+
continue;
|
|
410
|
+
}
|
|
411
|
+
// «\» al final de línea es un salto de párrafo en el RTF de macOS y de Word.
|
|
412
|
+
if (next === "\n" || next === "\r") {
|
|
413
|
+
if (!skipping)
|
|
414
|
+
out += "\n";
|
|
415
|
+
i += 2;
|
|
416
|
+
continue;
|
|
417
|
+
}
|
|
418
|
+
if (next === "'") {
|
|
419
|
+
if (!skipping)
|
|
420
|
+
out += cp1252.decode(new Uint8Array([parseInt(rtf.slice(i + 2, i + 4), 16)]));
|
|
421
|
+
i += 4;
|
|
422
|
+
continue;
|
|
423
|
+
}
|
|
424
|
+
const m = rtf.slice(i).match(/^\\([a-z]+)(-?\d+)? ?/i);
|
|
425
|
+
if (!m) {
|
|
426
|
+
i += 2;
|
|
427
|
+
continue;
|
|
428
|
+
}
|
|
429
|
+
const [whole, word, num] = m;
|
|
430
|
+
i += whole.length;
|
|
431
|
+
if (SKIP.has(word))
|
|
432
|
+
skipping = true;
|
|
433
|
+
if (skipping)
|
|
434
|
+
continue;
|
|
435
|
+
if (word === "par" || word === "line" || word === "sect" || word === "row")
|
|
436
|
+
out += "\n";
|
|
437
|
+
else if (word === "tab" || word === "cell")
|
|
438
|
+
out += "\t";
|
|
439
|
+
else if (word === "u" && num) {
|
|
440
|
+
out += String.fromCharCode(Number(num) < 0 ? Number(num) + 65536 : Number(num));
|
|
441
|
+
if (rtf[i] === "?")
|
|
442
|
+
i++;
|
|
443
|
+
}
|
|
444
|
+
continue;
|
|
445
|
+
}
|
|
446
|
+
if (ch === "\r" || ch === "\n") {
|
|
447
|
+
i++;
|
|
448
|
+
continue;
|
|
449
|
+
}
|
|
450
|
+
if (!skipping)
|
|
451
|
+
out += ch;
|
|
452
|
+
i++;
|
|
453
|
+
}
|
|
454
|
+
return out
|
|
455
|
+
.split("\n")
|
|
456
|
+
.map((l) => l.replace(/[ \t]+/g, " ").trim())
|
|
457
|
+
.join("\n")
|
|
458
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
459
|
+
.trim();
|
|
460
|
+
}
|
|
461
|
+
/**
|
|
462
|
+
* Convierte cualquier formato soportado a Markdown. PDF e imágenes no pasan
|
|
463
|
+
* por aquí: necesitan PDF.js y OCR, y quien llama ya los tiene.
|
|
464
|
+
*/
|
|
465
|
+
export function convertToMarkdown(bytes, filename = "") {
|
|
466
|
+
const format = detectFormat(bytes, filename);
|
|
467
|
+
const empty = (warning) => ({ format, markdown: "", parts: [], tables: [], warnings: [warning] });
|
|
468
|
+
const text = () => new TextDecoder("utf-8").decode(bytes).replace(/^/, "");
|
|
469
|
+
switch (format) {
|
|
470
|
+
case "docx":
|
|
471
|
+
case "xlsx":
|
|
472
|
+
case "pptx":
|
|
473
|
+
case "odt":
|
|
474
|
+
case "ods":
|
|
475
|
+
case "odp":
|
|
476
|
+
case "epub": {
|
|
477
|
+
const files = unzipSync(bytes);
|
|
478
|
+
if (format === "docx")
|
|
479
|
+
return docxToMarkdown(files);
|
|
480
|
+
if (format === "xlsx")
|
|
481
|
+
return xlsxToMarkdown(files);
|
|
482
|
+
if (format === "pptx")
|
|
483
|
+
return pptxToMarkdown(files);
|
|
484
|
+
if (format === "epub")
|
|
485
|
+
return epubToMarkdown(files);
|
|
486
|
+
return odfToMarkdown(files, format);
|
|
487
|
+
}
|
|
488
|
+
case "html": {
|
|
489
|
+
const { markdown, tables } = htmlToMarkdown(text());
|
|
490
|
+
return { format, markdown, parts: [{ title: "Página", markdown }], tables: tables.map((rows, i) => ({ title: `Tabla ${i + 1}`, rows })), warnings: [] };
|
|
491
|
+
}
|
|
492
|
+
case "csv":
|
|
493
|
+
case "tsv": {
|
|
494
|
+
const rows = parseDelimited(text(), format === "tsv" ? "\t" : undefined);
|
|
495
|
+
const markdown = rowsToMarkdown(rows);
|
|
496
|
+
return { format, markdown, parts: [{ title: filename || "Tabla", markdown }], tables: [{ title: filename || "Tabla", rows }], warnings: [] };
|
|
497
|
+
}
|
|
498
|
+
case "json": {
|
|
499
|
+
const raw = text();
|
|
500
|
+
try {
|
|
501
|
+
return { format, markdown: "```json\n" + JSON.stringify(JSON.parse(raw), null, 2) + "\n```", parts: [], tables: [], warnings: [] };
|
|
502
|
+
}
|
|
503
|
+
catch {
|
|
504
|
+
return { format: "txt", markdown: raw, parts: [], tables: [], warnings: ["JSON inválido: se leyó como texto."] };
|
|
505
|
+
}
|
|
506
|
+
}
|
|
507
|
+
case "eml":
|
|
508
|
+
return emlToMarkdown(text());
|
|
509
|
+
case "rtf":
|
|
510
|
+
return { format, markdown: rtfToText(text()), parts: [], tables: [], warnings: [] };
|
|
511
|
+
case "md":
|
|
512
|
+
case "txt":
|
|
513
|
+
return { format, markdown: text(), parts: [], tables: [], warnings: [] };
|
|
514
|
+
case "pdf":
|
|
515
|
+
return empty("PDF: usa pdfToMarkdown (necesita PDF.js).");
|
|
516
|
+
case "image":
|
|
517
|
+
return empty("Imagen: necesita OCR.");
|
|
518
|
+
default:
|
|
519
|
+
return empty("Formato no reconocido.");
|
|
520
|
+
}
|
|
521
|
+
}
|
package/dist/forms.d.ts
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
export interface FormField {
|
|
2
|
+
name: string;
|
|
3
|
+
type: "text" | "checkbox" | "radio" | "dropdown" | "list" | "signature" | "other";
|
|
4
|
+
/** Texto, opción elegida o lista de opciones elegidas; `null` si está vacío. */
|
|
5
|
+
value: string | string[] | null;
|
|
6
|
+
checked?: boolean;
|
|
7
|
+
options?: string[];
|
|
8
|
+
/** Página base 1 del primer widget, si se puede resolver. */
|
|
9
|
+
page: number | null;
|
|
10
|
+
}
|
|
11
|
+
export declare function readFormFields(buf: ArrayBuffer | Uint8Array): Promise<FormField[]>;
|
|
12
|
+
export interface TextCheckbox {
|
|
13
|
+
label: string;
|
|
14
|
+
checked: boolean;
|
|
15
|
+
/** La línea donde apareció, para citarla. */
|
|
16
|
+
line: string;
|
|
17
|
+
}
|
|
18
|
+
/** Casillas escritas como caracteres en el texto de cualquier documento. */
|
|
19
|
+
export declare function findTextCheckboxes(text: string): TextCheckbox[];
|
package/dist/forms.js
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
// Formularios: lo que LlamaParse vende como «checkboxes» y «forms», resuelto
|
|
2
|
+
// sin modelo de visión en los dos casos en que se puede hacer bien:
|
|
3
|
+
// 1. Formularios PDF rellenables (AcroForm): el valor está en el archivo.
|
|
4
|
+
// 2. Casillas dibujadas como caracteres (☐ ☑ ☒ [x] ( )): están en el texto.
|
|
5
|
+
// Una casilla marcada a mano sobre un escaneo NO entra aquí: eso sí necesita visión.
|
|
6
|
+
import { PDFCheckBox, PDFDocument, PDFDropdown, PDFOptionList, PDFRadioGroup, PDFTextField } from "pdf-lib";
|
|
7
|
+
export async function readFormFields(buf) {
|
|
8
|
+
const doc = await PDFDocument.load(buf, { updateMetadata: false });
|
|
9
|
+
const pages = doc.getPages();
|
|
10
|
+
const pageOf = (field) => {
|
|
11
|
+
const ref = field.acroField.getWidgets()[0]?.P();
|
|
12
|
+
const i = ref ? pages.findIndex((p) => p.ref === ref) : -1;
|
|
13
|
+
return i >= 0 ? i + 1 : null;
|
|
14
|
+
};
|
|
15
|
+
const out = [];
|
|
16
|
+
for (const f of doc.getForm().getFields()) {
|
|
17
|
+
const base = { name: f.getName(), page: pageOf(f) };
|
|
18
|
+
if (f instanceof PDFTextField)
|
|
19
|
+
out.push({ ...base, type: "text", value: f.getText() || null });
|
|
20
|
+
else if (f instanceof PDFCheckBox)
|
|
21
|
+
out.push({ ...base, type: "checkbox", value: f.isChecked() ? "on" : null, checked: f.isChecked() });
|
|
22
|
+
else if (f instanceof PDFRadioGroup)
|
|
23
|
+
out.push({ ...base, type: "radio", value: f.getSelected() ?? null, options: f.getOptions() });
|
|
24
|
+
else if (f instanceof PDFDropdown)
|
|
25
|
+
out.push({ ...base, type: "dropdown", value: f.getSelected()[0] ?? null, options: f.getOptions() });
|
|
26
|
+
else if (f instanceof PDFOptionList)
|
|
27
|
+
out.push({ ...base, type: "list", value: f.getSelected(), options: f.getOptions() });
|
|
28
|
+
else if (f.constructor.name.includes("Signature"))
|
|
29
|
+
out.push({ ...base, type: "signature", value: null });
|
|
30
|
+
else
|
|
31
|
+
out.push({ ...base, type: "other", value: null });
|
|
32
|
+
}
|
|
33
|
+
return out;
|
|
34
|
+
}
|
|
35
|
+
const MARK_ON = "☑☒✅✔✓■▣⊠";
|
|
36
|
+
const MARK_OFF = "☐□▢";
|
|
37
|
+
/** Una casilla y su etiqueta: «☒ Sí», «[x] Hipertensión», «( ) No». */
|
|
38
|
+
const BOX = new RegExp(String.raw `(?:([${MARK_ON}${MARK_OFF}])|\[\s*([xX✓✔*]?)\s*\]|\(\s*([xX✓✔*•]?)\s*\))\s*([^${MARK_ON}${MARK_OFF}\[\(\n]{1,60}?)(?=\s{2,}|\s*[${MARK_ON}${MARK_OFF}\[\(]|$)`, "gmu");
|
|
39
|
+
/** Casillas escritas como caracteres en el texto de cualquier documento. */
|
|
40
|
+
export function findTextCheckboxes(text) {
|
|
41
|
+
const out = [];
|
|
42
|
+
for (const line of text.split("\n")) {
|
|
43
|
+
// Sin al menos una marca de casilla, «(ver anexo)» o «[1]» no son casillas.
|
|
44
|
+
if (!new RegExp(`[${MARK_ON}${MARK_OFF}]|\\[\\s*[xX✓✔*]?\\s*\\]\\s*\\p{L}|\\(\\s*[xX✓✔*•]?\\s*\\)\\s*\\p{L}`, "u").test(line))
|
|
45
|
+
continue;
|
|
46
|
+
for (const m of line.matchAll(BOX)) {
|
|
47
|
+
const label = (m[4] ?? "").trim().replace(/[:;,.]$/, "");
|
|
48
|
+
if (!label || !/\p{L}/u.test(label))
|
|
49
|
+
continue;
|
|
50
|
+
const glyph = m[1];
|
|
51
|
+
const inner = m[2] ?? m[3];
|
|
52
|
+
const checked = glyph ? MARK_ON.includes(glyph) : !!inner;
|
|
53
|
+
out.push({ label, checked, line: line.trim().slice(0, 160) });
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return out;
|
|
57
|
+
}
|
package/dist/index.d.ts
CHANGED
package/dist/index.js
CHANGED
package/dist/locate.d.ts
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import type { Line } from "./markdown.js";
|
|
2
|
+
export interface Location {
|
|
3
|
+
page: number;
|
|
4
|
+
/** Caja 0–1, origen arriba a la izquierda: [x0, y0, x1, y1]. */
|
|
5
|
+
bbox: [number, number, number, number];
|
|
6
|
+
/** Líneas que abarca la cita (una cita puede cruzar un salto de línea). */
|
|
7
|
+
lines: number;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Ubica `text` entre las líneas extraídas. Tolera diferencias de espacios,
|
|
11
|
+
* tildes y puntuación, y citas que ocupan hasta tres líneas seguidas.
|
|
12
|
+
*/
|
|
13
|
+
export declare function locateText(lines: Line[], text: string, page?: number | null): Location | null;
|
package/dist/locate.js
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
const norm = (s) => s.toLowerCase().normalize("NFD").replace(/\p{Diacritic}/gu, "").replace(/[^\p{L}\p{N}]+/gu, " ").trim();
|
|
2
|
+
/**
|
|
3
|
+
* Ubica `text` entre las líneas extraídas. Tolera diferencias de espacios,
|
|
4
|
+
* tildes y puntuación, y citas que ocupan hasta tres líneas seguidas.
|
|
5
|
+
*/
|
|
6
|
+
export function locateText(lines, text, page) {
|
|
7
|
+
const needle = norm(text);
|
|
8
|
+
if (needle.length < 3)
|
|
9
|
+
return null;
|
|
10
|
+
const pool = lines.map((l, i) => ({ l, i })).filter(({ l }) => l.bbox && (!page || l.page === page));
|
|
11
|
+
for (let span = 1; span <= 3; span++) {
|
|
12
|
+
for (let k = 0; k + span <= pool.length; k++) {
|
|
13
|
+
const group = pool.slice(k, k + span);
|
|
14
|
+
if (group.some((g) => g.l.page !== group[0].l.page))
|
|
15
|
+
continue;
|
|
16
|
+
if (span > 1 && group.some((g, j) => j > 0 && g.i !== group[j - 1].i + 1))
|
|
17
|
+
continue;
|
|
18
|
+
const hay = norm(group.map((g) => g.l.text).join(" "));
|
|
19
|
+
if (!hay.includes(needle))
|
|
20
|
+
continue;
|
|
21
|
+
const boxes = group.map((g) => g.l.bbox);
|
|
22
|
+
return {
|
|
23
|
+
page: group[0].l.page,
|
|
24
|
+
bbox: [
|
|
25
|
+
Math.min(...boxes.map((b) => b[0])),
|
|
26
|
+
Math.min(...boxes.map((b) => b[1])),
|
|
27
|
+
Math.max(...boxes.map((b) => b[2])),
|
|
28
|
+
Math.max(...boxes.map((b) => b[3])),
|
|
29
|
+
],
|
|
30
|
+
lines: span,
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
return null;
|
|
35
|
+
}
|
package/dist/markdown.d.ts
CHANGED
|
@@ -8,6 +8,12 @@ export interface Line {
|
|
|
8
8
|
bold: boolean;
|
|
9
9
|
/** Posición vertical relativa: 0 = pie de página, 1 = parte superior. */
|
|
10
10
|
rel: number;
|
|
11
|
+
/**
|
|
12
|
+
* Caja de la línea normalizada 0–1 con origen arriba a la izquierda
|
|
13
|
+
* [x0, y0, x1, y1]: lo que necesita un visor para resaltar una cita sin
|
|
14
|
+
* saber nada de coordenadas PDF. Ausente en líneas que no vienen de un PDF.
|
|
15
|
+
*/
|
|
16
|
+
bbox?: [number, number, number, number];
|
|
11
17
|
}
|
|
12
18
|
type Progress = (ratio: number) => void;
|
|
13
19
|
/** Extrae las líneas de texto con tamaño y posición de un documento abierto. */
|
package/dist/markdown.js
CHANGED
|
@@ -23,7 +23,9 @@ export async function extractLines(pdf, onProgress) {
|
|
|
23
23
|
for (let p = 1; p <= pdf.numPages; p++) {
|
|
24
24
|
const page = await pdf.getPage(p);
|
|
25
25
|
const content = await page.getTextContent();
|
|
26
|
-
const
|
|
26
|
+
const viewport = page.getViewport({ scale: 1 });
|
|
27
|
+
const pageHeight = viewport.height || 792;
|
|
28
|
+
const pageWidth = viewport.width || 612;
|
|
27
29
|
const styles = content.styles;
|
|
28
30
|
const rows = new Map();
|
|
29
31
|
for (const it of content.items) {
|
|
@@ -55,7 +57,17 @@ export async function extractLines(pdf, onProgress) {
|
|
|
55
57
|
continue;
|
|
56
58
|
const size = Math.max(...row.map((r) => r.size));
|
|
57
59
|
const bold = row.filter((r) => r.bold).reduce((s, r) => s + r.str.length, 0) > text.length * 0.6;
|
|
58
|
-
|
|
60
|
+
const x0 = row[0].x;
|
|
61
|
+
const x1 = Math.max(...row.map((r) => r.x + r.w));
|
|
62
|
+
const r4 = (n) => Math.round(Math.min(1, Math.max(0, n)) * 10000) / 10000;
|
|
63
|
+
// y es la línea base: la caja sube un cuerpo de letra y baja lo que ocupan los descendentes.
|
|
64
|
+
const bbox = [
|
|
65
|
+
r4(x0 / pageWidth),
|
|
66
|
+
r4(1 - (y + size * 0.9) / pageHeight),
|
|
67
|
+
r4(x1 / pageWidth),
|
|
68
|
+
r4(1 - (y - size * 0.25) / pageHeight),
|
|
69
|
+
];
|
|
70
|
+
lines.push({ text, size, y, x: x0, page: p, bold, rel: y / pageHeight, bbox });
|
|
59
71
|
}
|
|
60
72
|
page.cleanup();
|
|
61
73
|
onProgress?.(p / pdf.numPages);
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import { type DocType } from "./classify.js";
|
|
2
|
+
export interface Segment {
|
|
3
|
+
/** Páginas base 1, inclusivas. */
|
|
4
|
+
from: number;
|
|
5
|
+
to: number;
|
|
6
|
+
type: DocType;
|
|
7
|
+
/** Por qué empieza aquí un documento nuevo (vacío en el primero). */
|
|
8
|
+
reasons: string[];
|
|
9
|
+
}
|
|
10
|
+
export declare function segmentPages(pages: string[]): Segment[];
|
package/dist/segment.js
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
// Separar un PDF que en realidad son varios documentos pegados: el lote que
|
|
2
|
+
// sale del escáner de la recepción, un expediente con factura + historia +
|
|
3
|
+
// autorización. Sin modelo: se buscan los cortes que una persona vería.
|
|
4
|
+
import { classifyDocument } from "./classify.js";
|
|
5
|
+
/** «Página 1 de 5», «Pág. 1/3», «Page 1 of 2», «1 de 4» al pie. */
|
|
6
|
+
const FIRST_PAGE = /\b(?:p[aá]g(?:ina)?\.?|page)\s*1\s*(?:de|of|\/)\s*\d+\b/i;
|
|
7
|
+
const OTHER_PAGE = /\b(?:p[aá]g(?:ina)?\.?|page)\s*([2-9]|\d{2,})\s*(?:de|of|\/)\s*\d+\b/i;
|
|
8
|
+
/** Encabezados con los que suele abrir un documento. */
|
|
9
|
+
const OPENING = /^(?:#+\s*)?(factura(?:\s+electr[oó]nica)?(?:\s+de\s+venta)?|historia\s+cl[ií]nica|epicrisis|certificado|certifica(?:ci[oó]n)?|contrato\s+de|registro\s+[uú]nico\s+tributario|extracto|autorizaci[oó]n|orden\s+m[eé]dica|f[oó]rmula\s+m[eé]dica|acta\s+de|poder\s+especial|constancia|cuenta\s+de\s+cobro|resultados?\s+de\s+laboratorio)\b/im;
|
|
10
|
+
export function segmentPages(pages) {
|
|
11
|
+
if (!pages.length)
|
|
12
|
+
return [];
|
|
13
|
+
const kinds = pages.map((t) => classifyDocument(t));
|
|
14
|
+
const segments = [];
|
|
15
|
+
let current = { from: 1, to: 1, type: kinds[0].type, reasons: [] };
|
|
16
|
+
for (let i = 1; i < pages.length; i++) {
|
|
17
|
+
const text = pages[i];
|
|
18
|
+
const head = text.split("\n").slice(0, 8).join("\n");
|
|
19
|
+
const reasons = [];
|
|
20
|
+
const prevType = current.type;
|
|
21
|
+
const k = kinds[i];
|
|
22
|
+
if (FIRST_PAGE.test(text))
|
|
23
|
+
reasons.push("numeración reinicia en «página 1»");
|
|
24
|
+
if (OPENING.test(head))
|
|
25
|
+
reasons.push(`abre con «${head.match(OPENING)[1]}»`);
|
|
26
|
+
// Un cambio de tipo solo cuenta con señales fuertes en ambos lados: una
|
|
27
|
+
// página de anexos sin palabras clave no debe partir un contrato.
|
|
28
|
+
if (k.type !== "desconocido" && prevType !== "desconocido" && k.type !== prevType && k.confidence >= 0.6) {
|
|
29
|
+
reasons.push(`cambia de ${prevType} a ${k.type}`);
|
|
30
|
+
}
|
|
31
|
+
// «Página 3 de 5» contradice un corte: sigue el mismo documento.
|
|
32
|
+
const continues = OTHER_PAGE.test(text) && !FIRST_PAGE.test(text);
|
|
33
|
+
const cut = !continues && (reasons.length >= 2 || reasons.some((r) => r.startsWith("numeración")));
|
|
34
|
+
if (cut) {
|
|
35
|
+
segments.push(current);
|
|
36
|
+
current = { from: i + 1, to: i + 1, type: k.type, reasons };
|
|
37
|
+
}
|
|
38
|
+
else {
|
|
39
|
+
current.to = i + 1;
|
|
40
|
+
if (current.type === "desconocido" && k.type !== "desconocido")
|
|
41
|
+
current.type = k.type;
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
segments.push(current);
|
|
45
|
+
return segments;
|
|
46
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@sirdaspdf/core",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.1",
|
|
4
4
|
"description": "Sırdaş core: PDF operations, PDF→Markdown for AI and local PII anonymization. Runs in the browser and in Node; files never leave the machine.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pdf",
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
"fine-tuning",
|
|
15
15
|
"deduplication"
|
|
16
16
|
],
|
|
17
|
-
"homepage": "https://
|
|
17
|
+
"homepage": "https://www.sirdaspdf.com",
|
|
18
18
|
"bugs": {
|
|
19
19
|
"url": "https://github.com/Brayan15p/SIRDAS-APP-PDF/issues"
|
|
20
20
|
},
|
|
@@ -55,6 +55,7 @@
|
|
|
55
55
|
"prepublishOnly": "npm run test"
|
|
56
56
|
},
|
|
57
57
|
"dependencies": {
|
|
58
|
+
"fflate": "0.8.3",
|
|
58
59
|
"pdf-lib": "1.17.1"
|
|
59
60
|
},
|
|
60
61
|
"peerDependencies": {
|