@sirdaspdf/core 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agentchunks.d.ts +18 -0
- package/dist/agentchunks.js +85 -0
- package/dist/anonymize.js +55 -7
- package/dist/classify.d.ts +12 -0
- package/dist/classify.js +89 -0
- package/dist/dataset.d.ts +25 -1
- package/dist/dataset.js +68 -2
- package/dist/extract.d.ts +16 -0
- package/dist/extract.js +103 -0
- package/dist/formats.d.ts +30 -0
- package/dist/formats.js +521 -0
- package/dist/forms.d.ts +19 -0
- package/dist/forms.js +57 -0
- package/dist/index.d.ts +9 -0
- package/dist/index.js +9 -0
- package/dist/locate.d.ts +13 -0
- package/dist/locate.js +35 -0
- package/dist/markdown.d.ts +16 -2
- package/dist/markdown.js +25 -6
- package/dist/ocr.js +34 -4
- package/dist/pages.d.ts +24 -0
- package/dist/pages.js +92 -0
- package/dist/security.d.ts +40 -0
- package/dist/security.js +100 -0
- package/dist/segment.d.ts +10 -0
- package/dist/segment.js +46 -0
- package/package.json +3 -1
package/dist/formats.js
ADDED
|
@@ -0,0 +1,521 @@
|
|
|
1
|
+
// Más allá del PDF: Word, Excel, PowerPoint, OpenDocument, HTML, EPUB, CSV,
|
|
2
|
+
// correos y texto a Markdown, con las mismas reglas que un PDF para que todo lo
|
|
3
|
+
// de después (clasificar, extraer, anonimizar, trocear) funcione igual.
|
|
4
|
+
// Los formatos de Office son ZIP con XML dentro: se leen con fflate, sin
|
|
5
|
+
// LibreOffice ni servidor. Se lee el contenido, no se reproduce la maquetación.
|
|
6
|
+
import { strFromU8, unzipSync } from "fflate";
|
|
7
|
+
const EXT = {
|
|
8
|
+
pdf: "pdf", docx: "docx", docm: "docx", xlsx: "xlsx", xlsm: "xlsx", pptx: "pptx", odt: "odt", ods: "ods", odp: "odp",
|
|
9
|
+
epub: "epub", html: "html", htm: "html", xhtml: "html", csv: "csv", tsv: "tsv", md: "md", markdown: "md", txt: "txt",
|
|
10
|
+
text: "txt", log: "txt", json: "json", jsonl: "txt", eml: "eml", rtf: "rtf", xml: "txt",
|
|
11
|
+
png: "image", jpg: "image", jpeg: "image", webp: "image", gif: "image", bmp: "image", tif: "image", tiff: "image", heic: "image",
|
|
12
|
+
};
|
|
13
|
+
/** Por contenido primero (la extensión miente a menudo), por nombre después. */
|
|
14
|
+
export function detectFormat(bytes, filename = "") {
|
|
15
|
+
const head = strFromU8(bytes.subarray(0, 8), true);
|
|
16
|
+
if (head.startsWith("%PDF"))
|
|
17
|
+
return "pdf";
|
|
18
|
+
if (bytes[0] === 0x89 && head.slice(1, 4) === "PNG")
|
|
19
|
+
return "image";
|
|
20
|
+
if (bytes[0] === 0xff && bytes[1] === 0xd8)
|
|
21
|
+
return "image";
|
|
22
|
+
if (head.startsWith("{\\rtf"))
|
|
23
|
+
return "rtf";
|
|
24
|
+
if (bytes[0] === 0x50 && bytes[1] === 0x4b) {
|
|
25
|
+
try {
|
|
26
|
+
const names = Object.keys(unzipSync(bytes, { filter: (f) => f.name === "mimetype" || f.name === "[Content_Types].xml" || /^(word|xl|ppt)\//.test(f.name) }));
|
|
27
|
+
if (names.some((n) => n.startsWith("word/")))
|
|
28
|
+
return "docx";
|
|
29
|
+
if (names.some((n) => n.startsWith("xl/")))
|
|
30
|
+
return "xlsx";
|
|
31
|
+
if (names.some((n) => n.startsWith("ppt/")))
|
|
32
|
+
return "pptx";
|
|
33
|
+
const mime = unzipSync(bytes, { filter: (f) => f.name === "mimetype" }).mimetype;
|
|
34
|
+
const m = mime ? strFromU8(mime) : "";
|
|
35
|
+
if (m.includes("opendocument.text"))
|
|
36
|
+
return "odt";
|
|
37
|
+
if (m.includes("opendocument.spreadsheet"))
|
|
38
|
+
return "ods";
|
|
39
|
+
if (m.includes("opendocument.presentation"))
|
|
40
|
+
return "odp";
|
|
41
|
+
if (m.includes("epub"))
|
|
42
|
+
return "epub";
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
/* ZIP dañado: se decide por extensión */
|
|
46
|
+
}
|
|
47
|
+
}
|
|
48
|
+
const ext = filename.toLowerCase().split(".").pop() ?? "";
|
|
49
|
+
return EXT[ext] ?? "unknown";
|
|
50
|
+
}
|
|
51
|
+
/* ---------- XML mínimo ---------- */
|
|
52
|
+
const decodeEntities = (s) => s
|
|
53
|
+
.replace(/</g, "<").replace(/>/g, ">").replace(/"/g, '"').replace(/'/g, "'").replace(/ /g, " ")
|
|
54
|
+
.replace(/&#x([0-9a-f]+);/gi, (_, h) => String.fromCodePoint(parseInt(h, 16)))
|
|
55
|
+
.replace(/&#(\d+);/g, (_, d) => String.fromCodePoint(Number(d)))
|
|
56
|
+
.replace(/&/g, "&");
|
|
57
|
+
/** Texto de todas las etiquetas `tag` dentro de un fragmento XML, en orden. */
|
|
58
|
+
const texts = (xml, tag) => [...xml.matchAll(new RegExp(`<${tag}(?:\\s[^>]*)?>([\\s\\S]*?)</${tag}>`, "g"))].map((m) => decodeEntities(m[1]));
|
|
59
|
+
/** Atributo de una etiqueta, sin depender del orden en que el programa los escribió. */
|
|
60
|
+
const attr = (tag, name) => {
|
|
61
|
+
const m = tag.match(new RegExp(`\\s${name.replace(":", "\\:")}="([^"]*)"`));
|
|
62
|
+
return m ? decodeEntities(m[1]) : undefined;
|
|
63
|
+
};
|
|
64
|
+
/** Nivel de título por tamaño de letra frente al cuerpo, como con un PDF. */
|
|
65
|
+
function headingBySize(size, body, bold, text) {
|
|
66
|
+
if (!size || !body)
|
|
67
|
+
return 0;
|
|
68
|
+
const r = size / body;
|
|
69
|
+
const short = text.length < 90 && !/[.,;:]$/.test(text);
|
|
70
|
+
if (r >= 1.6 && short)
|
|
71
|
+
return 1;
|
|
72
|
+
if (r >= 1.3 && short)
|
|
73
|
+
return 2;
|
|
74
|
+
if (((r >= 1.12 && short) || (bold && text.length < 70 && short)) && !/^[-•]/.test(text))
|
|
75
|
+
return 3;
|
|
76
|
+
return 0;
|
|
77
|
+
}
|
|
78
|
+
const modeOf = (values) => {
|
|
79
|
+
const c = new Map();
|
|
80
|
+
for (const v of values)
|
|
81
|
+
if (v)
|
|
82
|
+
c.set(v, (c.get(v) ?? 0) + 1);
|
|
83
|
+
return [...c.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? 0;
|
|
84
|
+
};
|
|
85
|
+
const cellMd = (s) => s.replace(/\|/g, "\\|").replace(/\s*\n\s*/g, " ").trim();
|
|
86
|
+
export function rowsToMarkdown(rows) {
|
|
87
|
+
const width = Math.max(0, ...rows.map((r) => r.length));
|
|
88
|
+
if (!width)
|
|
89
|
+
return "";
|
|
90
|
+
const pad = (r) => [...r, ...Array(width - r.length).fill("")].map(cellMd);
|
|
91
|
+
const [head, ...body] = rows;
|
|
92
|
+
return [`| ${pad(head).join(" | ")} |`, `| ${Array(width).fill("---").join(" | ")} |`, ...body.map((r) => `| ${pad(r).join(" | ")} |`)].join("\n");
|
|
93
|
+
}
|
|
94
|
+
/* ---------- Word (DOCX) ---------- */
|
|
95
|
+
function docxToMarkdown(files) {
|
|
96
|
+
const xml = strFromU8(files["word/document.xml"] ?? new Uint8Array());
|
|
97
|
+
const styles = strFromU8(files["word/styles.xml"] ?? new Uint8Array());
|
|
98
|
+
// Nivel de título por estilo: «Heading1», «Ttulo1» (Word en español) o outlineLvl.
|
|
99
|
+
const headingOf = new Map();
|
|
100
|
+
for (const m of styles.matchAll(/<w:style\b[^>]*w:styleId="([^"]+)"[^>]*>([\s\S]*?)<\/w:style>/g)) {
|
|
101
|
+
const lvl = m[2].match(/<w:outlineLvl w:val="(\d)"/)?.[1] ?? m[1].match(/(?:heading|t[ií]?tulo|ttulo)\s*(\d)/i)?.[1];
|
|
102
|
+
if (lvl !== undefined)
|
|
103
|
+
headingOf.set(m[1], Math.min(6, Number(lvl) + (m[2].includes("outlineLvl") ? 1 : 0)));
|
|
104
|
+
else if (/^(title|t[ií]tulo)$/i.test(m[1]))
|
|
105
|
+
headingOf.set(m[1], 1);
|
|
106
|
+
}
|
|
107
|
+
const out = [];
|
|
108
|
+
const tables = [];
|
|
109
|
+
const body = xml.match(/<w:body>([\s\S]*)<\/w:body>/)?.[1] ?? xml;
|
|
110
|
+
const blocks = [...body.matchAll(/<w:tbl>[\s\S]*?<\/w:tbl>|<w:p[ >][\s\S]*?<\/w:p>|<w:p\/>/g)].map((m) => m[0]);
|
|
111
|
+
const sizeOf = (b) => Math.max(0, ...[...b.matchAll(/<w:sz w:val="(\d+)"/g)].map((m) => Number(m[1])));
|
|
112
|
+
// El cuerpo es el tamaño más frecuente, ponderado por cuánto texto lo usa.
|
|
113
|
+
const bodySize = modeOf(blocks.filter((b) => !b.startsWith("<w:tbl")).flatMap((b) => Array(Math.min(20, texts(b, "w:t").join("").length)).fill(sizeOf(b))));
|
|
114
|
+
for (const b of blocks) {
|
|
115
|
+
if (b.startsWith("<w:tbl")) {
|
|
116
|
+
const rows = [...b.matchAll(/<w:tr[ >][\s\S]*?<\/w:tr>/g)].map((r) => [...r[0].matchAll(/<w:tc>[\s\S]*?<\/w:tc>/g)].map((c) => texts(c[0], "w:t").join("")));
|
|
117
|
+
if (rows.length) {
|
|
118
|
+
tables.push({ title: `Tabla ${tables.length + 1}`, rows });
|
|
119
|
+
out.push(rowsToMarkdown(rows), "");
|
|
120
|
+
}
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
const text = b.replace(/<w:tab\/>/g, "<w:t>\t</w:t>").replace(/<w:br\/>/g, "<w:t>\n</w:t>");
|
|
124
|
+
let line = texts(text, "w:t").join("").replace(/[ \t]+/g, " ").trim();
|
|
125
|
+
if (!line)
|
|
126
|
+
continue;
|
|
127
|
+
const bullet = /^[•●▪◦‣∙·\-–*]\s*/.test(line) || /<w:numPr>/.test(b);
|
|
128
|
+
line = line.replace(/^[•●▪◦‣∙·–*]\s*/, "");
|
|
129
|
+
const style = b.match(/<w:pStyle w:val="([^"]+)"/)?.[1] ?? "";
|
|
130
|
+
const runs = [...b.matchAll(/<w:r>[\s\S]*?<\/w:r>/g)].map((r) => r[0]).filter((r) => texts(r, "w:t").join("").trim());
|
|
131
|
+
const bold = runs.length > 0 && runs.every((r) => /<w:b\/>|<w:b w:val="(?:1|true)"\/>/.test(r));
|
|
132
|
+
const level = headingOf.get(style) ?? (bullet ? 0 : headingBySize(sizeOf(b), bodySize, bold, line));
|
|
133
|
+
if (level)
|
|
134
|
+
out.push(`${"#".repeat(level)} ${line}`, "");
|
|
135
|
+
else if (bullet)
|
|
136
|
+
out.push(`- ${line}`);
|
|
137
|
+
else
|
|
138
|
+
out.push(line, "");
|
|
139
|
+
}
|
|
140
|
+
const markdown = out.join("\n").replace(/\n{3,}/g, "\n\n").trim();
|
|
141
|
+
return { format: "docx", markdown, parts: [{ title: "Documento", markdown }], tables, warnings: [] };
|
|
142
|
+
}
|
|
143
|
+
/* ---------- Excel (XLSX) ---------- */
|
|
144
|
+
function colIndex(ref) {
|
|
145
|
+
const letters = ref.match(/^[A-Z]+/)?.[0] ?? "A";
|
|
146
|
+
return [...letters].reduce((n, ch) => n * 26 + ch.charCodeAt(0) - 64, 0) - 1;
|
|
147
|
+
}
|
|
148
|
+
function xlsxToMarkdown(files) {
|
|
149
|
+
const read = (p) => (files[p] ? strFromU8(files[p]) : "");
|
|
150
|
+
const shared = [...read("xl/sharedStrings.xml").matchAll(/<si>([\s\S]*?)<\/si>/g)].map((m) => texts(m[1], "t").join(""));
|
|
151
|
+
const workbook = read("xl/workbook.xml");
|
|
152
|
+
const rels = read("xl/_rels/workbook.xml.rels");
|
|
153
|
+
const target = new Map([...rels.matchAll(/<Relationship\b[^>]*>/g)].map((m) => [attr(m[0], "Id") ?? "", (attr(m[0], "Target") ?? "").replace(/^\/?xl\//, "")]));
|
|
154
|
+
const sheets = [...workbook.matchAll(/<sheet\b[^>]*>/g)].map((m) => ({ name: attr(m[0], "name") ?? "Hoja", path: `xl/${target.get(attr(m[0], "r:id") ?? "") ?? ""}` }));
|
|
155
|
+
const parts = [];
|
|
156
|
+
const tables = [];
|
|
157
|
+
const warnings = [];
|
|
158
|
+
for (const sheet of sheets) {
|
|
159
|
+
const xml = read(sheet.path);
|
|
160
|
+
const rows = [];
|
|
161
|
+
for (const r of xml.matchAll(/<row\b[^>]*>([\s\S]*?)<\/row>|<row\b[^>]*\/>/g)) {
|
|
162
|
+
if (!r[1])
|
|
163
|
+
continue;
|
|
164
|
+
const row = [];
|
|
165
|
+
for (const c of r[1].matchAll(/<c\b([^>]*?)(?:\/>|>([\s\S]*?)<\/c>)/g)) {
|
|
166
|
+
const attrs = c[1];
|
|
167
|
+
const ref = attrs.match(/r="([A-Z]+)\d+"/)?.[1] ?? "";
|
|
168
|
+
const type = attrs.match(/t="([^"]+)"/)?.[1];
|
|
169
|
+
const inner = c[2] ?? "";
|
|
170
|
+
const v = inner.match(/<v>([\s\S]*?)<\/v>/)?.[1];
|
|
171
|
+
const value = type === "s" ? shared[Number(v)] ?? "" : type === "inlineStr" ? texts(inner, "t").join("") : decodeEntities(v ?? "");
|
|
172
|
+
row[ref ? colIndex(ref) : row.length] = value;
|
|
173
|
+
}
|
|
174
|
+
if (row.some((x) => x?.trim()))
|
|
175
|
+
rows.push(Array.from(row, (x) => x ?? ""));
|
|
176
|
+
}
|
|
177
|
+
if (!rows.length)
|
|
178
|
+
continue;
|
|
179
|
+
if (xml.includes("<f>"))
|
|
180
|
+
warnings.push(`«${sheet.name}» tiene fórmulas: se usa el último valor calculado que guardó Excel.`);
|
|
181
|
+
tables.push({ title: sheet.name, rows });
|
|
182
|
+
parts.push({ title: sheet.name, markdown: `## ${sheet.name}\n\n${rowsToMarkdown(rows)}` });
|
|
183
|
+
}
|
|
184
|
+
return { format: "xlsx", markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables, warnings };
|
|
185
|
+
}
|
|
186
|
+
/* ---------- PowerPoint (PPTX) ---------- */
|
|
187
|
+
function pptxToMarkdown(files) {
|
|
188
|
+
const slides = Object.keys(files)
|
|
189
|
+
.filter((n) => /^ppt\/slides\/slide\d+\.xml$/.test(n))
|
|
190
|
+
.sort((a, b) => Number(a.match(/\d+/g).pop()) - Number(b.match(/\d+/g).pop()));
|
|
191
|
+
const parts = slides.map((name, i) => {
|
|
192
|
+
const xml = strFromU8(files[name]);
|
|
193
|
+
const paragraphs = [...xml.matchAll(/<a:p>([\s\S]*?)<\/a:p>/g)].map((m) => texts(m[1], "a:t").join("").trim()).filter(Boolean);
|
|
194
|
+
const notesName = `ppt/notesSlides/notesSlide${name.match(/\d+/g).pop()}.xml`;
|
|
195
|
+
const notes = files[notesName]
|
|
196
|
+
? [...strFromU8(files[notesName]).matchAll(/<a:p>([\s\S]*?)<\/a:p>/g)].map((m) => texts(m[1], "a:t").join("").trim()).filter((t) => t && !/^\d+$/.test(t))
|
|
197
|
+
: [];
|
|
198
|
+
const [title, ...rest] = paragraphs;
|
|
199
|
+
const md = [`## Diapositiva ${i + 1}${title ? `: ${title}` : ""}`, "", ...rest.map((p) => `- ${p}`), ...(notes.length ? ["", `> Notas: ${notes.join(" ")}`] : [])].join("\n");
|
|
200
|
+
return { title: title ?? `Diapositiva ${i + 1}`, markdown: md };
|
|
201
|
+
});
|
|
202
|
+
return { format: "pptx", markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables: [], warnings: [] };
|
|
203
|
+
}
|
|
204
|
+
/* ---------- OpenDocument ---------- */
|
|
205
|
+
function odfToMarkdown(files, format) {
|
|
206
|
+
const xml = strFromU8(files["content.xml"] ?? new Uint8Array());
|
|
207
|
+
const plain = (s) => decodeEntities(s.replace(/<text:(?:s|tab)\b[^>]*\/>/g, " ").replace(/<text:line-break\/>/g, "\n").replace(/<[^>]+>/g, "")).trim();
|
|
208
|
+
if (format === "ods") {
|
|
209
|
+
const tables = [];
|
|
210
|
+
for (const t of xml.matchAll(/<table:table\b[^>]*table:name="([^"]*)"[^>]*>([\s\S]*?)<\/table:table>/g)) {
|
|
211
|
+
const rows = [...t[2].matchAll(/<table:table-row\b[^>]*>([\s\S]*?)<\/table:table-row>/g)]
|
|
212
|
+
.map((r) => [...r[1].matchAll(/<table:table-cell\b([^>]*?)(?:\/>|>([\s\S]*?)<\/table:table-cell>)/g)].flatMap((c) => {
|
|
213
|
+
const repeat = Math.min(50, Number(c[1].match(/number-columns-repeated="(\d+)"/)?.[1] ?? 1));
|
|
214
|
+
return Array(repeat).fill(plain(c[2] ?? ""));
|
|
215
|
+
}))
|
|
216
|
+
.map((r) => { while (r.length && !r[r.length - 1])
|
|
217
|
+
r.pop(); return r; })
|
|
218
|
+
.filter((r) => r.length);
|
|
219
|
+
if (rows.length)
|
|
220
|
+
tables.push({ title: decodeEntities(t[1]), rows });
|
|
221
|
+
}
|
|
222
|
+
const parts = tables.map((t) => ({ title: t.title, markdown: `## ${t.title}\n\n${rowsToMarkdown(t.rows)}` }));
|
|
223
|
+
return { format, markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables, warnings: [] };
|
|
224
|
+
}
|
|
225
|
+
// Estilos automáticos: tamaño y negrita de cada estilo de párrafo y de texto.
|
|
226
|
+
const styleInfo = new Map();
|
|
227
|
+
for (const m of xml.matchAll(/<style:style\b([^>]*)>([\s\S]*?)<\/style:style>/g)) {
|
|
228
|
+
const size = Number(m[2].match(/fo:font-size="([\d.]+)pt"/)?.[1] ?? 0);
|
|
229
|
+
styleInfo.set(attr(m[0], "style:name") ?? "", { size, bold: /fo:font-weight="bold"/.test(m[2]) });
|
|
230
|
+
}
|
|
231
|
+
const body = xml.match(/<office:(?:text|presentation)>([\s\S]*)<\/office:(?:text|presentation)>/)?.[1] ?? xml;
|
|
232
|
+
const blocks = [...body.matchAll(/<table:table\b[\s\S]*?<\/table:table>|<text:list\b[\s\S]*?<\/text:list>|<text:(h|p)\b[^>]*>[\s\S]*?<\/text:\1>|<draw:page\b[^>]*>/g)].map((m) => m[0]);
|
|
233
|
+
const paraSize = (b) => styleInfo.get(attr(b.match(/^<[^>]+>/)[0], "text:style-name") ?? "")?.size ?? 0;
|
|
234
|
+
const bodySize = modeOf(blocks.filter((b) => b.startsWith("<text:p")).map(paraSize));
|
|
235
|
+
const out = [];
|
|
236
|
+
const tables = [];
|
|
237
|
+
let slide = 0;
|
|
238
|
+
for (const b of blocks) {
|
|
239
|
+
if (b.startsWith("<draw:page")) {
|
|
240
|
+
out.push(`## Diapositiva ${++slide}`, "");
|
|
241
|
+
}
|
|
242
|
+
else if (b.startsWith("<table:table")) {
|
|
243
|
+
const rows = [...b.matchAll(/<table:table-row\b[^>]*>([\s\S]*?)<\/table:table-row>/g)].map((r) => [...r[1].matchAll(/<table:table-cell\b[^>]*?(?:\/>|>([\s\S]*?)<\/table:table-cell>)/g)].map((c) => plain(c[1] ?? "")));
|
|
244
|
+
if (rows.length) {
|
|
245
|
+
tables.push({ title: attr(b.match(/^<[^>]+>/)[0], "table:name") ?? `Tabla ${tables.length + 1}`, rows });
|
|
246
|
+
out.push(rowsToMarkdown(rows), "");
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
else if (b.startsWith("<text:list")) {
|
|
250
|
+
for (const it of b.matchAll(/<text:list-item\b[^>]*>([\s\S]*?)<\/text:list-item>/g)) {
|
|
251
|
+
const t = plain(it[1]);
|
|
252
|
+
if (t)
|
|
253
|
+
out.push(`- ${t}`);
|
|
254
|
+
}
|
|
255
|
+
out.push("");
|
|
256
|
+
}
|
|
257
|
+
else {
|
|
258
|
+
const text = plain(b);
|
|
259
|
+
if (!text)
|
|
260
|
+
continue;
|
|
261
|
+
const open = b.match(/^<[^>]+>/)[0];
|
|
262
|
+
const spans = [...b.matchAll(/<text:span\b[^>]*>/g)].map((m) => styleInfo.get(attr(m[0], "text:style-name") ?? ""));
|
|
263
|
+
const bold = styleInfo.get(attr(open, "text:style-name") ?? "")?.bold || (spans.length > 0 && spans.every((x) => x?.bold));
|
|
264
|
+
const level = b.startsWith("<text:h") ? Number(attr(open, "text:outline-level") ?? 2) : headingBySize(paraSize(b), bodySize, !!bold, text);
|
|
265
|
+
out.push(level ? `${"#".repeat(Math.min(6, level))} ${text}` : text, "");
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
const markdown = out.join("\n").replace(/\n{3,}/g, "\n\n").trim();
|
|
269
|
+
return { format, markdown, parts: [{ title: "Documento", markdown }], tables, warnings: [] };
|
|
270
|
+
}
|
|
271
|
+
/* ---------- HTML / EPUB ---------- */
|
|
272
|
+
export function htmlToMarkdown(html) {
|
|
273
|
+
const tables = [];
|
|
274
|
+
let s = html
|
|
275
|
+
.replace(/<(script|style|noscript|svg|head|nav|footer)\b[\s\S]*?<\/\1>/gi, "")
|
|
276
|
+
.replace(/<!--[\s\S]*?-->/g, "");
|
|
277
|
+
s = s.replace(/<table\b[\s\S]*?<\/table>/gi, (t) => {
|
|
278
|
+
const rows = [...t.matchAll(/<tr\b[\s\S]*?<\/tr>/gi)].map((r) => [...r[0].matchAll(/<t[hd]\b[^>]*>([\s\S]*?)<\/t[hd]>/gi)].map((c) => decodeEntities(c[1].replace(/<[^>]+>/g, " ")).replace(/\s+/g, " ").trim())).filter((r) => r.length);
|
|
279
|
+
if (!rows.length)
|
|
280
|
+
return "";
|
|
281
|
+
tables.push(rows);
|
|
282
|
+
return `\n\n${rowsToMarkdown(rows)}\n\n`;
|
|
283
|
+
});
|
|
284
|
+
s = s
|
|
285
|
+
.replace(/<h([1-6])\b[^>]*>([\s\S]*?)<\/h\1>/gi, (_, n, t) => `\n\n${"#".repeat(Number(n))} ${t.replace(/<[^>]+>/g, "").trim()}\n\n`)
|
|
286
|
+
.replace(/<li\b[^>]*>/gi, "\n- ")
|
|
287
|
+
.replace(/<br\s*\/?>/gi, "\n")
|
|
288
|
+
.replace(/<\/(p|div|section|article|ul|ol|blockquote|tr)>/gi, "\n\n")
|
|
289
|
+
.replace(/<(strong|b)\b[^>]*>([\s\S]*?)<\/\1>/gi, "**$2**")
|
|
290
|
+
.replace(/<a\b[^>]*href="(https?:[^"]+)"[^>]*>([\s\S]*?)<\/a>/gi, "[$2]($1)")
|
|
291
|
+
.replace(/<[^>]+>/g, "");
|
|
292
|
+
const markdown = decodeEntities(s)
|
|
293
|
+
.split("\n")
|
|
294
|
+
.map((l) => l.replace(/[ \t]+/g, " ").trim())
|
|
295
|
+
.join("\n")
|
|
296
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
297
|
+
.trim();
|
|
298
|
+
return { markdown, tables };
|
|
299
|
+
}
|
|
300
|
+
function epubToMarkdown(files) {
|
|
301
|
+
const container = strFromU8(files["META-INF/container.xml"] ?? new Uint8Array());
|
|
302
|
+
const opfPath = container.match(/full-path="([^"]+)"/)?.[1] ?? Object.keys(files).find((n) => n.endsWith(".opf")) ?? "";
|
|
303
|
+
const opf = strFromU8(files[opfPath] ?? new Uint8Array());
|
|
304
|
+
const base = opfPath.includes("/") ? opfPath.slice(0, opfPath.lastIndexOf("/") + 1) : "";
|
|
305
|
+
const manifest = new Map([...opf.matchAll(/<item\b[^>]*id="([^"]+)"[^>]*href="([^"]+)"/g)].map((m) => [m[1], m[2]]));
|
|
306
|
+
const spine = [...opf.matchAll(/<itemref\b[^>]*idref="([^"]+)"/g)].map((m) => base + decodeURIComponent(manifest.get(m[1]) ?? ""));
|
|
307
|
+
const parts = spine
|
|
308
|
+
.filter((p) => files[p])
|
|
309
|
+
.map((p, i) => {
|
|
310
|
+
const { markdown } = htmlToMarkdown(strFromU8(files[p]));
|
|
311
|
+
return { title: markdown.match(/^#+\s+(.+)$/m)?.[1] ?? `Capítulo ${i + 1}`, markdown };
|
|
312
|
+
})
|
|
313
|
+
.filter((p) => p.markdown);
|
|
314
|
+
return { format: "epub", markdown: parts.map((p) => p.markdown).join("\n\n"), parts, tables: [], warnings: [] };
|
|
315
|
+
}
|
|
316
|
+
/* ---------- texto ---------- */
|
|
317
|
+
/** CSV con comillas, separador autodetectado (coma, punto y coma o tabulador). */
|
|
318
|
+
export function parseDelimited(text, delimiter) {
|
|
319
|
+
const first = text.split("\n", 1)[0];
|
|
320
|
+
const sep = delimiter ?? [";", "\t", ","].reduce((best, d) => (first.split(d).length > first.split(best).length ? d : best), ",");
|
|
321
|
+
const rows = [];
|
|
322
|
+
let row = [];
|
|
323
|
+
let cell = "";
|
|
324
|
+
let quoted = false;
|
|
325
|
+
for (let i = 0; i < text.length; i++) {
|
|
326
|
+
const ch = text[i];
|
|
327
|
+
if (quoted) {
|
|
328
|
+
if (ch === '"' && text[i + 1] === '"') {
|
|
329
|
+
cell += '"';
|
|
330
|
+
i++;
|
|
331
|
+
}
|
|
332
|
+
else if (ch === '"')
|
|
333
|
+
quoted = false;
|
|
334
|
+
else
|
|
335
|
+
cell += ch;
|
|
336
|
+
}
|
|
337
|
+
else if (ch === '"')
|
|
338
|
+
quoted = true;
|
|
339
|
+
else if (ch === sep) {
|
|
340
|
+
row.push(cell);
|
|
341
|
+
cell = "";
|
|
342
|
+
}
|
|
343
|
+
else if (ch === "\n" || ch === "\r") {
|
|
344
|
+
if (ch === "\r" && text[i + 1] === "\n")
|
|
345
|
+
i++;
|
|
346
|
+
row.push(cell);
|
|
347
|
+
cell = "";
|
|
348
|
+
if (row.some((c) => c.trim()))
|
|
349
|
+
rows.push(row);
|
|
350
|
+
row = [];
|
|
351
|
+
}
|
|
352
|
+
else
|
|
353
|
+
cell += ch;
|
|
354
|
+
}
|
|
355
|
+
row.push(cell);
|
|
356
|
+
if (row.some((c) => c.trim()))
|
|
357
|
+
rows.push(row);
|
|
358
|
+
return rows;
|
|
359
|
+
}
|
|
360
|
+
function emlToMarkdown(raw) {
|
|
361
|
+
const [headerBlock, ...bodyParts] = raw.split(/\r?\n\r?\n/);
|
|
362
|
+
const header = (name) => headerBlock.match(new RegExp(`^${name}:\\s*(.+(?:\\r?\\n[ \\t].+)*)`, "im"))?.[1].replace(/\r?\n[ \t]+/g, " ").trim();
|
|
363
|
+
let body = bodyParts.join("\n\n");
|
|
364
|
+
const htmlPart = body.match(/Content-Type:\s*text\/html[\s\S]*?\r?\n\r?\n([\s\S]*?)(?:\r?\n--|$)/i)?.[1];
|
|
365
|
+
const textPart = body.match(/Content-Type:\s*text\/plain[\s\S]*?\r?\n\r?\n([\s\S]*?)(?:\r?\n--|$)/i)?.[1];
|
|
366
|
+
body = textPart ?? (htmlPart ? htmlToMarkdown(htmlPart).markdown : body);
|
|
367
|
+
// quoted-printable, lo más común en correos en español.
|
|
368
|
+
body = body.replace(/=\r?\n/g, "").replace(/(?:=[0-9A-F]{2})+/g, (m) => {
|
|
369
|
+
try {
|
|
370
|
+
return new TextDecoder().decode(new Uint8Array(m.match(/[0-9A-F]{2}/g).map((h) => parseInt(h, 16))));
|
|
371
|
+
}
|
|
372
|
+
catch {
|
|
373
|
+
return m;
|
|
374
|
+
}
|
|
375
|
+
});
|
|
376
|
+
const meta = [["De", header("From")], ["Para", header("To")], ["Fecha", header("Date")]].filter(([, v]) => v).map(([k, v]) => `${k}: ${v}`);
|
|
377
|
+
const markdown = [`# ${header("Subject") ?? "(sin asunto)"}`, "", ...meta, "", body.trim()].join("\n");
|
|
378
|
+
const warnings = /Content-Disposition:\s*attachment/i.test(raw) ? ["El correo trae adjuntos: se leyó solo el cuerpo."] : [];
|
|
379
|
+
return { format: "eml", markdown, parts: [{ title: header("Subject") ?? "Correo", markdown }], tables: [], warnings };
|
|
380
|
+
}
|
|
381
|
+
/** RTF por grupos: se saltan tablas de fuentes, colores, estilos e info, y se decodifica cp1252. */
|
|
382
|
+
function rtfToText(rtf) {
|
|
383
|
+
const cp1252 = new TextDecoder("windows-1252");
|
|
384
|
+
const SKIP = new Set(["fonttbl", "colortbl", "stylesheet", "info", "pict", "header", "footer", "listtable", "listoverridetable", "generator", "expandedcolortbl", "themedata", "datastore", "xmlnstbl", "latentstyles"]);
|
|
385
|
+
let out = "";
|
|
386
|
+
const stack = [];
|
|
387
|
+
let skipping = false;
|
|
388
|
+
let i = 0;
|
|
389
|
+
while (i < rtf.length) {
|
|
390
|
+
const ch = rtf[i];
|
|
391
|
+
if (ch === "{") {
|
|
392
|
+
stack.push(skipping);
|
|
393
|
+
i++;
|
|
394
|
+
if (rtf.startsWith("\\*", i))
|
|
395
|
+
skipping = true;
|
|
396
|
+
continue;
|
|
397
|
+
}
|
|
398
|
+
if (ch === "}") {
|
|
399
|
+
skipping = stack.pop() ?? false;
|
|
400
|
+
i++;
|
|
401
|
+
continue;
|
|
402
|
+
}
|
|
403
|
+
if (ch === "\\") {
|
|
404
|
+
const next = rtf[i + 1];
|
|
405
|
+
if (next === "\\" || next === "{" || next === "}") {
|
|
406
|
+
if (!skipping)
|
|
407
|
+
out += next;
|
|
408
|
+
i += 2;
|
|
409
|
+
continue;
|
|
410
|
+
}
|
|
411
|
+
// «\» al final de línea es un salto de párrafo en el RTF de macOS y de Word.
|
|
412
|
+
if (next === "\n" || next === "\r") {
|
|
413
|
+
if (!skipping)
|
|
414
|
+
out += "\n";
|
|
415
|
+
i += 2;
|
|
416
|
+
continue;
|
|
417
|
+
}
|
|
418
|
+
if (next === "'") {
|
|
419
|
+
if (!skipping)
|
|
420
|
+
out += cp1252.decode(new Uint8Array([parseInt(rtf.slice(i + 2, i + 4), 16)]));
|
|
421
|
+
i += 4;
|
|
422
|
+
continue;
|
|
423
|
+
}
|
|
424
|
+
const m = rtf.slice(i).match(/^\\([a-z]+)(-?\d+)? ?/i);
|
|
425
|
+
if (!m) {
|
|
426
|
+
i += 2;
|
|
427
|
+
continue;
|
|
428
|
+
}
|
|
429
|
+
const [whole, word, num] = m;
|
|
430
|
+
i += whole.length;
|
|
431
|
+
if (SKIP.has(word))
|
|
432
|
+
skipping = true;
|
|
433
|
+
if (skipping)
|
|
434
|
+
continue;
|
|
435
|
+
if (word === "par" || word === "line" || word === "sect" || word === "row")
|
|
436
|
+
out += "\n";
|
|
437
|
+
else if (word === "tab" || word === "cell")
|
|
438
|
+
out += "\t";
|
|
439
|
+
else if (word === "u" && num) {
|
|
440
|
+
out += String.fromCharCode(Number(num) < 0 ? Number(num) + 65536 : Number(num));
|
|
441
|
+
if (rtf[i] === "?")
|
|
442
|
+
i++;
|
|
443
|
+
}
|
|
444
|
+
continue;
|
|
445
|
+
}
|
|
446
|
+
if (ch === "\r" || ch === "\n") {
|
|
447
|
+
i++;
|
|
448
|
+
continue;
|
|
449
|
+
}
|
|
450
|
+
if (!skipping)
|
|
451
|
+
out += ch;
|
|
452
|
+
i++;
|
|
453
|
+
}
|
|
454
|
+
return out
|
|
455
|
+
.split("\n")
|
|
456
|
+
.map((l) => l.replace(/[ \t]+/g, " ").trim())
|
|
457
|
+
.join("\n")
|
|
458
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
459
|
+
.trim();
|
|
460
|
+
}
|
|
461
|
+
/**
|
|
462
|
+
* Convierte cualquier formato soportado a Markdown. PDF e imágenes no pasan
|
|
463
|
+
* por aquí: necesitan PDF.js y OCR, y quien llama ya los tiene.
|
|
464
|
+
*/
|
|
465
|
+
export function convertToMarkdown(bytes, filename = "") {
|
|
466
|
+
const format = detectFormat(bytes, filename);
|
|
467
|
+
const empty = (warning) => ({ format, markdown: "", parts: [], tables: [], warnings: [warning] });
|
|
468
|
+
const text = () => new TextDecoder("utf-8").decode(bytes).replace(/^/, "");
|
|
469
|
+
switch (format) {
|
|
470
|
+
case "docx":
|
|
471
|
+
case "xlsx":
|
|
472
|
+
case "pptx":
|
|
473
|
+
case "odt":
|
|
474
|
+
case "ods":
|
|
475
|
+
case "odp":
|
|
476
|
+
case "epub": {
|
|
477
|
+
const files = unzipSync(bytes);
|
|
478
|
+
if (format === "docx")
|
|
479
|
+
return docxToMarkdown(files);
|
|
480
|
+
if (format === "xlsx")
|
|
481
|
+
return xlsxToMarkdown(files);
|
|
482
|
+
if (format === "pptx")
|
|
483
|
+
return pptxToMarkdown(files);
|
|
484
|
+
if (format === "epub")
|
|
485
|
+
return epubToMarkdown(files);
|
|
486
|
+
return odfToMarkdown(files, format);
|
|
487
|
+
}
|
|
488
|
+
case "html": {
|
|
489
|
+
const { markdown, tables } = htmlToMarkdown(text());
|
|
490
|
+
return { format, markdown, parts: [{ title: "Página", markdown }], tables: tables.map((rows, i) => ({ title: `Tabla ${i + 1}`, rows })), warnings: [] };
|
|
491
|
+
}
|
|
492
|
+
case "csv":
|
|
493
|
+
case "tsv": {
|
|
494
|
+
const rows = parseDelimited(text(), format === "tsv" ? "\t" : undefined);
|
|
495
|
+
const markdown = rowsToMarkdown(rows);
|
|
496
|
+
return { format, markdown, parts: [{ title: filename || "Tabla", markdown }], tables: [{ title: filename || "Tabla", rows }], warnings: [] };
|
|
497
|
+
}
|
|
498
|
+
case "json": {
|
|
499
|
+
const raw = text();
|
|
500
|
+
try {
|
|
501
|
+
return { format, markdown: "```json\n" + JSON.stringify(JSON.parse(raw), null, 2) + "\n```", parts: [], tables: [], warnings: [] };
|
|
502
|
+
}
|
|
503
|
+
catch {
|
|
504
|
+
return { format: "txt", markdown: raw, parts: [], tables: [], warnings: ["JSON inválido: se leyó como texto."] };
|
|
505
|
+
}
|
|
506
|
+
}
|
|
507
|
+
case "eml":
|
|
508
|
+
return emlToMarkdown(text());
|
|
509
|
+
case "rtf":
|
|
510
|
+
return { format, markdown: rtfToText(text()), parts: [], tables: [], warnings: [] };
|
|
511
|
+
case "md":
|
|
512
|
+
case "txt":
|
|
513
|
+
return { format, markdown: text(), parts: [], tables: [], warnings: [] };
|
|
514
|
+
case "pdf":
|
|
515
|
+
return empty("PDF: usa pdfToMarkdown (necesita PDF.js).");
|
|
516
|
+
case "image":
|
|
517
|
+
return empty("Imagen: necesita OCR.");
|
|
518
|
+
default:
|
|
519
|
+
return empty("Formato no reconocido.");
|
|
520
|
+
}
|
|
521
|
+
}
|
package/dist/forms.d.ts
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
export interface FormField {
|
|
2
|
+
name: string;
|
|
3
|
+
type: "text" | "checkbox" | "radio" | "dropdown" | "list" | "signature" | "other";
|
|
4
|
+
/** Texto, opción elegida o lista de opciones elegidas; `null` si está vacío. */
|
|
5
|
+
value: string | string[] | null;
|
|
6
|
+
checked?: boolean;
|
|
7
|
+
options?: string[];
|
|
8
|
+
/** Página base 1 del primer widget, si se puede resolver. */
|
|
9
|
+
page: number | null;
|
|
10
|
+
}
|
|
11
|
+
export declare function readFormFields(buf: ArrayBuffer | Uint8Array): Promise<FormField[]>;
|
|
12
|
+
export interface TextCheckbox {
|
|
13
|
+
label: string;
|
|
14
|
+
checked: boolean;
|
|
15
|
+
/** La línea donde apareció, para citarla. */
|
|
16
|
+
line: string;
|
|
17
|
+
}
|
|
18
|
+
/** Casillas escritas como caracteres en el texto de cualquier documento. */
|
|
19
|
+
export declare function findTextCheckboxes(text: string): TextCheckbox[];
|
package/dist/forms.js
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
// Formularios: lo que LlamaParse vende como «checkboxes» y «forms», resuelto
|
|
2
|
+
// sin modelo de visión en los dos casos en que se puede hacer bien:
|
|
3
|
+
// 1. Formularios PDF rellenables (AcroForm): el valor está en el archivo.
|
|
4
|
+
// 2. Casillas dibujadas como caracteres (☐ ☑ ☒ [x] ( )): están en el texto.
|
|
5
|
+
// Una casilla marcada a mano sobre un escaneo NO entra aquí: eso sí necesita visión.
|
|
6
|
+
import { PDFCheckBox, PDFDocument, PDFDropdown, PDFOptionList, PDFRadioGroup, PDFTextField } from "pdf-lib";
|
|
7
|
+
export async function readFormFields(buf) {
|
|
8
|
+
const doc = await PDFDocument.load(buf, { updateMetadata: false });
|
|
9
|
+
const pages = doc.getPages();
|
|
10
|
+
const pageOf = (field) => {
|
|
11
|
+
const ref = field.acroField.getWidgets()[0]?.P();
|
|
12
|
+
const i = ref ? pages.findIndex((p) => p.ref === ref) : -1;
|
|
13
|
+
return i >= 0 ? i + 1 : null;
|
|
14
|
+
};
|
|
15
|
+
const out = [];
|
|
16
|
+
for (const f of doc.getForm().getFields()) {
|
|
17
|
+
const base = { name: f.getName(), page: pageOf(f) };
|
|
18
|
+
if (f instanceof PDFTextField)
|
|
19
|
+
out.push({ ...base, type: "text", value: f.getText() || null });
|
|
20
|
+
else if (f instanceof PDFCheckBox)
|
|
21
|
+
out.push({ ...base, type: "checkbox", value: f.isChecked() ? "on" : null, checked: f.isChecked() });
|
|
22
|
+
else if (f instanceof PDFRadioGroup)
|
|
23
|
+
out.push({ ...base, type: "radio", value: f.getSelected() ?? null, options: f.getOptions() });
|
|
24
|
+
else if (f instanceof PDFDropdown)
|
|
25
|
+
out.push({ ...base, type: "dropdown", value: f.getSelected()[0] ?? null, options: f.getOptions() });
|
|
26
|
+
else if (f instanceof PDFOptionList)
|
|
27
|
+
out.push({ ...base, type: "list", value: f.getSelected(), options: f.getOptions() });
|
|
28
|
+
else if (f.constructor.name.includes("Signature"))
|
|
29
|
+
out.push({ ...base, type: "signature", value: null });
|
|
30
|
+
else
|
|
31
|
+
out.push({ ...base, type: "other", value: null });
|
|
32
|
+
}
|
|
33
|
+
return out;
|
|
34
|
+
}
|
|
35
|
+
const MARK_ON = "☑☒✅✔✓■▣⊠";
|
|
36
|
+
const MARK_OFF = "☐□▢";
|
|
37
|
+
/** Una casilla y su etiqueta: «☒ Sí», «[x] Hipertensión», «( ) No». */
|
|
38
|
+
const BOX = new RegExp(String.raw `(?:([${MARK_ON}${MARK_OFF}])|\[\s*([xX✓✔*]?)\s*\]|\(\s*([xX✓✔*•]?)\s*\))\s*([^${MARK_ON}${MARK_OFF}\[\(\n]{1,60}?)(?=\s{2,}|\s*[${MARK_ON}${MARK_OFF}\[\(]|$)`, "gmu");
|
|
39
|
+
/** Casillas escritas como caracteres en el texto de cualquier documento. */
|
|
40
|
+
export function findTextCheckboxes(text) {
|
|
41
|
+
const out = [];
|
|
42
|
+
for (const line of text.split("\n")) {
|
|
43
|
+
// Sin al menos una marca de casilla, «(ver anexo)» o «[1]» no son casillas.
|
|
44
|
+
if (!new RegExp(`[${MARK_ON}${MARK_OFF}]|\\[\\s*[xX✓✔*]?\\s*\\]\\s*\\p{L}|\\(\\s*[xX✓✔*•]?\\s*\\)\\s*\\p{L}`, "u").test(line))
|
|
45
|
+
continue;
|
|
46
|
+
for (const m of line.matchAll(BOX)) {
|
|
47
|
+
const label = (m[4] ?? "").trim().replace(/[:;,.]$/, "");
|
|
48
|
+
if (!label || !/\p{L}/u.test(label))
|
|
49
|
+
continue;
|
|
50
|
+
const glyph = m[1];
|
|
51
|
+
const inner = m[2] ?? m[3];
|
|
52
|
+
const checked = glyph ? MARK_ON.includes(glyph) : !!inner;
|
|
53
|
+
out.push({ label, checked, line: line.trim().slice(0, 160) });
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
return out;
|
|
57
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -17,3 +17,12 @@ export * from "./datasplit.js";
|
|
|
17
17
|
export * from "./dataset.js";
|
|
18
18
|
export * from "./verifiable.js";
|
|
19
19
|
export * from "./paragraphs.js";
|
|
20
|
+
export * from "./agentchunks.js";
|
|
21
|
+
export * from "./classify.js";
|
|
22
|
+
export * from "./pages.js";
|
|
23
|
+
export * from "./security.js";
|
|
24
|
+
export * from "./extract.js";
|
|
25
|
+
export * from "./locate.js";
|
|
26
|
+
export * from "./forms.js";
|
|
27
|
+
export * from "./segment.js";
|
|
28
|
+
export * from "./formats.js";
|
package/dist/index.js
CHANGED
|
@@ -19,3 +19,12 @@ export * from "./datasplit.js";
|
|
|
19
19
|
export * from "./dataset.js";
|
|
20
20
|
export * from "./verifiable.js";
|
|
21
21
|
export * from "./paragraphs.js";
|
|
22
|
+
export * from "./agentchunks.js";
|
|
23
|
+
export * from "./classify.js";
|
|
24
|
+
export * from "./pages.js";
|
|
25
|
+
export * from "./security.js";
|
|
26
|
+
export * from "./extract.js";
|
|
27
|
+
export * from "./locate.js";
|
|
28
|
+
export * from "./forms.js";
|
|
29
|
+
export * from "./segment.js";
|
|
30
|
+
export * from "./formats.js";
|
package/dist/locate.d.ts
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import type { Line } from "./markdown.js";
|
|
2
|
+
export interface Location {
|
|
3
|
+
page: number;
|
|
4
|
+
/** Caja 0–1, origen arriba a la izquierda: [x0, y0, x1, y1]. */
|
|
5
|
+
bbox: [number, number, number, number];
|
|
6
|
+
/** Líneas que abarca la cita (una cita puede cruzar un salto de línea). */
|
|
7
|
+
lines: number;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Ubica `text` entre las líneas extraídas. Tolera diferencias de espacios,
|
|
11
|
+
* tildes y puntuación, y citas que ocupan hasta tres líneas seguidas.
|
|
12
|
+
*/
|
|
13
|
+
export declare function locateText(lines: Line[], text: string, page?: number | null): Location | null;
|