zerochat 7.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- zerochat-7.3.0.dist-info/METADATA +70 -0
- zerochat-7.3.0.dist-info/RECORD +130 -0
- zerochat-7.3.0.dist-info/WHEEL +5 -0
- zerochat-7.3.0.dist-info/entry_points.txt +2 -0
- zerochat-7.3.0.dist-info/licenses/LICENSE +34 -0
- zerochat-7.3.0.dist-info/top_level.txt +2 -0
- zerochat.py +2004 -0
- zerochat_runtime/__init__.py +1 -0
- zerochat_runtime/assets/css/base.css +199 -0
- zerochat_runtime/assets/css/components/composer.css +1258 -0
- zerochat_runtime/assets/css/components/debug.css +777 -0
- zerochat_runtime/assets/css/components/header.css +97 -0
- zerochat_runtime/assets/css/components/icons.css +39 -0
- zerochat_runtime/assets/css/components/markdown.css +685 -0
- zerochat_runtime/assets/css/components/messages.css +473 -0
- zerochat_runtime/assets/css/components/modals.css +1319 -0
- zerochat_runtime/assets/css/components/sidebar.css +622 -0
- zerochat_runtime/assets/css/components/tools.css +1223 -0
- zerochat_runtime/assets/css/layout.css +171 -0
- zerochat_runtime/assets/css/print.css +120 -0
- zerochat_runtime/assets/css/styles.css +18 -0
- zerochat_runtime/assets/css/theme-overrides.css +555 -0
- zerochat_runtime/assets/css/tokens.css +165 -0
- zerochat_runtime/assets/help/architecture.html +247 -0
- zerochat_runtime/assets/help/composer.html +134 -0
- zerochat_runtime/assets/help/conversations.html +105 -0
- zerochat_runtime/assets/help/debug.html +108 -0
- zerochat_runtime/assets/help/en/architecture.html +247 -0
- zerochat_runtime/assets/help/en/composer.html +134 -0
- zerochat_runtime/assets/help/en/conversations.html +104 -0
- zerochat_runtime/assets/help/en/debug.html +105 -0
- zerochat_runtime/assets/help/en/gemini-free.html +39 -0
- zerochat_runtime/assets/help/en/index.html +191 -0
- zerochat_runtime/assets/help/en/learning.html +71 -0
- zerochat_runtime/assets/help/en/mcp.html +143 -0
- zerochat_runtime/assets/help/en/openrouter-free.html +39 -0
- zerochat_runtime/assets/help/en/profiles.html +173 -0
- zerochat_runtime/assets/help/en/rag.html +130 -0
- zerochat_runtime/assets/help/en/reasoning-telemetry.html +108 -0
- zerochat_runtime/assets/help/en/tools-agent.html +133 -0
- zerochat_runtime/assets/help/en/webllm.html +355 -0
- zerochat_runtime/assets/help/gemini-free.html +39 -0
- zerochat_runtime/assets/help/help.css +604 -0
- zerochat_runtime/assets/help/help.js +51 -0
- zerochat_runtime/assets/help/index.html +194 -0
- zerochat_runtime/assets/help/learning.html +71 -0
- zerochat_runtime/assets/help/mcp.html +143 -0
- zerochat_runtime/assets/help/openrouter-free.html +39 -0
- zerochat_runtime/assets/help/profiles.html +174 -0
- zerochat_runtime/assets/help/rag.html +130 -0
- zerochat_runtime/assets/help/reasoning-telemetry.html +111 -0
- zerochat_runtime/assets/help/tools-agent.html +133 -0
- zerochat_runtime/assets/help/webllm.html +359 -0
- zerochat_runtime/assets/js/agent-core.js +1553 -0
- zerochat_runtime/assets/js/api.js +844 -0
- zerochat_runtime/assets/js/app.js +2558 -0
- zerochat_runtime/assets/js/attachments.js +220 -0
- zerochat_runtime/assets/js/charts.js +338 -0
- zerochat_runtime/assets/js/chat-engine.js +566 -0
- zerochat_runtime/assets/js/config-store.js +207 -0
- zerochat_runtime/assets/js/context-manager.js +584 -0
- zerochat_runtime/assets/js/conversation-service.js +484 -0
- zerochat_runtime/assets/js/cookies.js +688 -0
- zerochat_runtime/assets/js/data-reset-service.js +136 -0
- zerochat_runtime/assets/js/debug.js +508 -0
- zerochat_runtime/assets/js/defaults.js +16 -0
- zerochat_runtime/assets/js/export.js +179 -0
- zerochat_runtime/assets/js/file-parser.js +2088 -0
- zerochat_runtime/assets/js/generation-controller.js +417 -0
- zerochat_runtime/assets/js/i18n.js +1483 -0
- zerochat_runtime/assets/js/icons.js +156 -0
- zerochat_runtime/assets/js/ingestionEngine.js +436 -0
- zerochat_runtime/assets/js/markdown.js +588 -0
- zerochat_runtime/assets/js/mcp.js +1271 -0
- zerochat_runtime/assets/js/message-turns.js +62 -0
- zerochat_runtime/assets/js/profile-backup.js +118 -0
- zerochat_runtime/assets/js/profile-export-bundle.js +481 -0
- zerochat_runtime/assets/js/profile-repository.js +213 -0
- zerochat_runtime/assets/js/providers-webllm.js +486 -0
- zerochat_runtime/assets/js/providers.js +1560 -0
- zerochat_runtime/assets/js/rag-index.js +295 -0
- zerochat_runtime/assets/js/rag-service.js +530 -0
- zerochat_runtime/assets/js/rag-ui.js +997 -0
- zerochat_runtime/assets/js/ragStorage.js +676 -0
- zerochat_runtime/assets/js/sandbox.js +449 -0
- zerochat_runtime/assets/js/state.js +704 -0
- zerochat_runtime/assets/js/storage-db.js +144 -0
- zerochat_runtime/assets/js/tool-cards.js +330 -0
- zerochat_runtime/assets/js/tool-security.js +653 -0
- zerochat_runtime/assets/js/tools/README.md +50 -0
- zerochat_runtime/assets/js/tools/builtin/agent-checkpoint.tool.js +212 -0
- zerochat_runtime/assets/js/tools/builtin/download-pdf.tool.js +111 -0
- zerochat_runtime/assets/js/tools/builtin/execute-javascript.tool.js +219 -0
- zerochat_runtime/assets/js/tools/builtin/fetch-web-page.tool.js +112 -0
- zerochat_runtime/assets/js/tools/builtin/list-documents.tool.js +117 -0
- zerochat_runtime/assets/js/tools/builtin/read-knowledge-chunk.tool.js +129 -0
- zerochat_runtime/assets/js/tools/builtin/read-knowledge-image.tool.js +95 -0
- zerochat_runtime/assets/js/tools/builtin/render-chart.tool.js +90 -0
- zerochat_runtime/assets/js/tools/builtin/search-knowledge-base.tool.js +124 -0
- zerochat_runtime/assets/js/tools/builtin/search-web.tool.js +125 -0
- zerochat_runtime/assets/js/tools/tool-manifest.js +53 -0
- zerochat_runtime/assets/js/tools/tool-runtime.js +77 -0
- zerochat_runtime/assets/js/ui-composer.js +319 -0
- zerochat_runtime/assets/js/ui-conversation.js +655 -0
- zerochat_runtime/assets/js/ui-dialogs.js +132 -0
- zerochat_runtime/assets/js/ui-generation-status.js +119 -0
- zerochat_runtime/assets/js/ui-inspector.js +804 -0
- zerochat_runtime/assets/js/ui-mcp.js +604 -0
- zerochat_runtime/assets/js/ui-profiles.js +824 -0
- zerochat_runtime/assets/js/ui-reasoning.js +146 -0
- zerochat_runtime/assets/js/ui-settings.js +825 -0
- zerochat_runtime/assets/js/ui-shell.js +215 -0
- zerochat_runtime/assets/js/ui-sidebar.js +429 -0
- zerochat_runtime/assets/js/ui-telemetry.js +404 -0
- zerochat_runtime/assets/js/ui-transfer.js +262 -0
- zerochat_runtime/assets/js/utils.js +216 -0
- zerochat_runtime/assets/js/vendor/orama.browser.js +11 -0
- zerochat_runtime/assets/js/web-browser.js +569 -0
- zerochat_runtime/assets/js/web-search.js +519 -0
- zerochat_runtime/assets/manifest.webmanifest +20 -0
- zerochat_runtime/assets/services/dummy_mcp/dummy_mcp_server.py +47 -0
- zerochat_runtime/assets/services/dummy_mcp/service.json +22 -0
- zerochat_runtime/assets/services/lsp/installer.json +9 -0
- zerochat_runtime/assets/services/lsp/service.json +22 -0
- zerochat_runtime/assets/services/memory/installer.json +9 -0
- zerochat_runtime/assets/services/memory/service.json +24 -0
- zerochat_runtime/assets/services/playwright/installer.json +10 -0
- zerochat_runtime/assets/services/playwright/service.json +39 -0
- zerochat_runtime/assets/sw.js +164 -0
- zerochat_runtime/assets/zerochat.html +671 -0
|
@@ -0,0 +1,2088 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Módulo de procesamiento y extracción de contenido de archivos (ChatFileParser).
|
|
3
|
+
* Compatible con file:// y http:// sin dependencias externas.
|
|
4
|
+
* Soporta:
|
|
5
|
+
* - Documentos PDF (extracción y descompresión de texto en streams FlateDecode y objetos BT/ET).
|
|
6
|
+
* - Archivos de código y texto plano (.txt, .md, .js, .py, .html, .css, .json, .csv, etc.).
|
|
7
|
+
* - Imágenes (.png, .jpg, .jpeg, .webp, .gif, .svg).
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
(function (root, factory) {
|
|
11
|
+
if (typeof exports === 'object' && typeof module !== 'undefined') {
|
|
12
|
+
module.exports = factory();
|
|
13
|
+
} else {
|
|
14
|
+
root.ChatFileParser = factory();
|
|
15
|
+
}
|
|
16
|
+
})(typeof self !== 'undefined' ? self : this, function () {
|
|
17
|
+
'use strict';
|
|
18
|
+
|
|
19
|
+
function formatBytes(bytes) {
|
|
20
|
+
if (bytes === 0) return '0 B';
|
|
21
|
+
const k = 1024;
|
|
22
|
+
const sizes = ['B', 'KB', 'MB', 'GB'];
|
|
23
|
+
const i = Math.floor(Math.log(bytes) / Math.log(k));
|
|
24
|
+
return parseFloat((bytes / Math.pow(k, i)).toFixed(1)) + ' ' + sizes[i];
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function isPdfDelimiterOrWs(charCode) {
|
|
28
|
+
return charCode <= 32 || charCode === 40 || charCode === 41 || charCode === 60 ||
|
|
29
|
+
charCode === 62 || charCode === 91 || charCode === 93 || charCode === 47 || charCode === 37;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function decodePdfHexString(hex) {
|
|
33
|
+
if (!hex) return '';
|
|
34
|
+
let cleanHex = hex.replace(/\s+/g, '');
|
|
35
|
+
if (cleanHex.length % 2 !== 0) cleanHex += '0';
|
|
36
|
+
let str = '';
|
|
37
|
+
|
|
38
|
+
if (cleanHex.toUpperCase().startsWith('FEFF')) {
|
|
39
|
+
for (let i = 4; i < cleanHex.length; i += 4) {
|
|
40
|
+
const code = parseInt(cleanHex.slice(i, i + 4), 16);
|
|
41
|
+
if (!isNaN(code) && code > 0) str += String.fromCharCode(code);
|
|
42
|
+
}
|
|
43
|
+
return str;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
for (let i = 0; i < cleanHex.length; i += 2) {
|
|
47
|
+
const code = parseInt(cleanHex.slice(i, i + 2), 16);
|
|
48
|
+
if (!isNaN(code) && code >= 32) str += String.fromCharCode(code);
|
|
49
|
+
}
|
|
50
|
+
return str;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
function decodePdfEscapes(str) {
|
|
54
|
+
if (!str) return '';
|
|
55
|
+
return str.replace(/\\([0-7]{1,3})/g, (m, oct) => {
|
|
56
|
+
const code = parseInt(oct, 8);
|
|
57
|
+
return String.fromCharCode(code);
|
|
58
|
+
}).replace(/\\([nrtbf\\()])/g, (m, esc) => {
|
|
59
|
+
switch (esc) {
|
|
60
|
+
case 'n': return '\n';
|
|
61
|
+
case 'r': return '\r';
|
|
62
|
+
case 't': return '\t';
|
|
63
|
+
case 'b': return '\b';
|
|
64
|
+
case 'f': return '\f';
|
|
65
|
+
default: return esc;
|
|
66
|
+
}
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function parseCMapData(text, cmap) {
|
|
71
|
+
if (!text || typeof text !== 'string') return;
|
|
72
|
+
|
|
73
|
+
// 1. Extraer bloques beginbfchar ... endbfchar
|
|
74
|
+
const bfcharBlockRegex = /beginbfchar([\s\S]*?)endbfchar/g;
|
|
75
|
+
let block;
|
|
76
|
+
while ((block = bfcharBlockRegex.exec(text)) !== null) {
|
|
77
|
+
const blockText = block[1];
|
|
78
|
+
const bfcharRegex = /<([0-9a-fA-F]+)>\s*<([0-9a-fA-F]+)>/g;
|
|
79
|
+
let cm;
|
|
80
|
+
while ((cm = bfcharRegex.exec(blockText)) !== null) {
|
|
81
|
+
const src = cm[1].toLowerCase();
|
|
82
|
+
const dstHex = cm[2];
|
|
83
|
+
let dstChar = '';
|
|
84
|
+
for (let k = 0; k < dstHex.length; k += 4) {
|
|
85
|
+
const code = parseInt(dstHex.slice(k, k + 4), 16);
|
|
86
|
+
if (!isNaN(code)) dstChar += String.fromCharCode(code);
|
|
87
|
+
}
|
|
88
|
+
if (dstChar) {
|
|
89
|
+
cmap.set(src, dstChar);
|
|
90
|
+
if (src.length === 2) cmap.set('00' + src, dstChar);
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// 2. Extraer bloques beginbfrange ... endbfrange
|
|
96
|
+
const bfrangeBlockRegex = /beginbfrange([\s\S]*?)endbfrange/g;
|
|
97
|
+
while ((block = bfrangeBlockRegex.exec(text)) !== null) {
|
|
98
|
+
const blockText = block[1];
|
|
99
|
+
|
|
100
|
+
// 2a. beginbfrange con array de destinos: <start> <end> [ <dest1> <dest2> ... ]
|
|
101
|
+
const bfrangeArrayRegex = /<([0-9a-fA-F]+)>\s*<([0-9a-fA-F]+)>\s*\[([\s\S]*?)\]/g;
|
|
102
|
+
let cm;
|
|
103
|
+
while ((cm = bfrangeArrayRegex.exec(blockText)) !== null) {
|
|
104
|
+
const start = parseInt(cm[1], 16);
|
|
105
|
+
const end = parseInt(cm[2], 16);
|
|
106
|
+
const len = cm[1].length;
|
|
107
|
+
const destMatches = cm[3].match(/<([0-9a-fA-F]+)>/g) || [];
|
|
108
|
+
for (let s = start, idx = 0; s <= end && idx < destMatches.length; s++, idx++) {
|
|
109
|
+
const srcHex = s.toString(16).padStart(len, '0').toLowerCase();
|
|
110
|
+
const dstHex = destMatches[idx].replace(/[<>]/g, '');
|
|
111
|
+
let dstChar = '';
|
|
112
|
+
for (let k = 0; k < dstHex.length; k += 4) {
|
|
113
|
+
const code = parseInt(dstHex.slice(k, k + 4), 16);
|
|
114
|
+
if (!isNaN(code)) dstChar += String.fromCharCode(code);
|
|
115
|
+
}
|
|
116
|
+
if (dstChar) {
|
|
117
|
+
cmap.set(srcHex, dstChar);
|
|
118
|
+
if (srcHex.length === 2) cmap.set('00' + srcHex, dstChar);
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
// 2b. beginbfrange con destino simple: <start> <end> <destStart>
|
|
124
|
+
const cleanBlockText = blockText.replace(/<[0-9a-fA-F]+>\s*<[0-9a-fA-F]+>\s*\[[\s\S]*?\]/g, '');
|
|
125
|
+
const bfrangeSimpleRegex = /<([0-9a-fA-F]+)>\s*<([0-9a-fA-F]+)>\s*<([0-9a-fA-F]+)>/g;
|
|
126
|
+
while ((cm = bfrangeSimpleRegex.exec(cleanBlockText)) !== null) {
|
|
127
|
+
const start = parseInt(cm[1], 16);
|
|
128
|
+
const end = parseInt(cm[2], 16);
|
|
129
|
+
const destStart = parseInt(cm[3], 16);
|
|
130
|
+
const len = cm[1].length;
|
|
131
|
+
for (let s = start; s <= end; s++) {
|
|
132
|
+
const srcHex = s.toString(16).padStart(len, '0').toLowerCase();
|
|
133
|
+
const dstCode = destStart + (s - start);
|
|
134
|
+
const dstChar = String.fromCharCode(dstCode);
|
|
135
|
+
cmap.set(srcHex, dstChar);
|
|
136
|
+
if (srcHex.length === 2) cmap.set('00' + srcHex, dstChar);
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
async function parseCMaps(allObjects, fullText, bytes, objOffsets, security = null) {
|
|
143
|
+
const cmap = new Map();
|
|
144
|
+
const toUnicodeObjNums = new Set();
|
|
145
|
+
const cmapByObject = new Map();
|
|
146
|
+
|
|
147
|
+
// Buscar referencias /ToUnicode en fullText y en todos los objetos (incluidos los de ObjStm)
|
|
148
|
+
const toUnicodeRegex = /\/ToUnicode\s+(\d+)\s+\d+\s+R/g;
|
|
149
|
+
let m;
|
|
150
|
+
while ((m = toUnicodeRegex.exec(fullText)) !== null) {
|
|
151
|
+
toUnicodeObjNums.add(m[1]);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
for (const [num, body] of allObjects.entries()) {
|
|
155
|
+
let bm;
|
|
156
|
+
const bRegex = /\/ToUnicode\s+(\d+)\s+\d+\s+R/g;
|
|
157
|
+
while ((bm = bRegex.exec(body)) !== null) {
|
|
158
|
+
toUnicodeObjNums.add(bm[1]);
|
|
159
|
+
}
|
|
160
|
+
// Si el objeto ya contiene directamente CMap
|
|
161
|
+
if (body.includes('beginbfchar') || body.includes('beginbfrange')) {
|
|
162
|
+
parseCMapData(body, cmap);
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
for (const objNum of toUnicodeObjNums) {
|
|
167
|
+
const localCmap = new Map();
|
|
168
|
+
const body = allObjects.get(String(objNum));
|
|
169
|
+
if (body && (body.includes('beginbfchar') || body.includes('beginbfrange'))) {
|
|
170
|
+
parseCMapData(body, localCmap);
|
|
171
|
+
parseCMapData(body, cmap);
|
|
172
|
+
cmapByObject.set(String(objNum), localCmap);
|
|
173
|
+
continue;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
const offset = objOffsets.get(String(objNum));
|
|
177
|
+
if (offset !== undefined) {
|
|
178
|
+
const streamIdx = fullText.indexOf('stream', offset);
|
|
179
|
+
const endStreamIdx = fullText.indexOf('endstream', streamIdx);
|
|
180
|
+
if (streamIdx !== -1 && endStreamIdx !== -1) {
|
|
181
|
+
let dataStart = streamIdx + 6;
|
|
182
|
+
if (fullText.charCodeAt(dataStart) === 13) dataStart++;
|
|
183
|
+
if (fullText.charCodeAt(dataStart) === 10) dataStart++;
|
|
184
|
+
let dataEnd = endStreamIdx;
|
|
185
|
+
while (dataEnd > dataStart && (fullText.charCodeAt(dataEnd - 1) === 10 || fullText.charCodeAt(dataEnd - 1) === 13 || fullText.charCodeAt(dataEnd - 1) === 32)) {
|
|
186
|
+
dataEnd--;
|
|
187
|
+
}
|
|
188
|
+
try {
|
|
189
|
+
const rawBytes = bytes.subarray(dataStart, dataEnd);
|
|
190
|
+
const generation = Number(body?.match(/^\s*\d+\s+(\d+)\s+obj/)?.[1] || 0);
|
|
191
|
+
const streamBytes = security?.encrypted
|
|
192
|
+
? decryptPdfObjectBytes(rawBytes, security, Number(objNum), generation)
|
|
193
|
+
: rawBytes;
|
|
194
|
+
const decomp = await decompressDeflateData(streamBytes);
|
|
195
|
+
const toParse = decomp ? new TextDecoder('latin1').decode(decomp) : (streamBytes ? new TextDecoder('latin1').decode(streamBytes) : '');
|
|
196
|
+
if (toParse && (toParse.includes('beginbfchar') || toParse.includes('beginbfrange'))) {
|
|
197
|
+
parseCMapData(toParse, localCmap);
|
|
198
|
+
parseCMapData(toParse, cmap);
|
|
199
|
+
cmapByObject.set(String(objNum), localCmap);
|
|
200
|
+
}
|
|
201
|
+
} catch (e) {}
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
if (localCmap.size > 0) cmapByObject.set(String(objNum), localCmap);
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
// Los códigos CID solo son únicos dentro de una fuente. Mantener también
|
|
208
|
+
// los mapas por nombre de recurso (/F1, /F2...) evita que el último CMap
|
|
209
|
+
// leído corrompa texto y cifras pertenecientes a otra fuente.
|
|
210
|
+
const cmapByFontObject = new Map();
|
|
211
|
+
for (const [objNum, body] of allObjects.entries()) {
|
|
212
|
+
const match = body.match(/\/ToUnicode\s+(\d+)\s+\d+\s+R/);
|
|
213
|
+
if (!match) continue;
|
|
214
|
+
const localCmap = cmapByObject.get(String(match[1]));
|
|
215
|
+
if (localCmap) cmapByFontObject.set(String(objNum), localCmap);
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
const byFontName = new Map();
|
|
219
|
+
for (const body of allObjects.values()) {
|
|
220
|
+
const resourceRegex = /\/([^\s/<>\[\]()]+)\s+(\d+)\s+\d+\s+R/g;
|
|
221
|
+
let resourceMatch;
|
|
222
|
+
while ((resourceMatch = resourceRegex.exec(body)) !== null) {
|
|
223
|
+
const localCmap = cmapByFontObject.get(String(resourceMatch[2]));
|
|
224
|
+
if (localCmap && !byFontName.has(resourceMatch[1])) {
|
|
225
|
+
byFontName.set(resourceMatch[1], localCmap);
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
cmap.byFontName = byFontName;
|
|
230
|
+
|
|
231
|
+
return cmap;
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
function isReadablePdfText(str) {
|
|
235
|
+
if (!str || str.length < 3) return false;
|
|
236
|
+
let printable = 0;
|
|
237
|
+
for (let i = 0; i < str.length; i++) {
|
|
238
|
+
const code = str.charCodeAt(i);
|
|
239
|
+
if ((code >= 32 && code <= 126) || (code >= 160 && code <= 255) || code === 10 || code === 13 || code === 9) {
|
|
240
|
+
printable++;
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
const ratio = printable / str.length;
|
|
244
|
+
if (/SF\d{6}|afii\d+|upblock|dnblock|triagup|dmacron/.test(str)) return false;
|
|
245
|
+
return ratio >= 0.70;
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
function mapPdfLiteralString(lit, cmap) {
|
|
249
|
+
const raw = decodePdfEscapes(lit);
|
|
250
|
+
if (!cmap || cmap.size === 0 || !raw) return raw;
|
|
251
|
+
|
|
252
|
+
let decoded = '';
|
|
253
|
+
let hasMapping = false;
|
|
254
|
+
for (let k = 0; k < raw.length; k++) {
|
|
255
|
+
const c1 = raw.charCodeAt(k);
|
|
256
|
+
const hex1 = c1.toString(16).padStart(2, '0').toLowerCase();
|
|
257
|
+
|
|
258
|
+
// Probar secuencia de 2 bytes (UTF-16BE / 2-byte CID)
|
|
259
|
+
if (k + 1 < raw.length) {
|
|
260
|
+
const c2 = raw.charCodeAt(k + 1);
|
|
261
|
+
const hex2 = hex1 + c2.toString(16).padStart(2, '0').toLowerCase();
|
|
262
|
+
if (cmap.has(hex2)) {
|
|
263
|
+
decoded += cmap.get(hex2);
|
|
264
|
+
hasMapping = true;
|
|
265
|
+
k++;
|
|
266
|
+
continue;
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// Probar byte nulo + carácter en UTF-16BE estándar
|
|
271
|
+
if (c1 === 0 && k + 1 < raw.length) {
|
|
272
|
+
const c2 = raw.charCodeAt(k + 1);
|
|
273
|
+
const hex2 = '00' + c2.toString(16).padStart(2, '0').toLowerCase();
|
|
274
|
+
if (cmap.has(hex2)) {
|
|
275
|
+
decoded += cmap.get(hex2);
|
|
276
|
+
hasMapping = true;
|
|
277
|
+
} else if (c2 >= 32 && c2 <= 126) {
|
|
278
|
+
decoded += raw.charAt(k + 1);
|
|
279
|
+
}
|
|
280
|
+
k++;
|
|
281
|
+
continue;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
// Probar 1 byte mapeado en CMap
|
|
285
|
+
if (cmap.has(hex1)) {
|
|
286
|
+
decoded += cmap.get(hex1);
|
|
287
|
+
hasMapping = true;
|
|
288
|
+
} else if (c1 >= 32 && c1 <= 126) {
|
|
289
|
+
decoded += raw.charAt(k);
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
return (hasMapping || decoded.length >= raw.length * 0.5) ? decoded : raw;
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
function parsePdfStreamText(streamString, cmap = new Map()) {
|
|
297
|
+
if (!streamString || typeof streamString !== 'string') return '';
|
|
298
|
+
|
|
299
|
+
let out = [];
|
|
300
|
+
let inTextObject = false;
|
|
301
|
+
let activeCmap = cmap;
|
|
302
|
+
const len = streamString.length;
|
|
303
|
+
let i = 0;
|
|
304
|
+
|
|
305
|
+
while (i < len) {
|
|
306
|
+
const c = streamString.charCodeAt(i);
|
|
307
|
+
|
|
308
|
+
// Check BT (Begin Text)
|
|
309
|
+
if (c === 66 /* B */ && i + 1 < len && streamString.charCodeAt(i + 1) === 84 /* T */) {
|
|
310
|
+
const prev = i > 0 ? streamString.charCodeAt(i - 1) : 32;
|
|
311
|
+
const next = i + 2 < len ? streamString.charCodeAt(i + 2) : 32;
|
|
312
|
+
if (isPdfDelimiterOrWs(prev) && isPdfDelimiterOrWs(next)) {
|
|
313
|
+
inTextObject = true;
|
|
314
|
+
i += 2;
|
|
315
|
+
continue;
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
// Check ET (End Text)
|
|
320
|
+
if (c === 69 /* E */ && i + 1 < len && streamString.charCodeAt(i + 1) === 84 /* T */) {
|
|
321
|
+
const prev = i > 0 ? streamString.charCodeAt(i - 1) : 32;
|
|
322
|
+
const next = i + 2 < len ? streamString.charCodeAt(i + 2) : 32;
|
|
323
|
+
if (isPdfDelimiterOrWs(prev) && isPdfDelimiterOrWs(next)) {
|
|
324
|
+
inTextObject = false;
|
|
325
|
+
out.push('\n');
|
|
326
|
+
i += 2;
|
|
327
|
+
continue;
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
if (!inTextObject) {
|
|
332
|
+
i++;
|
|
333
|
+
continue;
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
// Seleccionar el ToUnicode de la fuente activa indicada por el operador
|
|
337
|
+
// PDF "/Fname size Tf". El mapa agregado queda como fallback.
|
|
338
|
+
if (c === 47 /* / */ && cmap && cmap.byFontName) {
|
|
339
|
+
const fontMatch = streamString.substring(i).match(/^\/([^\s/<>\[\]()]+)\s+[-+]?(?:\d+\.?\d*|\.\d+)\s+Tf\b/);
|
|
340
|
+
if (fontMatch) {
|
|
341
|
+
activeCmap = cmap.byFontName.get(fontMatch[1]) || cmap;
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
// Literal string: (texto)
|
|
346
|
+
if (c === 40 /* ( */) {
|
|
347
|
+
let depth = 1;
|
|
348
|
+
let j = i + 1;
|
|
349
|
+
let lit = '';
|
|
350
|
+
while (j < len && depth > 0) {
|
|
351
|
+
const sc = streamString.charCodeAt(j);
|
|
352
|
+
if (sc === 92 /* \ */) {
|
|
353
|
+
lit += streamString.charAt(j) + (j + 1 < len ? streamString.charAt(j + 1) : '');
|
|
354
|
+
j += 2;
|
|
355
|
+
continue;
|
|
356
|
+
}
|
|
357
|
+
if (sc === 40 /* ( */) depth++;
|
|
358
|
+
else if (sc === 41 /* ) */) {
|
|
359
|
+
depth--;
|
|
360
|
+
if (depth === 0) { j++; break; }
|
|
361
|
+
}
|
|
362
|
+
lit += streamString.charAt(j);
|
|
363
|
+
j++;
|
|
364
|
+
}
|
|
365
|
+
if (lit) out.push(mapPdfLiteralString(lit, activeCmap));
|
|
366
|
+
i = j;
|
|
367
|
+
continue;
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
// Hex string: <hex>
|
|
371
|
+
if (c === 60 /* < */) {
|
|
372
|
+
if (i + 1 < len && streamString.charCodeAt(i + 1) === 60) {
|
|
373
|
+
i += 2;
|
|
374
|
+
continue;
|
|
375
|
+
}
|
|
376
|
+
let j = i + 1;
|
|
377
|
+
let hex = '';
|
|
378
|
+
while (j < len && streamString.charCodeAt(j) !== 62 /* > */) {
|
|
379
|
+
const hc = streamString.charCodeAt(j);
|
|
380
|
+
if ((hc >= 48 && hc <= 57) || (hc >= 65 && hc <= 70) || (hc >= 97 && hc <= 102)) {
|
|
381
|
+
hex += streamString.charAt(j);
|
|
382
|
+
}
|
|
383
|
+
j++;
|
|
384
|
+
}
|
|
385
|
+
if (j < len && streamString.charCodeAt(j) === 62) j++;
|
|
386
|
+
let decoded = '';
|
|
387
|
+
for (let k = 0; k < hex.length; k += 4) {
|
|
388
|
+
const chunk = hex.slice(k, k + 4).toLowerCase();
|
|
389
|
+
if (activeCmap.has(chunk)) decoded += activeCmap.get(chunk);
|
|
390
|
+
else {
|
|
391
|
+
const sub2 = hex.slice(k, k + 2).toLowerCase();
|
|
392
|
+
if (activeCmap.has(sub2)) { decoded += activeCmap.get(sub2); k -= 2; }
|
|
393
|
+
else {
|
|
394
|
+
const code = parseInt(chunk, 16);
|
|
395
|
+
if (!isNaN(code) && code >= 32 && code < 127) decoded += String.fromCharCode(code);
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
}
|
|
399
|
+
if (decoded) out.push(decoded);
|
|
400
|
+
i = j;
|
|
401
|
+
continue;
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
// Array TJ: [(str) num (str)] TJ
|
|
405
|
+
if (c === 91 /* [ */) {
|
|
406
|
+
let j = i + 1;
|
|
407
|
+
let arrText = '';
|
|
408
|
+
while (j < len && streamString.charCodeAt(j) !== 93 /* ] */) {
|
|
409
|
+
const ac = streamString.charCodeAt(j);
|
|
410
|
+
if (ac === 40 /* ( */) {
|
|
411
|
+
let depth = 1;
|
|
412
|
+
let k = j + 1;
|
|
413
|
+
let lit = '';
|
|
414
|
+
while (k < len && depth > 0) {
|
|
415
|
+
const sc = streamString.charCodeAt(k);
|
|
416
|
+
if (sc === 92) {
|
|
417
|
+
lit += streamString.charAt(k) + (k + 1 < len ? streamString.charAt(k + 1) : '');
|
|
418
|
+
k += 2;
|
|
419
|
+
continue;
|
|
420
|
+
}
|
|
421
|
+
if (sc === 40) depth++;
|
|
422
|
+
else if (sc === 41) {
|
|
423
|
+
depth--;
|
|
424
|
+
if (depth === 0) { k++; break; }
|
|
425
|
+
}
|
|
426
|
+
lit += streamString.charAt(k);
|
|
427
|
+
k++;
|
|
428
|
+
}
|
|
429
|
+
arrText += mapPdfLiteralString(lit, activeCmap);
|
|
430
|
+
j = k;
|
|
431
|
+
continue;
|
|
432
|
+
} else if (ac === 60 /* < */) {
|
|
433
|
+
if (j + 1 < len && streamString.charCodeAt(j + 1) === 60) { j += 2; continue; }
|
|
434
|
+
let k = j + 1;
|
|
435
|
+
let hex = '';
|
|
436
|
+
while (k < len && streamString.charCodeAt(k) !== 62) {
|
|
437
|
+
const hc = streamString.charCodeAt(k);
|
|
438
|
+
if ((hc >= 48 && hc <= 57) || (hc >= 65 && hc <= 70) || (hc >= 97 && hc <= 102)) {
|
|
439
|
+
hex += streamString.charAt(k);
|
|
440
|
+
}
|
|
441
|
+
k++;
|
|
442
|
+
}
|
|
443
|
+
if (k < len && streamString.charCodeAt(k) === 62) k++;
|
|
444
|
+
for (let m = 0; m < hex.length; m += 4) {
|
|
445
|
+
const chunk = hex.slice(m, m + 4).toLowerCase();
|
|
446
|
+
if (activeCmap.has(chunk)) arrText += activeCmap.get(chunk);
|
|
447
|
+
else {
|
|
448
|
+
const sub2 = hex.slice(m, m + 2).toLowerCase();
|
|
449
|
+
if (activeCmap.has(sub2)) { arrText += activeCmap.get(sub2); m -= 2; }
|
|
450
|
+
}
|
|
451
|
+
}
|
|
452
|
+
j = k;
|
|
453
|
+
continue;
|
|
454
|
+
} else if ((ac >= 48 && ac <= 57) || ac === 45 /* - */) {
|
|
455
|
+
let numStr = '';
|
|
456
|
+
while (j < len && ((streamString.charCodeAt(j) >= 48 && streamString.charCodeAt(j) <= 57) || streamString.charCodeAt(j) === 45 || streamString.charCodeAt(j) === 46)) {
|
|
457
|
+
numStr += streamString.charAt(j);
|
|
458
|
+
j++;
|
|
459
|
+
}
|
|
460
|
+
const num = parseFloat(numStr);
|
|
461
|
+
if (!isNaN(num) && num < -100) arrText += ' ';
|
|
462
|
+
continue;
|
|
463
|
+
}
|
|
464
|
+
j++;
|
|
465
|
+
}
|
|
466
|
+
if (j < len && streamString.charCodeAt(j) === 93) j++;
|
|
467
|
+
if (arrText) out.push(arrText);
|
|
468
|
+
i = j;
|
|
469
|
+
continue;
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
// Operadores de salto de línea T*, Td, TD
|
|
473
|
+
if (c === 84 /* T */ && i + 1 < len) {
|
|
474
|
+
const nextC = streamString.charCodeAt(i + 1);
|
|
475
|
+
if (nextC === 42 /* * */ || nextC === 100 /* d */ || nextC === 68 /* D */) {
|
|
476
|
+
const after = i + 2 < len ? streamString.charCodeAt(i + 2) : 32;
|
|
477
|
+
if (isPdfDelimiterOrWs(after)) {
|
|
478
|
+
out.push('\n');
|
|
479
|
+
i += 2;
|
|
480
|
+
continue;
|
|
481
|
+
}
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
|
|
485
|
+
i++;
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
const res = out.join('').replace(/[ \t]+/g, ' ').replace(/\n\s*\n+/g, '\n\n').trim();
|
|
489
|
+
return decodePdfShiftedText(res);
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
const KNOWN_PDF_ANCHORS = new Set([
|
|
493
|
+
// English
|
|
494
|
+
'LIQUIDITY', 'COAL', 'MINED', 'ASSET', 'ASSETS', 'LIABILITY', 'LIABILITIES',
|
|
495
|
+
'EQUITY', 'REVENUE', 'REVENUES', 'PROFIT', 'PROFITS', 'INCOME', 'EXPENSE', 'EXPENSES',
|
|
496
|
+
'CASH', 'FLOW', 'FLOWS', 'BALANCE', 'SHEET', 'TOTAL', 'MARGIN', 'MARGINS',
|
|
497
|
+
'DIVIDEND', 'DIVIDENDS', 'EARNING', 'EARNINGS', 'SHARE', 'SHARES', 'DEBT',
|
|
498
|
+
'SALES', 'COST', 'COSTS', 'OPERATING', 'FINANCIAL', 'REPORT', 'REPORTS',
|
|
499
|
+
'TAX', 'TAXES', 'NET', 'GROSS', 'CAPITAL', 'EXPENDITURE', 'EXPENDITURES', 'INTEREST', 'PERIOD', 'QUARTER',
|
|
500
|
+
'ANNUAL', 'CURRENT', 'INVESTMENT', 'INVESTMENTS', 'DEPRECIATION', 'AMORTIZATION',
|
|
501
|
+
'PRODUCTION', 'TONNES', 'TONS', 'PRICE', 'PRICES', 'VOLUME', 'SEGMENT', 'RESULTS',
|
|
502
|
+
'AUDIT', 'AUDITED', 'COMPANY', 'CORPORATION', 'GROUP', 'CONSOLIDATED', 'MILLION', 'THOUSAND',
|
|
503
|
+
// Spanish
|
|
504
|
+
'LIQUIDEZ', 'ACTIVO', 'ACTIVOS', 'PASIVO', 'PASIVOS', 'PATRIMONIO', 'NETO',
|
|
505
|
+
'INGRESO', 'INGRESOS', 'GASTO', 'GASTOS', 'BENEFICIO', 'BENEFICIOS',
|
|
506
|
+
'RESULTADO', 'RESULTADOS', 'BALANCE', 'TOTAL', 'MARGEN', 'MARGENES',
|
|
507
|
+
'DIVIDENDO', 'DIVIDENDOS', 'CUENTA', 'CUENTAS', 'PERIODO', 'PERIODOS',
|
|
508
|
+
'EJERCICIO', 'EJERCICIOS', 'VENTA', 'VENTAS', 'COSTE', 'COSTES',
|
|
509
|
+
'FINANCIERO', 'FINANCIEROS', 'FINANCIERA', 'FINANCIERAS', 'INFORME', 'INFORMES',
|
|
510
|
+
'IMPUESTO', 'IMPUESTOS', 'EXPLOTACION', 'CONSOLIDADO', 'CONSOLIDADA',
|
|
511
|
+
'AUDITORIA', 'MEMORIA', 'CAPITAL', 'INTERES', 'INTERESES', 'INVERSION',
|
|
512
|
+
'INVERSIONES', 'DEPRECIACION', 'AMORTIZACION', 'PRODUCCION', 'TONELADAS',
|
|
513
|
+
'PRECIO', 'PRECIOS', 'VOLUMEN', 'EMPRESA', 'SOCIEDAD', 'GRUPO', 'MILLONES', 'MILES'
|
|
514
|
+
]);
|
|
515
|
+
|
|
516
|
+
function unshiftAsciiString(str, offset = 3) {
|
|
517
|
+
if (!str) return '';
|
|
518
|
+
let res = '';
|
|
519
|
+
for (let i = 0; i < str.length; i++) {
|
|
520
|
+
const code = str.charCodeAt(i);
|
|
521
|
+
if (code >= 33 && code <= 126) {
|
|
522
|
+
const unshifted = code - offset;
|
|
523
|
+
res += (unshifted >= 32 && unshifted <= 126) ? String.fromCharCode(unshifted) : str.charAt(i);
|
|
524
|
+
} else {
|
|
525
|
+
res += str.charAt(i);
|
|
526
|
+
}
|
|
527
|
+
}
|
|
528
|
+
return res;
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
function splitKnownConcatenatedWords(str) {
|
|
532
|
+
for (const kw of KNOWN_PDF_ANCHORS) {
|
|
533
|
+
if (str.startsWith(kw) && str.length > kw.length) {
|
|
534
|
+
const rest = str.slice(kw.length);
|
|
535
|
+
if (KNOWN_PDF_ANCHORS.has(rest)) {
|
|
536
|
+
return `${kw} ${rest}`;
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
}
|
|
540
|
+
return str;
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
function collapseSpacedLettersAndNumbers(text) {
|
|
544
|
+
if (!text || text.length < 3) return text;
|
|
545
|
+
let out = text.replace(/((?:[A-Za-z]\s+){2,}[A-Za-z])/g, (match) => {
|
|
546
|
+
const parts = match.split(/\s{2,}/);
|
|
547
|
+
return parts.map(p => {
|
|
548
|
+
const joined = p.replace(/\s+/g, '');
|
|
549
|
+
return splitKnownConcatenatedWords(joined.toUpperCase());
|
|
550
|
+
}).join(' ');
|
|
551
|
+
});
|
|
552
|
+
|
|
553
|
+
out = out.replace(/((?:[\d.,\-+()\/]\s+){2,}[\d.,\-+()\/])/g, (match) => {
|
|
554
|
+
return match.replace(/\s+/g, '');
|
|
555
|
+
});
|
|
556
|
+
|
|
557
|
+
return out;
|
|
558
|
+
}
|
|
559
|
+
|
|
560
|
+
/**
|
|
561
|
+
* Algunos generadores PDF emiten cada glifo como una operación de texto
|
|
562
|
+
* independiente. El extractor conserva esas operaciones como líneas, por lo
|
|
563
|
+
* que una página termina siendo "C\na\ns\nh\n\nF\nl\no\nw". Solo normalizamos
|
|
564
|
+
* páginas donde este patrón es dominante para no alterar documentos que
|
|
565
|
+
* realmente contienen listas de una letra o tablas verticales.
|
|
566
|
+
*/
|
|
567
|
+
function collapseVerticallySplitGlyphs(text) {
|
|
568
|
+
if (!text || typeof text !== 'string') return text;
|
|
569
|
+
|
|
570
|
+
return text.split(/(?=--- Página \d+ ---\n)/).map(section => {
|
|
571
|
+
const firstNewline = section.indexOf('\n');
|
|
572
|
+
if (firstNewline < 0) return section;
|
|
573
|
+
|
|
574
|
+
const heading = section.slice(0, firstNewline + 1);
|
|
575
|
+
const body = section.slice(firstNewline + 1);
|
|
576
|
+
const nonEmptyLines = body.split(/\r?\n/).map(line => line.trim()).filter(Boolean);
|
|
577
|
+
if (nonEmptyLines.length < 20) return section;
|
|
578
|
+
|
|
579
|
+
const singleGlyphLines = nonEmptyLines.filter(line => Array.from(line).length === 1).length;
|
|
580
|
+
if (singleGlyphLines / nonEmptyLines.length < 0.65) return section;
|
|
581
|
+
|
|
582
|
+
const blocks = body.split(/\r?\n\s*\r?\n+/).map(block => {
|
|
583
|
+
const lines = block.split(/\r?\n/).map(line => line.trim()).filter(Boolean);
|
|
584
|
+
if (lines.length >= 2 && lines.every(line => Array.from(line).length === 1)) {
|
|
585
|
+
return lines.join('');
|
|
586
|
+
}
|
|
587
|
+
return lines.join(' ');
|
|
588
|
+
}).filter(Boolean);
|
|
589
|
+
|
|
590
|
+
return heading + blocks.join(' ') + '\n\n';
|
|
591
|
+
}).join('');
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
function decodePdfShiftedText(text, offset = 3) {
|
|
595
|
+
if (!text || typeof text !== 'string' || text.length < 3) return text;
|
|
596
|
+
const lines = text.split(/\r?\n/);
|
|
597
|
+
let anyDecoded = false;
|
|
598
|
+
let anySpacingNormalized = false;
|
|
599
|
+
let tableContextActive = false;
|
|
600
|
+
|
|
601
|
+
const decodedLines = lines.map(line => {
|
|
602
|
+
const trimmed = line.trim();
|
|
603
|
+
if (!trimmed) {
|
|
604
|
+
tableContextActive = false;
|
|
605
|
+
return line;
|
|
606
|
+
}
|
|
607
|
+
|
|
608
|
+
// Mantener la ruta original para fuentes con desplazamiento +3. En los
|
|
609
|
+
// PDF normales, además, compactamos glifos separados para que una línea
|
|
610
|
+
// como "C a s h F l o w" sea indexable.
|
|
611
|
+
const candidate = unshiftAsciiString(trimmed, offset);
|
|
612
|
+
const collapsed = collapseSpacedLettersAndNumbers(candidate);
|
|
613
|
+
const rawCollapsed = collapseSpacedLettersAndNumbers(trimmed);
|
|
614
|
+
if (rawCollapsed !== trimmed) anySpacingNormalized = true;
|
|
615
|
+
|
|
616
|
+
const candidateUpper = collapsed.toUpperCase();
|
|
617
|
+
const tokens = candidateUpper.split(/[^A-Z]+/);
|
|
618
|
+
let keywordHits = 0;
|
|
619
|
+
for (const tok of tokens) {
|
|
620
|
+
if (tok.length >= 3 && KNOWN_PDF_ANCHORS.has(tok)) {
|
|
621
|
+
keywordHits++;
|
|
622
|
+
}
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
const hasBackslashGlitch = /\b[A-Za-z0-9\s]{2,}\\[\s\d]*/.test(trimmed) || /\\(?:\s+|$)/.test(trimmed);
|
|
626
|
+
const hasShiftedNumberFormat = /[0-9:<;]\s*[\/1]\s*[0-9:<;]/.test(trimmed);
|
|
627
|
+
|
|
628
|
+
if (keywordHits > 0 || hasBackslashGlitch || (tableContextActive && hasShiftedNumberFormat)) {
|
|
629
|
+
anyDecoded = true;
|
|
630
|
+
tableContextActive = true;
|
|
631
|
+
return collapsed.replace(/[ \t]+/g, ' ').trim();
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
return rawCollapsed;
|
|
635
|
+
});
|
|
636
|
+
|
|
637
|
+
return (anyDecoded || anySpacingNormalized) ? decodedLines.join('\n') : text;
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
async function decompressDeflateData(uint8Array) {
|
|
641
|
+
if (!uint8Array || uint8Array.length === 0) return null;
|
|
642
|
+
|
|
643
|
+
if (typeof DecompressionStream !== 'undefined') {
|
|
644
|
+
try {
|
|
645
|
+
const ds = new DecompressionStream('deflate');
|
|
646
|
+
const writer = ds.writable.getWriter();
|
|
647
|
+
const writePromise = writer.write(uint8Array).then(() => writer.close()).catch(() => {});
|
|
648
|
+
const res = new Response(ds.readable);
|
|
649
|
+
const buf = await res.arrayBuffer();
|
|
650
|
+
await writePromise;
|
|
651
|
+
return new Uint8Array(buf);
|
|
652
|
+
} catch (e) {}
|
|
653
|
+
|
|
654
|
+
try {
|
|
655
|
+
let rawSlice = uint8Array;
|
|
656
|
+
const isZlib = uint8Array.length > 6 && (uint8Array[0] & 0x0F) === 8 && (((uint8Array[0] << 8) | uint8Array[1]) % 31 === 0);
|
|
657
|
+
if (isZlib) {
|
|
658
|
+
rawSlice = uint8Array.subarray(2, uint8Array.length - 4);
|
|
659
|
+
}
|
|
660
|
+
const dsRaw = new DecompressionStream('deflate-raw');
|
|
661
|
+
const writer = dsRaw.writable.getWriter();
|
|
662
|
+
const writePromise = writer.write(rawSlice).then(() => writer.close()).catch(() => {});
|
|
663
|
+
const res = new Response(dsRaw.readable);
|
|
664
|
+
const buf = await res.arrayBuffer();
|
|
665
|
+
await writePromise;
|
|
666
|
+
return new Uint8Array(buf);
|
|
667
|
+
} catch (e) {}
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
if (typeof require !== 'undefined') {
|
|
671
|
+
try {
|
|
672
|
+
const zlib = require('zlib');
|
|
673
|
+
return zlib.inflateSync(uint8Array);
|
|
674
|
+
} catch (e) {
|
|
675
|
+
try {
|
|
676
|
+
const zlib = require('zlib');
|
|
677
|
+
return zlib.inflateRawSync(uint8Array);
|
|
678
|
+
} catch (e2) {}
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
return null;
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
function bytesToBase64(uint8Array) {
|
|
686
|
+
if (!uint8Array || uint8Array.length === 0) return '';
|
|
687
|
+
if (typeof Buffer !== 'undefined') return Buffer.from(uint8Array).toString('base64');
|
|
688
|
+
let binary = '';
|
|
689
|
+
const len = uint8Array.byteLength;
|
|
690
|
+
const chunkSize = 0x8000;
|
|
691
|
+
for (let i = 0; i < len; i += chunkSize) {
|
|
692
|
+
binary += String.fromCharCode.apply(null, uint8Array.subarray(i, Math.min(i + chunkSize, len)));
|
|
693
|
+
}
|
|
694
|
+
return btoa(binary);
|
|
695
|
+
}
|
|
696
|
+
|
|
697
|
+
/**
|
|
698
|
+
* Decodifica y convierte JPEGs en espacio de color CMYK / Adobe YCCK a formato sRGB legible.
|
|
699
|
+
*/
|
|
700
|
+
function convertCmykJpegToRgbDataUrl(data) {
|
|
701
|
+
try {
|
|
702
|
+
let offset = 0;
|
|
703
|
+
function readUint16() {
|
|
704
|
+
const val = (data[offset] << 8) | data[offset + 1];
|
|
705
|
+
offset += 2;
|
|
706
|
+
return val;
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
if (readUint16() !== 0xFFD8) return null;
|
|
710
|
+
|
|
711
|
+
const quantTables = [];
|
|
712
|
+
const huffmanTablesDC = [];
|
|
713
|
+
const huffmanTablesAC = [];
|
|
714
|
+
let frame = null;
|
|
715
|
+
let adobeTransform = -1;
|
|
716
|
+
|
|
717
|
+
while (offset < data.length) {
|
|
718
|
+
if (data[offset] !== 0xFF) { offset++; continue; }
|
|
719
|
+
while (data[offset] === 0xFF) offset++;
|
|
720
|
+
const marker = data[offset++];
|
|
721
|
+
|
|
722
|
+
if (marker === 0xD9) break;
|
|
723
|
+
if (marker === 0xDA) { // SOS
|
|
724
|
+
readUint16();
|
|
725
|
+
const numScanComponents = data[offset++];
|
|
726
|
+
const scanComponents = [];
|
|
727
|
+
for (let i = 0; i < numScanComponents; i++) {
|
|
728
|
+
const id = data[offset++];
|
|
729
|
+
const byte = data[offset++];
|
|
730
|
+
scanComponents.push({ id, dcTable: (byte >> 4) & 0x0F, acTable: byte & 0x0F });
|
|
731
|
+
}
|
|
732
|
+
offset += 3;
|
|
733
|
+
|
|
734
|
+
const scanBytes = [];
|
|
735
|
+
while (offset < data.length) {
|
|
736
|
+
if (data[offset] === 0xFF) {
|
|
737
|
+
if (data[offset + 1] === 0x00) {
|
|
738
|
+
scanBytes.push(0xFF);
|
|
739
|
+
offset += 2;
|
|
740
|
+
} else if (data[offset + 1] >= 0xD0 && data[offset + 1] <= 0xD7) {
|
|
741
|
+
offset += 2;
|
|
742
|
+
} else {
|
|
743
|
+
break;
|
|
744
|
+
}
|
|
745
|
+
} else {
|
|
746
|
+
scanBytes.push(data[offset++]);
|
|
747
|
+
}
|
|
748
|
+
}
|
|
749
|
+
|
|
750
|
+
if (!frame || frame.numComponents !== 4) return null;
|
|
751
|
+
|
|
752
|
+
let maxH = 1, maxV = 1;
|
|
753
|
+
for (const comp of frame.components) {
|
|
754
|
+
if (comp.hSample > maxH) maxH = comp.hSample;
|
|
755
|
+
if (comp.vSample > maxV) maxV = comp.vSample;
|
|
756
|
+
}
|
|
757
|
+
|
|
758
|
+
const mcuWidth = maxH * 8;
|
|
759
|
+
const mcuHeight = maxV * 8;
|
|
760
|
+
const mcusPerRow = Math.ceil(frame.width / mcuWidth);
|
|
761
|
+
const mcusPerCol = Math.ceil(frame.height / mcuHeight);
|
|
762
|
+
|
|
763
|
+
let bitPos = 0;
|
|
764
|
+
function readBit() {
|
|
765
|
+
const byteIdx = bitPos >> 3;
|
|
766
|
+
const bitIdx = 7 - (bitPos & 7);
|
|
767
|
+
bitPos++;
|
|
768
|
+
return (scanBytes[byteIdx] >> bitIdx) & 1;
|
|
769
|
+
}
|
|
770
|
+
function readBits(n) {
|
|
771
|
+
let val = 0;
|
|
772
|
+
for (let i = 0; i < n; i++) val = (val << 1) | readBit();
|
|
773
|
+
return val;
|
|
774
|
+
}
|
|
775
|
+
function readHuffman(tree) {
|
|
776
|
+
let node = tree;
|
|
777
|
+
while (node.sym === undefined) {
|
|
778
|
+
const bit = readBit();
|
|
779
|
+
node = node[bit];
|
|
780
|
+
if (!node) throw new Error('Invalid Huffman code');
|
|
781
|
+
}
|
|
782
|
+
return node.sym;
|
|
783
|
+
}
|
|
784
|
+
function extend(val, bits) {
|
|
785
|
+
const vt = 1 << (bits - 1);
|
|
786
|
+
if (val < vt) return val + (-1 << bits) + 1;
|
|
787
|
+
return val;
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
function idct(block, out) {
|
|
791
|
+
const temp = new Float64Array(64);
|
|
792
|
+
const C = Math.PI / 16;
|
|
793
|
+
for (let i = 0; i < 8; i++) {
|
|
794
|
+
for (let j = 0; j < 8; j++) {
|
|
795
|
+
let sum = 0;
|
|
796
|
+
for (let k = 0; k < 8; k++) {
|
|
797
|
+
const s = block[i * 8 + k];
|
|
798
|
+
if (s === 0) continue;
|
|
799
|
+
const c = k === 0 ? 0.7071067811865475 : 1;
|
|
800
|
+
sum += c * s * Math.cos((2 * j + 1) * k * C);
|
|
801
|
+
}
|
|
802
|
+
temp[i * 8 + j] = sum * 0.5;
|
|
803
|
+
}
|
|
804
|
+
}
|
|
805
|
+
for (let j = 0; j < 8; j++) {
|
|
806
|
+
for (let i = 0; i < 8; i++) {
|
|
807
|
+
let sum = 0;
|
|
808
|
+
for (let k = 0; k < 8; k++) {
|
|
809
|
+
const s = temp[k * 8 + j];
|
|
810
|
+
if (s === 0) continue;
|
|
811
|
+
const c = k === 0 ? 0.7071067811865475 : 1;
|
|
812
|
+
sum += c * s * Math.cos((2 * i + 1) * k * C);
|
|
813
|
+
}
|
|
814
|
+
let val = Math.round(sum * 0.5) + 128;
|
|
815
|
+
if (val < 0) val = 0;
|
|
816
|
+
else if (val > 255) val = 255;
|
|
817
|
+
out[i * 8 + j] = val;
|
|
818
|
+
}
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
const ZIGZAG = [
|
|
823
|
+
0, 1, 8, 16, 9, 2, 3, 10,
|
|
824
|
+
17, 24, 32, 25, 18, 11, 4, 5,
|
|
825
|
+
12, 19, 26, 33, 40, 48, 41, 34,
|
|
826
|
+
27, 20, 13, 6, 7, 14, 21, 28,
|
|
827
|
+
35, 42, 49, 56, 57, 50, 43, 36,
|
|
828
|
+
29, 22, 15, 23, 30, 37, 44, 51,
|
|
829
|
+
58, 59, 52, 45, 38, 31, 39, 46,
|
|
830
|
+
53, 60, 61, 54, 47, 55, 62, 63
|
|
831
|
+
];
|
|
832
|
+
|
|
833
|
+
const compBuffers = frame.components.map(comp => {
|
|
834
|
+
const w = mcusPerRow * comp.hSample * 8;
|
|
835
|
+
const h = mcusPerCol * comp.vSample * 8;
|
|
836
|
+
return {
|
|
837
|
+
data: new Uint8Array(w * h),
|
|
838
|
+
width: w,
|
|
839
|
+
height: h,
|
|
840
|
+
hSample: comp.hSample,
|
|
841
|
+
vSample: comp.vSample,
|
|
842
|
+
quantTable: quantTables[comp.quantId],
|
|
843
|
+
dcPred: 0
|
|
844
|
+
};
|
|
845
|
+
});
|
|
846
|
+
|
|
847
|
+
for (let mcuY = 0; mcuY < mcusPerCol; mcuY++) {
|
|
848
|
+
for (let mcuX = 0; mcuX < mcusPerRow; mcuX++) {
|
|
849
|
+
for (let c = 0; c < frame.numComponents; c++) {
|
|
850
|
+
const comp = compBuffers[c];
|
|
851
|
+
const scanComp = scanComponents.find(sc => sc.id === frame.components[c].id);
|
|
852
|
+
const dcTree = huffmanTablesDC[scanComp.dcTable];
|
|
853
|
+
const acTree = huffmanTablesAC[scanComp.acTable];
|
|
854
|
+
|
|
855
|
+
for (let v = 0; v < comp.vSample; v++) {
|
|
856
|
+
for (let h = 0; h < comp.hSample; h++) {
|
|
857
|
+
const block = new Int32Array(64);
|
|
858
|
+
const dcLen = readHuffman(dcTree);
|
|
859
|
+
let dcDiff = 0;
|
|
860
|
+
if (dcLen > 0) dcDiff = extend(readBits(dcLen), dcLen);
|
|
861
|
+
comp.dcPred += dcDiff;
|
|
862
|
+
block[0] = comp.dcPred * comp.quantTable[0];
|
|
863
|
+
|
|
864
|
+
let k = 1;
|
|
865
|
+
while (k < 64) {
|
|
866
|
+
const acByte = readHuffman(acTree);
|
|
867
|
+
const rrr = (acByte >> 4) & 0x0F;
|
|
868
|
+
const sss = acByte & 0x0F;
|
|
869
|
+
if (sss === 0) {
|
|
870
|
+
if (rrr === 0) break;
|
|
871
|
+
if (rrr === 15) { k += 16; continue; }
|
|
872
|
+
}
|
|
873
|
+
k += rrr;
|
|
874
|
+
if (k >= 64) break;
|
|
875
|
+
const acVal = extend(readBits(sss), sss);
|
|
876
|
+
block[ZIGZAG[k]] = acVal * comp.quantTable[ZIGZAG[k]];
|
|
877
|
+
k++;
|
|
878
|
+
}
|
|
879
|
+
|
|
880
|
+
const blockOut = new Uint8Array(64);
|
|
881
|
+
idct(block, blockOut);
|
|
882
|
+
|
|
883
|
+
const startX = (mcuX * comp.hSample + h) * 8;
|
|
884
|
+
const startY = (mcuY * comp.vSample + v) * 8;
|
|
885
|
+
for (let row = 0; row < 8; row++) {
|
|
886
|
+
const destOffset = (startY + row) * comp.width + startX;
|
|
887
|
+
for (let col = 0; col < 8; col++) {
|
|
888
|
+
comp.data[destOffset + col] = blockOut[row * 8 + col];
|
|
889
|
+
}
|
|
890
|
+
}
|
|
891
|
+
}
|
|
892
|
+
}
|
|
893
|
+
}
|
|
894
|
+
}
|
|
895
|
+
}
|
|
896
|
+
|
|
897
|
+
function getCompVal(cIdx, x, y) {
|
|
898
|
+
const b = compBuffers[cIdx];
|
|
899
|
+
return b.data[Math.floor(y * b.vSample / maxV) * b.width + Math.floor(x * b.hSample / maxH)];
|
|
900
|
+
}
|
|
901
|
+
|
|
902
|
+
const isYCCK = (adobeTransform === 2 || adobeTransform === -1);
|
|
903
|
+
const rgbBytes = new Uint8Array(frame.width * frame.height * 3);
|
|
904
|
+
let rgbPos = 0;
|
|
905
|
+
|
|
906
|
+
for (let y = 0; y < frame.height; y++) {
|
|
907
|
+
for (let x = 0; x < frame.width; x++) {
|
|
908
|
+
const c0 = getCompVal(0, x, y);
|
|
909
|
+
const c1 = getCompVal(1, x, y);
|
|
910
|
+
const c2 = getCompVal(2, x, y);
|
|
911
|
+
const c3 = getCompVal(3, x, y);
|
|
912
|
+
|
|
913
|
+
let r, g, b;
|
|
914
|
+
if (isYCCK) {
|
|
915
|
+
const yVal = c0;
|
|
916
|
+
const cb = c1 - 128;
|
|
917
|
+
const cr = c2 - 128;
|
|
918
|
+
const kNorm = (255 - c3) / 255;
|
|
919
|
+
r = (yVal + 1.402 * cr) * kNorm;
|
|
920
|
+
g = (yVal - 0.344136 * cb - 0.714136 * cr) * kNorm;
|
|
921
|
+
b = (yVal + 1.772 * cb) * kNorm;
|
|
922
|
+
} else {
|
|
923
|
+
const c = c0 / 255, m = c1 / 255, y_ = c2 / 255, k = c3 / 255;
|
|
924
|
+
r = 255 * (1 - c) * (1 - k);
|
|
925
|
+
g = 255 * (1 - m) * (1 - k);
|
|
926
|
+
b = 255 * (1 - y_) * (1 - k);
|
|
927
|
+
}
|
|
928
|
+
rgbBytes[rgbPos++] = Math.max(0, Math.min(255, Math.round(r)));
|
|
929
|
+
rgbBytes[rgbPos++] = Math.max(0, Math.min(255, Math.round(g)));
|
|
930
|
+
rgbBytes[rgbPos++] = Math.max(0, Math.min(255, Math.round(b)));
|
|
931
|
+
}
|
|
932
|
+
}
|
|
933
|
+
|
|
934
|
+
if (typeof document !== 'undefined' && typeof document.createElement === 'function') {
|
|
935
|
+
try {
|
|
936
|
+
const canvas = document.createElement('canvas');
|
|
937
|
+
canvas.width = frame.width;
|
|
938
|
+
canvas.height = frame.height;
|
|
939
|
+
const ctx = canvas.getContext('2d');
|
|
940
|
+
if (ctx) {
|
|
941
|
+
const imgData = ctx.createImageData(frame.width, frame.height);
|
|
942
|
+
const d = imgData.data;
|
|
943
|
+
let sIdx = 0;
|
|
944
|
+
for (let i = 0; i < d.length; i += 4) {
|
|
945
|
+
d[i] = rgbBytes[sIdx++];
|
|
946
|
+
d[i + 1] = rgbBytes[sIdx++];
|
|
947
|
+
d[i + 2] = rgbBytes[sIdx++];
|
|
948
|
+
d[i + 3] = 255;
|
|
949
|
+
}
|
|
950
|
+
ctx.putImageData(imgData, 0, 0);
|
|
951
|
+
return canvas.toDataURL('image/jpeg', 0.85);
|
|
952
|
+
}
|
|
953
|
+
} catch (e) {}
|
|
954
|
+
}
|
|
955
|
+
|
|
956
|
+
return encodeRgbToJpegDataUrl(frame.width, frame.height, rgbBytes, 80);
|
|
957
|
+
}
|
|
958
|
+
|
|
959
|
+
const length = readUint16();
|
|
960
|
+
const nextMarkerOffset = offset + length - 2;
|
|
961
|
+
|
|
962
|
+
if (marker === 0xDB) {
|
|
963
|
+
let p = offset;
|
|
964
|
+
while (p < nextMarkerOffset) {
|
|
965
|
+
const info = data[p++];
|
|
966
|
+
const tableId = info & 0x0F;
|
|
967
|
+
const is16Bit = (info >> 4) !== 0;
|
|
968
|
+
const table = new Int32Array(64);
|
|
969
|
+
for (let i = 0; i < 64; i++) {
|
|
970
|
+
table[i] = is16Bit ? ((data[p++] << 8) | data[p++]) : data[p++];
|
|
971
|
+
}
|
|
972
|
+
quantTables[tableId] = table;
|
|
973
|
+
}
|
|
974
|
+
} else if (marker === 0xC0 || marker === 0xC2) {
|
|
975
|
+
const precision = data[offset++];
|
|
976
|
+
const height = readUint16();
|
|
977
|
+
const width = readUint16();
|
|
978
|
+
const numComponents = data[offset++];
|
|
979
|
+
const components = [];
|
|
980
|
+
for (let i = 0; i < numComponents; i++) {
|
|
981
|
+
const id = data[offset++];
|
|
982
|
+
const byte = data[offset++];
|
|
983
|
+
const quantId = data[offset++];
|
|
984
|
+
components.push({ id, hSample: (byte >> 4) & 0x0F, vSample: byte & 0x0F, quantId });
|
|
985
|
+
}
|
|
986
|
+
frame = { precision, height, width, numComponents, components };
|
|
987
|
+
} else if (marker === 0xC4) {
|
|
988
|
+
let p = offset;
|
|
989
|
+
while (p < nextMarkerOffset) {
|
|
990
|
+
const info = data[p++];
|
|
991
|
+
const isAC = (info >> 4) !== 0;
|
|
992
|
+
const tableId = info & 0x0F;
|
|
993
|
+
const counts = new Uint8Array(16);
|
|
994
|
+
for (let i = 0; i < 16; i++) counts[i] = data[p++];
|
|
995
|
+
const symbols = [];
|
|
996
|
+
for (let i = 0; i < 16; i++) {
|
|
997
|
+
const syms = [];
|
|
998
|
+
for (let j = 0; j < counts[i]; j++) syms.push(data[p++]);
|
|
999
|
+
symbols.push(syms);
|
|
1000
|
+
}
|
|
1001
|
+
const tree = buildHuffmanTree(counts, symbols);
|
|
1002
|
+
if (isAC) huffmanTablesAC[tableId] = tree;
|
|
1003
|
+
else huffmanTablesDC[tableId] = tree;
|
|
1004
|
+
}
|
|
1005
|
+
} else if (marker === 0xEE) {
|
|
1006
|
+
if (length >= 14 && data[offset] === 0x41 && data[offset+1] === 0x64 && data[offset+2] === 0x6F && data[offset+3] === 0x62 && data[offset+4] === 0x65) {
|
|
1007
|
+
adobeTransform = data[offset + 11];
|
|
1008
|
+
}
|
|
1009
|
+
}
|
|
1010
|
+
offset = nextMarkerOffset;
|
|
1011
|
+
}
|
|
1012
|
+
} catch (e) {}
|
|
1013
|
+
return null;
|
|
1014
|
+
}
|
|
1015
|
+
|
|
1016
|
+
function buildHuffmanTree(counts, symbols) {
|
|
1017
|
+
const root = {};
|
|
1018
|
+
let code = 0;
|
|
1019
|
+
for (let len = 1; len <= 16; len++) {
|
|
1020
|
+
const syms = symbols[len - 1];
|
|
1021
|
+
for (let i = 0; i < syms.length; i++) {
|
|
1022
|
+
const sym = syms[i];
|
|
1023
|
+
let node = root;
|
|
1024
|
+
for (let bit = len - 1; bit >= 0; bit--) {
|
|
1025
|
+
const b = (code >> bit) & 1;
|
|
1026
|
+
if (!node[b]) node[b] = {};
|
|
1027
|
+
node = node[b];
|
|
1028
|
+
}
|
|
1029
|
+
node.sym = sym;
|
|
1030
|
+
code++;
|
|
1031
|
+
}
|
|
1032
|
+
code <<= 1;
|
|
1033
|
+
}
|
|
1034
|
+
return root;
|
|
1035
|
+
}
|
|
1036
|
+
|
|
1037
|
+
function encodeRgbToJpegDataUrl(width, height, rgb, quality = 80) {
|
|
1038
|
+
try {
|
|
1039
|
+
const q = Math.max(1, Math.min(100, quality));
|
|
1040
|
+
const scale = q < 50 ? Math.floor(5000 / q) : Math.floor(200 - q * 2);
|
|
1041
|
+
|
|
1042
|
+
const defaultYTable = [
|
|
1043
|
+
16, 11, 10, 16, 24, 40, 51, 61,
|
|
1044
|
+
12, 12, 14, 19, 26, 58, 60, 55,
|
|
1045
|
+
14, 13, 16, 24, 40, 57, 69, 56,
|
|
1046
|
+
14, 17, 22, 29, 51, 87, 80, 62,
|
|
1047
|
+
18, 22, 37, 56, 68, 109, 103, 77,
|
|
1048
|
+
24, 35, 55, 64, 81, 104, 113, 92,
|
|
1049
|
+
49, 64, 78, 87, 103, 121, 120, 101,
|
|
1050
|
+
72, 92, 95, 98, 112, 100, 103, 99
|
|
1051
|
+
];
|
|
1052
|
+
|
|
1053
|
+
const defaultUVTable = [
|
|
1054
|
+
17, 18, 24, 47, 99, 99, 99, 99,
|
|
1055
|
+
18, 21, 26, 66, 99, 99, 99, 99,
|
|
1056
|
+
24, 26, 56, 99, 99, 99, 99, 99,
|
|
1057
|
+
47, 66, 99, 99, 99, 99, 99, 99,
|
|
1058
|
+
99, 99, 99, 99, 99, 99, 99, 99,
|
|
1059
|
+
99, 99, 99, 99, 99, 99, 99, 99,
|
|
1060
|
+
99, 99, 99, 99, 99, 99, 99, 99,
|
|
1061
|
+
99, 99, 99, 99, 99, 99, 99, 99
|
|
1062
|
+
];
|
|
1063
|
+
|
|
1064
|
+
const yTable = new Uint8Array(64);
|
|
1065
|
+
const uvTable = new Uint8Array(64);
|
|
1066
|
+
for (let i = 0; i < 64; i++) {
|
|
1067
|
+
const yVal = Math.floor((defaultYTable[i] * scale + 50) / 100);
|
|
1068
|
+
const uvVal = Math.floor((defaultUVTable[i] * scale + 50) / 100);
|
|
1069
|
+
yTable[i] = Math.max(1, Math.min(255, yVal));
|
|
1070
|
+
uvTable[i] = Math.max(1, Math.min(255, uvVal));
|
|
1071
|
+
}
|
|
1072
|
+
|
|
1073
|
+
const ZIGZAG = [
|
|
1074
|
+
0, 1, 8, 16, 9, 2, 3, 10,
|
|
1075
|
+
17, 24, 32, 25, 18, 11, 4, 5,
|
|
1076
|
+
12, 19, 26, 33, 40, 48, 41, 34,
|
|
1077
|
+
27, 20, 13, 6, 7, 14, 21, 28,
|
|
1078
|
+
35, 42, 49, 56, 57, 50, 43, 36,
|
|
1079
|
+
29, 22, 15, 23, 30, 37, 44, 51,
|
|
1080
|
+
58, 59, 52, 45, 38, 31, 39, 46,
|
|
1081
|
+
53, 60, 61, 54, 47, 55, 62, 63
|
|
1082
|
+
];
|
|
1083
|
+
|
|
1084
|
+
const yQuant = new Float32Array(64);
|
|
1085
|
+
const uvQuant = new Float32Array(64);
|
|
1086
|
+
const AAN_SCALE = [1.0, 1.387039845, 1.306562965, 1.175875602, 1.0, 0.785694958, 0.541196100, 0.275899379];
|
|
1087
|
+
for (let row = 0; row < 8; row++) {
|
|
1088
|
+
for (let col = 0; col < 8; col++) {
|
|
1089
|
+
const idx = row * 8 + col;
|
|
1090
|
+
yQuant[idx] = 1.0 / (yTable[ZIGZAG[idx]] * AAN_SCALE[row] * AAN_SCALE[col] * 8);
|
|
1091
|
+
uvQuant[idx] = 1.0 / (uvTable[ZIGZAG[idx]] * AAN_SCALE[row] * AAN_SCALE[col] * 8);
|
|
1092
|
+
}
|
|
1093
|
+
}
|
|
1094
|
+
|
|
1095
|
+
const std_dc_lum_nr = [0, 0, 1, 5, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0, 0, 0];
|
|
1096
|
+
const std_dc_lum_val = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11];
|
|
1097
|
+
const std_dc_chr_nr = [0, 0, 3, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, 0, 0, 0, 0];
|
|
1098
|
+
const std_dc_chr_val = [0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11];
|
|
1099
|
+
|
|
1100
|
+
const std_ac_lum_nr = [0, 0, 2, 1, 3, 3, 2, 4, 3, 5, 5, 4, 4, 0, 0, 1, 0x7d];
|
|
1101
|
+
const std_ac_lum_val = [
|
|
1102
|
+
0x01, 0x02, 0x03, 0x00, 0x04, 0x11, 0x05, 0x12, 0x21, 0x31, 0x41, 0x06, 0x13, 0x51, 0x61, 0x07,
|
|
1103
|
+
0x22, 0x71, 0x14, 0x32, 0x81, 0x91, 0xa1, 0x08, 0x23, 0x42, 0xb1, 0xc1, 0x15, 0x52, 0xd1, 0xf0,
|
|
1104
|
+
0x24, 0x33, 0x62, 0x72, 0x82, 0x09, 0x0a, 0x16, 0x17, 0x18, 0x19, 0x1a, 0x25, 0x26, 0x27, 0x28,
|
|
1105
|
+
0x29, 0x2a, 0x34, 0x35, 0x36, 0x37, 0x38, 0x39, 0x3a, 0x43, 0x44, 0x45, 0x46, 0x47, 0x48, 0x49,
|
|
1106
|
+
0x4a, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58, 0x59, 0x5a, 0x63, 0x64, 0x65, 0x66, 0x67, 0x68, 0x69,
|
|
1107
|
+
0x6a, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, 0x79, 0x7a, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88, 0x89,
|
|
1108
|
+
0x8a, 0x92, 0x93, 0x94, 0x95, 0x96, 0x97, 0x98, 0x99, 0x9a, 0xa2, 0xa3, 0xa4, 0xa5, 0xa6, 0xa7,
|
|
1109
|
+
0xa8, 0xa9, 0xaa, 0xb2, 0xb3, 0xb4, 0xb5, 0xb6, 0xb7, 0xb8, 0xb9, 0xba, 0xc2, 0xc3, 0xc4, 0xc5,
|
|
1110
|
+
0xc6, 0xc7, 0xc8, 0xc9, 0xca, 0xd2, 0xd3, 0xd4, 0xd5, 0xd6, 0xd7, 0xd8, 0xd9, 0xda, 0xe1, 0xe2,
|
|
1111
|
+
0xe3, 0xe4, 0xe5, 0xe6, 0xe7, 0xe8, 0xe9, 0xea, 0xf1, 0xf2, 0xf3, 0xf4, 0xf5, 0xf6, 0xf7, 0xf8,
|
|
1112
|
+
0xf9, 0xfa
|
|
1113
|
+
];
|
|
1114
|
+
|
|
1115
|
+
const std_ac_chr_nr = [0, 0, 2, 1, 2, 4, 4, 3, 4, 7, 5, 4, 4, 0, 1, 2, 0x77];
|
|
1116
|
+
const std_ac_chr_val = [
|
|
1117
|
+
0x00, 0x01, 0x02, 0x03, 0x11, 0x04, 0x05, 0x21, 0x31, 0x06, 0x12, 0x41, 0x51, 0x07, 0x61, 0x71,
|
|
1118
|
+
0x13, 0x22, 0x32, 0x81, 0x08, 0x14, 0x42, 0x91, 0xa1, 0xb1, 0xc1, 0x09, 0x23, 0x33, 0x52, 0xf0,
|
|
1119
|
+
0x15, 0x62, 0x72, 0xd1, 0x0a, 0x16, 0x24, 0x34, 0xe1, 0x25, 0xf1, 0x17, 0x18, 0x19, 0x1a, 0x26,
|
|
1120
|
+
0x27, 0x28, 0x29, 0x2a, 0x35, 0x36, 0x37, 0x38, 0x39, 0x3a, 0x43, 0x44, 0x45, 0x46, 0x47, 0x48,
|
|
1121
|
+
0x49, 0x4a, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58, 0x59, 0x5a, 0x63, 0x64, 0x65, 0x66, 0x67, 0x68,
|
|
1122
|
+
0x69, 0x6a, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, 0x79, 0x7a, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87,
|
|
1123
|
+
0x88, 0x89, 0x8a, 0x92, 0x93, 0x94, 0x95, 0x96, 0x97, 0x98, 0x99, 0x9a, 0xa2, 0xa3, 0xa4, 0xa5,
|
|
1124
|
+
0xa6, 0xa7, 0xa8, 0xa9, 0xaa, 0xb2, 0xb3, 0xb4, 0xb5, 0xb6, 0xb7, 0xb8, 0xb9, 0xba, 0xc2, 0xc3,
|
|
1125
|
+
0xc4, 0xc5, 0xc6, 0xc7, 0xc8, 0xc9, 0xca, 0xd2, 0xd3, 0xd4, 0xd5, 0xd6, 0xd7, 0xd8, 0xd9, 0xda,
|
|
1126
|
+
0xe2, 0xe3, 0xe4, 0xe5, 0xe6, 0xe7, 0xe8, 0xe9, 0xea, 0xf2, 0xf3, 0xf4, 0xf5, 0xf6, 0xf7, 0xf8,
|
|
1127
|
+
0xf9, 0xfa
|
|
1128
|
+
];
|
|
1129
|
+
|
|
1130
|
+
function computeHuffmanTable(nrcodes, values) {
|
|
1131
|
+
const huff = [];
|
|
1132
|
+
let code = 0;
|
|
1133
|
+
let k = 0;
|
|
1134
|
+
for (let i = 1; i <= 16; i++) {
|
|
1135
|
+
for (let j = 1; j <= nrcodes[i]; j++) {
|
|
1136
|
+
huff[values[k]] = { code, len: i };
|
|
1137
|
+
k++;
|
|
1138
|
+
code++;
|
|
1139
|
+
}
|
|
1140
|
+
code <<= 1;
|
|
1141
|
+
}
|
|
1142
|
+
return huff;
|
|
1143
|
+
}
|
|
1144
|
+
|
|
1145
|
+
const dcLumHuff = computeHuffmanTable(std_dc_lum_nr, std_dc_lum_val);
|
|
1146
|
+
const dcChrHuff = computeHuffmanTable(std_dc_chr_nr, std_dc_chr_val);
|
|
1147
|
+
const acLumHuff = computeHuffmanTable(std_ac_lum_nr, std_ac_lum_val);
|
|
1148
|
+
const acChrHuff = computeHuffmanTable(std_ac_chr_nr, std_ac_chr_val);
|
|
1149
|
+
|
|
1150
|
+
const byteStream = [];
|
|
1151
|
+
let bitBuf = 0;
|
|
1152
|
+
let bitCnt = 0;
|
|
1153
|
+
|
|
1154
|
+
function writeBits(code, len) {
|
|
1155
|
+
bitBuf = (bitBuf << len) | code;
|
|
1156
|
+
bitCnt += len;
|
|
1157
|
+
while (bitCnt >= 8) {
|
|
1158
|
+
const b = (bitBuf >> (bitCnt - 8)) & 0xFF;
|
|
1159
|
+
byteStream.push(b);
|
|
1160
|
+
if (b === 0xFF) byteStream.push(0x00);
|
|
1161
|
+
bitCnt -= 8;
|
|
1162
|
+
}
|
|
1163
|
+
}
|
|
1164
|
+
|
|
1165
|
+
function flushBits() {
|
|
1166
|
+
if (bitCnt > 0) {
|
|
1167
|
+
const b = (bitBuf << (8 - bitCnt)) & 0xFF;
|
|
1168
|
+
byteStream.push(b);
|
|
1169
|
+
if (b === 0xFF) byteStream.push(0x00);
|
|
1170
|
+
bitBuf = 0;
|
|
1171
|
+
bitCnt = 0;
|
|
1172
|
+
}
|
|
1173
|
+
}
|
|
1174
|
+
|
|
1175
|
+
function writeWord(w) {
|
|
1176
|
+
byteStream.push((w >> 8) & 0xFF, w & 0xFF);
|
|
1177
|
+
}
|
|
1178
|
+
|
|
1179
|
+
writeWord(0xFFD8);
|
|
1180
|
+
writeWord(0xFFE0);
|
|
1181
|
+
writeWord(16);
|
|
1182
|
+
byteStream.push(0x4A, 0x46, 0x49, 0x46, 0x00, 1, 1, 0, 0, 1, 0, 1, 0, 0);
|
|
1183
|
+
|
|
1184
|
+
writeWord(0xFFDB);
|
|
1185
|
+
writeWord(132);
|
|
1186
|
+
byteStream.push(0x00);
|
|
1187
|
+
for (let i = 0; i < 64; i++) byteStream.push(yTable[ZIGZAG[i]]);
|
|
1188
|
+
byteStream.push(0x01);
|
|
1189
|
+
for (let i = 0; i < 64; i++) byteStream.push(uvTable[ZIGZAG[i]]);
|
|
1190
|
+
|
|
1191
|
+
writeWord(0xFFC0);
|
|
1192
|
+
writeWord(17);
|
|
1193
|
+
byteStream.push(8);
|
|
1194
|
+
writeWord(height);
|
|
1195
|
+
writeWord(width);
|
|
1196
|
+
byteStream.push(3, 1, 0x11, 0, 2, 0x11, 1, 3, 0x11, 1);
|
|
1197
|
+
|
|
1198
|
+
function writeDHT(nr, val, cls, id) {
|
|
1199
|
+
writeWord(0xFFC4);
|
|
1200
|
+
writeWord(2 + 1 + 16 + val.length);
|
|
1201
|
+
byteStream.push((cls << 4) | id);
|
|
1202
|
+
for (let i = 1; i <= 16; i++) byteStream.push(nr[i]);
|
|
1203
|
+
for (let i = 0; i < val.length; i++) byteStream.push(val[i]);
|
|
1204
|
+
}
|
|
1205
|
+
writeDHT(std_dc_lum_nr, std_dc_lum_val, 0, 0);
|
|
1206
|
+
writeDHT(std_ac_lum_nr, std_ac_lum_val, 1, 0);
|
|
1207
|
+
writeDHT(std_dc_chr_nr, std_dc_chr_val, 0, 1);
|
|
1208
|
+
writeDHT(std_ac_chr_nr, std_ac_chr_val, 1, 1);
|
|
1209
|
+
|
|
1210
|
+
writeWord(0xFFDA);
|
|
1211
|
+
writeWord(12);
|
|
1212
|
+
byteStream.push(3, 1, 0x00, 2, 0x11, 3, 0x11, 0, 63, 0);
|
|
1213
|
+
|
|
1214
|
+
function fdct(block, quant) {
|
|
1215
|
+
const out = new Int32Array(64);
|
|
1216
|
+
const tmp = new Float32Array(64);
|
|
1217
|
+
|
|
1218
|
+
for (let i = 0; i < 8; i++) {
|
|
1219
|
+
const i8 = i * 8;
|
|
1220
|
+
const d0 = block[i8] + block[i8 + 7];
|
|
1221
|
+
const d7 = block[i8] - block[i8 + 7];
|
|
1222
|
+
const d1 = block[i8 + 1] + block[i8 + 6];
|
|
1223
|
+
const d6 = block[i8 + 1] - block[i8 + 6];
|
|
1224
|
+
const d2 = block[i8 + 2] + block[i8 + 5];
|
|
1225
|
+
const d5 = block[i8 + 2] - block[i8 + 5];
|
|
1226
|
+
const d3 = block[i8 + 3] + block[i8 + 4];
|
|
1227
|
+
const d4 = block[i8 + 3] - block[i8 + 4];
|
|
1228
|
+
|
|
1229
|
+
const e0 = d0 + d3;
|
|
1230
|
+
const e3 = d0 - d3;
|
|
1231
|
+
const e1 = d1 + d2;
|
|
1232
|
+
const e2 = d1 - d2;
|
|
1233
|
+
|
|
1234
|
+
tmp[i8] = e0 + e1;
|
|
1235
|
+
tmp[i8 + 4] = e0 - e1;
|
|
1236
|
+
const z1 = (e2 + e3) * 0.707106781;
|
|
1237
|
+
tmp[i8 + 2] = e3 + z1;
|
|
1238
|
+
tmp[i8 + 6] = e3 - z1;
|
|
1239
|
+
|
|
1240
|
+
const f0 = d4 + d5;
|
|
1241
|
+
const f1 = d5 + d6;
|
|
1242
|
+
const f2 = d6 + d7;
|
|
1243
|
+
|
|
1244
|
+
const z2 = (f0 - f2) * 0.382683432;
|
|
1245
|
+
const z3 = f0 * 0.541196100 + z2;
|
|
1246
|
+
const z4 = f2 * 1.306562965 + z2;
|
|
1247
|
+
const z5 = f1 * 0.707106781;
|
|
1248
|
+
|
|
1249
|
+
const g0 = d7 + z5;
|
|
1250
|
+
const g1 = d7 - z5;
|
|
1251
|
+
|
|
1252
|
+
tmp[i8 + 5] = g1 + z3;
|
|
1253
|
+
tmp[i8 + 3] = g1 - z3;
|
|
1254
|
+
tmp[i8 + 1] = g0 + z4;
|
|
1255
|
+
tmp[i8 + 7] = g0 - z4;
|
|
1256
|
+
}
|
|
1257
|
+
|
|
1258
|
+
for (let j = 0; j < 8; j++) {
|
|
1259
|
+
const d0 = tmp[j] + tmp[56 + j];
|
|
1260
|
+
const d7 = tmp[j] - tmp[56 + j];
|
|
1261
|
+
const d1 = tmp[8 + j] + tmp[48 + j];
|
|
1262
|
+
const d6 = tmp[8 + j] - tmp[48 + j];
|
|
1263
|
+
const d2 = tmp[16 + j] + tmp[40 + j];
|
|
1264
|
+
const d5 = tmp[16 + j] - tmp[40 + j];
|
|
1265
|
+
const d3 = tmp[24 + j] + tmp[32 + j];
|
|
1266
|
+
const d4 = tmp[24 + j] - tmp[32 + j];
|
|
1267
|
+
|
|
1268
|
+
const e0 = d0 + d3;
|
|
1269
|
+
const e3 = d0 - d3;
|
|
1270
|
+
const e1 = d1 + d2;
|
|
1271
|
+
const e2 = d1 - d2;
|
|
1272
|
+
|
|
1273
|
+
out[j] = Math.round((e0 + e1) * quant[j]);
|
|
1274
|
+
out[32 + j] = Math.round((e0 - e1) * quant[32 + j]);
|
|
1275
|
+
const z1 = (e2 + e3) * 0.707106781;
|
|
1276
|
+
out[16 + j] = Math.round((e3 + z1) * quant[16 + j]);
|
|
1277
|
+
out[48 + j] = Math.round((e3 - z1) * quant[48 + j]);
|
|
1278
|
+
|
|
1279
|
+
const f0 = d4 + d5;
|
|
1280
|
+
const f1 = d5 + d6;
|
|
1281
|
+
const f2 = d6 + d7;
|
|
1282
|
+
|
|
1283
|
+
const z2 = (f0 - f2) * 0.382683432;
|
|
1284
|
+
const z3 = f0 * 0.541196100 + z2;
|
|
1285
|
+
const z4 = f2 * 1.306562965 + z2;
|
|
1286
|
+
const z5 = f1 * 0.707106781;
|
|
1287
|
+
|
|
1288
|
+
const g0 = d7 + z5;
|
|
1289
|
+
const g1 = d7 - z5;
|
|
1290
|
+
|
|
1291
|
+
out[40 + j] = Math.round((g1 + z3) * quant[40 + j]);
|
|
1292
|
+
out[24 + j] = Math.round((g1 - z3) * quant[24 + j]);
|
|
1293
|
+
out[8 + j] = Math.round((g0 + z4) * quant[8 + j]);
|
|
1294
|
+
out[56 + j] = Math.round((g0 - z4) * quant[56 + j]);
|
|
1295
|
+
}
|
|
1296
|
+
|
|
1297
|
+
return out;
|
|
1298
|
+
}
|
|
1299
|
+
|
|
1300
|
+
function encodeCategory(val) {
|
|
1301
|
+
if (val === 0) return { bits: 0, len: 0 };
|
|
1302
|
+
const absVal = Math.abs(val);
|
|
1303
|
+
let len = 0;
|
|
1304
|
+
while (absVal >= (1 << len)) len++;
|
|
1305
|
+
const bits = val < 0 ? (val + (1 << len) - 1) : val;
|
|
1306
|
+
return { bits, len };
|
|
1307
|
+
}
|
|
1308
|
+
|
|
1309
|
+
function encodeBlock(block, quant, dcHuff, acHuff, prevDC) {
|
|
1310
|
+
const dct = fdct(block, quant);
|
|
1311
|
+
const dcDiff = dct[0] - prevDC;
|
|
1312
|
+
const dcCat = encodeCategory(dcDiff);
|
|
1313
|
+
const dcH = dcHuff[dcCat.len];
|
|
1314
|
+
writeBits(dcH.code, dcH.len);
|
|
1315
|
+
if (dcCat.len > 0) writeBits(dcCat.bits, dcCat.len);
|
|
1316
|
+
|
|
1317
|
+
let r = 0;
|
|
1318
|
+
for (let k = 1; k < 64; k++) {
|
|
1319
|
+
const val = dct[ZIGZAG[k]];
|
|
1320
|
+
if (val === 0) {
|
|
1321
|
+
r++;
|
|
1322
|
+
} else {
|
|
1323
|
+
while (r > 15) {
|
|
1324
|
+
const zrl = acHuff[0xF0];
|
|
1325
|
+
writeBits(zrl.code, zrl.len);
|
|
1326
|
+
r -= 16;
|
|
1327
|
+
}
|
|
1328
|
+
const acCat = encodeCategory(val);
|
|
1329
|
+
const acSym = (r << 4) | acCat.len;
|
|
1330
|
+
const acH = acHuff[acSym];
|
|
1331
|
+
writeBits(acH.code, acH.len);
|
|
1332
|
+
writeBits(acCat.bits, acCat.len);
|
|
1333
|
+
r = 0;
|
|
1334
|
+
}
|
|
1335
|
+
}
|
|
1336
|
+
if (r > 0) {
|
|
1337
|
+
const eob = acHuff[0x00];
|
|
1338
|
+
writeBits(eob.code, eob.len);
|
|
1339
|
+
}
|
|
1340
|
+
return dct[0];
|
|
1341
|
+
}
|
|
1342
|
+
|
|
1343
|
+
const mcusX = Math.ceil(width / 8);
|
|
1344
|
+
const mcusY = Math.ceil(height / 8);
|
|
1345
|
+
const yData = new Float32Array(mcusX * 8 * mcusY * 8);
|
|
1346
|
+
const cbData = new Float32Array(mcusX * 8 * mcusY * 8);
|
|
1347
|
+
const crData = new Float32Array(mcusX * 8 * mcusY * 8);
|
|
1348
|
+
|
|
1349
|
+
for (let y = 0; y < height; y++) {
|
|
1350
|
+
for (let x = 0; x < width; x++) {
|
|
1351
|
+
const sIdx = (y * width + x) * 3;
|
|
1352
|
+
const r = rgb[sIdx];
|
|
1353
|
+
const g = rgb[sIdx + 1];
|
|
1354
|
+
const b = rgb[sIdx + 2];
|
|
1355
|
+
const dIdx = y * (mcusX * 8) + x;
|
|
1356
|
+
yData[dIdx] = (0.299 * r + 0.587 * g + 0.114 * b) - 128;
|
|
1357
|
+
cbData[dIdx] = (-0.168736 * r - 0.331264 * g + 0.5 * b);
|
|
1358
|
+
crData[dIdx] = (0.5 * r - 0.418688 * g - 0.081312 * b);
|
|
1359
|
+
}
|
|
1360
|
+
}
|
|
1361
|
+
|
|
1362
|
+
let prevYDC = 0, prevCbDC = 0, prevCrDC = 0;
|
|
1363
|
+
const yBlock = new Float32Array(64);
|
|
1364
|
+
const cbBlock = new Float32Array(64);
|
|
1365
|
+
const crBlock = new Float32Array(64);
|
|
1366
|
+
|
|
1367
|
+
for (let mY = 0; mY < mcusY; mY++) {
|
|
1368
|
+
for (let mX = 0; mX < mcusX; mX++) {
|
|
1369
|
+
for (let r = 0; r < 8; r++) {
|
|
1370
|
+
for (let c = 0; c < 8; c++) {
|
|
1371
|
+
const sX = Math.min(width - 1, mX * 8 + c);
|
|
1372
|
+
const sY = Math.min(height - 1, mY * 8 + r);
|
|
1373
|
+
const idx = sY * (mcusX * 8) + sX;
|
|
1374
|
+
const bIdx = r * 8 + c;
|
|
1375
|
+
yBlock[bIdx] = yData[idx];
|
|
1376
|
+
cbBlock[bIdx] = cbData[idx];
|
|
1377
|
+
crBlock[bIdx] = crData[idx];
|
|
1378
|
+
}
|
|
1379
|
+
}
|
|
1380
|
+
prevYDC = encodeBlock(yBlock, yQuant, dcLumHuff, acLumHuff, prevYDC);
|
|
1381
|
+
prevCbDC = encodeBlock(cbBlock, uvQuant, dcChrHuff, acChrHuff, prevCbDC);
|
|
1382
|
+
prevCrDC = encodeBlock(crBlock, uvQuant, dcChrHuff, acChrHuff, prevCrDC);
|
|
1383
|
+
}
|
|
1384
|
+
}
|
|
1385
|
+
|
|
1386
|
+
flushBits();
|
|
1387
|
+
writeWord(0xFFD9);
|
|
1388
|
+
|
|
1389
|
+
const uint8 = new Uint8Array(byteStream);
|
|
1390
|
+
return `data:image/jpeg;base64,${bytesToBase64(uint8)}`;
|
|
1391
|
+
} catch (e) {
|
|
1392
|
+
return null;
|
|
1393
|
+
}
|
|
1394
|
+
}
|
|
1395
|
+
|
|
1396
|
+
const PDF_PASSWORD_PADDING = new Uint8Array([
|
|
1397
|
+
0x28, 0xBF, 0x4E, 0x5E, 0x4E, 0x75, 0x8A, 0x41,
|
|
1398
|
+
0x64, 0x00, 0x4E, 0x56, 0xFF, 0xFA, 0x01, 0x08,
|
|
1399
|
+
0x2E, 0x2E, 0x00, 0xB6, 0xD0, 0x68, 0x3E, 0x80,
|
|
1400
|
+
0x2F, 0x0C, 0xA9, 0xFE, 0x64, 0x53, 0x69, 0x7A
|
|
1401
|
+
]);
|
|
1402
|
+
|
|
1403
|
+
function concatByteArrays(...arrays) {
|
|
1404
|
+
const result = new Uint8Array(arrays.reduce((total, value) => total + value.length, 0));
|
|
1405
|
+
let offset = 0;
|
|
1406
|
+
for (const value of arrays) {
|
|
1407
|
+
result.set(value, offset);
|
|
1408
|
+
offset += value.length;
|
|
1409
|
+
}
|
|
1410
|
+
return result;
|
|
1411
|
+
}
|
|
1412
|
+
|
|
1413
|
+
function hexToBytes(hex) {
|
|
1414
|
+
const clean = String(hex || '').replace(/\s+/g, '');
|
|
1415
|
+
if (!clean || clean.length % 2 !== 0 || !/^[0-9a-f]+$/i.test(clean)) return null;
|
|
1416
|
+
const bytes = new Uint8Array(clean.length / 2);
|
|
1417
|
+
for (let index = 0; index < bytes.length; index++) bytes[index] = parseInt(clean.slice(index * 2, index * 2 + 2), 16);
|
|
1418
|
+
return bytes;
|
|
1419
|
+
}
|
|
1420
|
+
|
|
1421
|
+
function md5Bytes(input) {
|
|
1422
|
+
const source = input instanceof Uint8Array ? input : new Uint8Array(input || []);
|
|
1423
|
+
const paddedLength = Math.ceil((source.length + 9) / 64) * 64;
|
|
1424
|
+
const padded = new Uint8Array(paddedLength);
|
|
1425
|
+
padded.set(source);
|
|
1426
|
+
padded[source.length] = 0x80;
|
|
1427
|
+
const view = new DataView(padded.buffer);
|
|
1428
|
+
const bitLength = source.length * 8;
|
|
1429
|
+
view.setUint32(paddedLength - 8, bitLength >>> 0, true);
|
|
1430
|
+
view.setUint32(paddedLength - 4, Math.floor(bitLength / 0x100000000), true);
|
|
1431
|
+
|
|
1432
|
+
const shifts = [
|
|
1433
|
+
7, 12, 17, 22, 7, 12, 17, 22, 7, 12, 17, 22, 7, 12, 17, 22,
|
|
1434
|
+
5, 9, 14, 20, 5, 9, 14, 20, 5, 9, 14, 20, 5, 9, 14, 20,
|
|
1435
|
+
4, 11, 16, 23, 4, 11, 16, 23, 4, 11, 16, 23, 4, 11, 16, 23,
|
|
1436
|
+
6, 10, 15, 21, 6, 10, 15, 21, 6, 10, 15, 21, 6, 10, 15, 21
|
|
1437
|
+
];
|
|
1438
|
+
const constants = Array.from({ length: 64 }, (_, index) => Math.floor(Math.abs(Math.sin(index + 1)) * 0x100000000) >>> 0);
|
|
1439
|
+
let a0 = 0x67452301;
|
|
1440
|
+
let b0 = 0xEFCDAB89;
|
|
1441
|
+
let c0 = 0x98BADCFE;
|
|
1442
|
+
let d0 = 0x10325476;
|
|
1443
|
+
|
|
1444
|
+
for (let offset = 0; offset < paddedLength; offset += 64) {
|
|
1445
|
+
const words = Array.from({ length: 16 }, (_, index) => view.getUint32(offset + index * 4, true));
|
|
1446
|
+
let a = a0, b = b0, c = c0, d = d0;
|
|
1447
|
+
for (let index = 0; index < 64; index++) {
|
|
1448
|
+
let f, wordIndex;
|
|
1449
|
+
if (index < 16) {
|
|
1450
|
+
f = (b & c) | (~b & d);
|
|
1451
|
+
wordIndex = index;
|
|
1452
|
+
} else if (index < 32) {
|
|
1453
|
+
f = (d & b) | (~d & c);
|
|
1454
|
+
wordIndex = (5 * index + 1) % 16;
|
|
1455
|
+
} else if (index < 48) {
|
|
1456
|
+
f = b ^ c ^ d;
|
|
1457
|
+
wordIndex = (3 * index + 5) % 16;
|
|
1458
|
+
} else {
|
|
1459
|
+
f = c ^ (b | ~d);
|
|
1460
|
+
wordIndex = (7 * index) % 16;
|
|
1461
|
+
}
|
|
1462
|
+
const sum = (a + f + constants[index] + words[wordIndex]) >>> 0;
|
|
1463
|
+
const rotated = ((sum << shifts[index]) | (sum >>> (32 - shifts[index]))) >>> 0;
|
|
1464
|
+
const previousD = d;
|
|
1465
|
+
d = c;
|
|
1466
|
+
c = b;
|
|
1467
|
+
b = (b + rotated) >>> 0;
|
|
1468
|
+
a = previousD;
|
|
1469
|
+
}
|
|
1470
|
+
a0 = (a0 + a) >>> 0;
|
|
1471
|
+
b0 = (b0 + b) >>> 0;
|
|
1472
|
+
c0 = (c0 + c) >>> 0;
|
|
1473
|
+
d0 = (d0 + d) >>> 0;
|
|
1474
|
+
}
|
|
1475
|
+
|
|
1476
|
+
const digest = new Uint8Array(16);
|
|
1477
|
+
const digestView = new DataView(digest.buffer);
|
|
1478
|
+
digestView.setUint32(0, a0, true);
|
|
1479
|
+
digestView.setUint32(4, b0, true);
|
|
1480
|
+
digestView.setUint32(8, c0, true);
|
|
1481
|
+
digestView.setUint32(12, d0, true);
|
|
1482
|
+
return digest;
|
|
1483
|
+
}
|
|
1484
|
+
|
|
1485
|
+
function rc4Bytes(key, input) {
|
|
1486
|
+
const state = new Uint8Array(256);
|
|
1487
|
+
for (let index = 0; index < 256; index++) state[index] = index;
|
|
1488
|
+
let j = 0;
|
|
1489
|
+
for (let index = 0; index < 256; index++) {
|
|
1490
|
+
j = (j + state[index] + key[index % key.length]) & 0xFF;
|
|
1491
|
+
const swap = state[index]; state[index] = state[j]; state[j] = swap;
|
|
1492
|
+
}
|
|
1493
|
+
const output = new Uint8Array(input.length);
|
|
1494
|
+
let i = 0;
|
|
1495
|
+
j = 0;
|
|
1496
|
+
for (let index = 0; index < input.length; index++) {
|
|
1497
|
+
i = (i + 1) & 0xFF;
|
|
1498
|
+
j = (j + state[i]) & 0xFF;
|
|
1499
|
+
const swap = state[i]; state[i] = state[j]; state[j] = swap;
|
|
1500
|
+
output[index] = input[index] ^ state[(state[i] + state[j]) & 0xFF];
|
|
1501
|
+
}
|
|
1502
|
+
return output;
|
|
1503
|
+
}
|
|
1504
|
+
|
|
1505
|
+
function bytesEqual(left, right, length = Math.min(left?.length || 0, right?.length || 0)) {
|
|
1506
|
+
if (!left || !right || left.length < length || right.length < length) return false;
|
|
1507
|
+
for (let index = 0; index < length; index++) if (left[index] !== right[index]) return false;
|
|
1508
|
+
return true;
|
|
1509
|
+
}
|
|
1510
|
+
|
|
1511
|
+
function createPdfSecurityContext(allObjects, fullText) {
|
|
1512
|
+
const encryptRef = String(fullText || '').match(/\/Encrypt\s+(\d+)\s+(\d+)\s+R/i);
|
|
1513
|
+
if (!encryptRef) return null;
|
|
1514
|
+
const encryptionBody = allObjects.get(encryptRef[1]);
|
|
1515
|
+
if (!encryptionBody) return { encrypted: true, supported: false };
|
|
1516
|
+
const filter = encryptionBody.match(/\/Filter\s*\/([A-Za-z0-9]+)/)?.[1] || '';
|
|
1517
|
+
const version = Number(encryptionBody.match(/\/V\s+(\d+)/)?.[1] || 0);
|
|
1518
|
+
const revision = Number(encryptionBody.match(/\/R\s+(\d+)/)?.[1] || 0);
|
|
1519
|
+
const ownerKey = hexToBytes(encryptionBody.match(/\/O\s*<([0-9A-Fa-f\s]+)>/)?.[1]);
|
|
1520
|
+
const userKey = hexToBytes(encryptionBody.match(/\/U\s*<([0-9A-Fa-f\s]+)>/)?.[1]);
|
|
1521
|
+
const fileId = hexToBytes(String(fullText || '').match(/\/ID\s*\[\s*<([0-9A-Fa-f\s]+)>/)?.[1]);
|
|
1522
|
+
const permissions = Number(encryptionBody.match(/\/P\s+(-?\d+)/)?.[1]);
|
|
1523
|
+
if (filter !== 'Standard' || ![1, 2].includes(version) || ![2, 3].includes(revision) || !ownerKey || !userKey || !fileId || !Number.isFinite(permissions)) {
|
|
1524
|
+
return { encrypted: true, supported: false };
|
|
1525
|
+
}
|
|
1526
|
+
|
|
1527
|
+
const keyLength = revision === 2 ? 5 : Math.min(16, Math.max(5, Number(encryptionBody.match(/\/Length\s+(\d+)/)?.[1] || 40) / 8));
|
|
1528
|
+
const permissionBytes = new Uint8Array(4);
|
|
1529
|
+
new DataView(permissionBytes.buffer).setInt32(0, permissions, true);
|
|
1530
|
+
let digest = md5Bytes(concatByteArrays(PDF_PASSWORD_PADDING, ownerKey, permissionBytes, fileId));
|
|
1531
|
+
if (revision >= 3) {
|
|
1532
|
+
for (let round = 0; round < 50; round++) digest = md5Bytes(digest.slice(0, keyLength));
|
|
1533
|
+
}
|
|
1534
|
+
const fileKey = digest.slice(0, keyLength);
|
|
1535
|
+
let expectedUserKey;
|
|
1536
|
+
if (revision === 2) {
|
|
1537
|
+
expectedUserKey = rc4Bytes(fileKey, PDF_PASSWORD_PADDING);
|
|
1538
|
+
} else {
|
|
1539
|
+
expectedUserKey = md5Bytes(concatByteArrays(PDF_PASSWORD_PADDING, fileId));
|
|
1540
|
+
expectedUserKey = rc4Bytes(fileKey, expectedUserKey);
|
|
1541
|
+
for (let round = 1; round <= 19; round++) {
|
|
1542
|
+
const roundKey = fileKey.map(value => value ^ round);
|
|
1543
|
+
expectedUserKey = rc4Bytes(roundKey, expectedUserKey);
|
|
1544
|
+
}
|
|
1545
|
+
}
|
|
1546
|
+
const valid = bytesEqual(expectedUserKey, userKey, revision === 2 ? 32 : 16);
|
|
1547
|
+
return { encrypted: true, supported: valid, fileKey };
|
|
1548
|
+
}
|
|
1549
|
+
|
|
1550
|
+
function decryptPdfObjectBytes(input, security, objectNumber, generationNumber = 0) {
|
|
1551
|
+
if (!security?.supported || !security.fileKey) return null;
|
|
1552
|
+
const suffix = new Uint8Array([
|
|
1553
|
+
objectNumber & 0xFF, (objectNumber >>> 8) & 0xFF, (objectNumber >>> 16) & 0xFF,
|
|
1554
|
+
generationNumber & 0xFF, (generationNumber >>> 8) & 0xFF
|
|
1555
|
+
]);
|
|
1556
|
+
const digest = md5Bytes(concatByteArrays(security.fileKey, suffix));
|
|
1557
|
+
const objectKey = digest.slice(0, Math.min(security.fileKey.length + 5, 16));
|
|
1558
|
+
return rc4Bytes(objectKey, input);
|
|
1559
|
+
}
|
|
1560
|
+
|
|
1561
|
+
function extractImagesFromPdfObjects(allObjects, objOffsets, bytes, security = null) {
|
|
1562
|
+
const imagesByObjNum = new Map();
|
|
1563
|
+
let imgCounter = 1;
|
|
1564
|
+
|
|
1565
|
+
for (const [num, body] of allObjects.entries()) {
|
|
1566
|
+
if (!/\/Subtype\s*\/Image\b/i.test(body)) continue;
|
|
1567
|
+
|
|
1568
|
+
const widthMatch = body.match(/\/Width\s+(\d+)/);
|
|
1569
|
+
const heightMatch = body.match(/\/Height\s+(\d+)/);
|
|
1570
|
+
const width = widthMatch ? parseInt(widthMatch[1], 10) : 0;
|
|
1571
|
+
const height = heightMatch ? parseInt(heightMatch[1], 10) : 0;
|
|
1572
|
+
|
|
1573
|
+
// Filtrar elementos gráficos irrelevantes o viñetas diminutas
|
|
1574
|
+
if (width < 30 || height < 30 || (width * height < 2500)) continue;
|
|
1575
|
+
|
|
1576
|
+
const sIdx = body.indexOf('stream');
|
|
1577
|
+
const eIdx = body.indexOf('endstream', sIdx);
|
|
1578
|
+
if (sIdx === -1 || eIdx === -1) continue;
|
|
1579
|
+
|
|
1580
|
+
let dStart = sIdx + 6;
|
|
1581
|
+
if (body.charCodeAt(dStart) === 13) dStart++;
|
|
1582
|
+
if (body.charCodeAt(dStart) === 10) dStart++;
|
|
1583
|
+
let dEnd = eIdx;
|
|
1584
|
+
while (dEnd > dStart && (body.charCodeAt(dEnd - 1) === 10 || body.charCodeAt(dEnd - 1) === 13 || body.charCodeAt(dEnd - 1) === 32)) {
|
|
1585
|
+
dEnd--;
|
|
1586
|
+
}
|
|
1587
|
+
|
|
1588
|
+
const offset = objOffsets.get(String(num)) || 0;
|
|
1589
|
+
const rawBytes = bytes.subarray(offset + dStart, offset + dEnd);
|
|
1590
|
+
if (!rawBytes || rawBytes.length < 50) continue;
|
|
1591
|
+
|
|
1592
|
+
const generation = Number(body.match(/^\s*\d+\s+(\d+)\s+obj/)?.[1] || 0);
|
|
1593
|
+
const imageBytes = security?.encrypted
|
|
1594
|
+
? decryptPdfObjectBytes(rawBytes, security, Number(num), generation)
|
|
1595
|
+
: rawBytes;
|
|
1596
|
+
if (!imageBytes || imageBytes.length < 50) continue;
|
|
1597
|
+
|
|
1598
|
+
const isDct = /\/Filter\s*(?:\/DCTDecode|\[\s*\/DCTDecode\s*\])/i.test(body);
|
|
1599
|
+
const isJpx = /\/Filter\s*(?:\/JPXDecode|\[\s*\/JPXDecode\s*\])/i.test(body);
|
|
1600
|
+
|
|
1601
|
+
let mimeType = '';
|
|
1602
|
+
let dataUrl = '';
|
|
1603
|
+
|
|
1604
|
+
const isCmyk = body.includes('DeviceCMYK') || body.includes('/ColorSpace/DeviceCMYK') || body.includes('/ColorSpace /DeviceCMYK');
|
|
1605
|
+
|
|
1606
|
+
if ((isDct || (imageBytes[0] === 0xFF && imageBytes[1] === 0xD8)) && imageBytes[0] === 0xFF && imageBytes[1] === 0xD8) {
|
|
1607
|
+
mimeType = 'image/jpeg';
|
|
1608
|
+
dataUrl = `data:image/jpeg;base64,${bytesToBase64(imageBytes)}`;
|
|
1609
|
+
} else if (isJpx && imageBytes.length >= 12 && imageBytes[4] === 0x6A && imageBytes[5] === 0x50) {
|
|
1610
|
+
mimeType = 'image/jp2';
|
|
1611
|
+
dataUrl = `data:image/jp2;base64,${bytesToBase64(imageBytes)}`;
|
|
1612
|
+
} else if (imageBytes[0] === 0x89 && imageBytes[1] === 0x50 && imageBytes[2] === 0x4E && imageBytes[3] === 0x47) {
|
|
1613
|
+
mimeType = 'image/png';
|
|
1614
|
+
dataUrl = `data:image/png;base64,${bytesToBase64(imageBytes)}`;
|
|
1615
|
+
}
|
|
1616
|
+
|
|
1617
|
+
if (dataUrl) {
|
|
1618
|
+
imagesByObjNum.set(String(num), {
|
|
1619
|
+
id: `img_${imgCounter++}`,
|
|
1620
|
+
objNum: String(num),
|
|
1621
|
+
width,
|
|
1622
|
+
height,
|
|
1623
|
+
sizeBytes: imageBytes.length,
|
|
1624
|
+
mimeType: mimeType || 'image/jpeg',
|
|
1625
|
+
isCmyk: Boolean(isCmyk),
|
|
1626
|
+
dataUrl
|
|
1627
|
+
});
|
|
1628
|
+
}
|
|
1629
|
+
}
|
|
1630
|
+
|
|
1631
|
+
return imagesByObjNum;
|
|
1632
|
+
}
|
|
1633
|
+
|
|
1634
|
+
/** Extrae texto y objetos de imagen de un PDF de forma interna. */
|
|
1635
|
+
async function extractPdfInternal(arrayBuffer) {
|
|
1636
|
+
const bytes = new Uint8Array(arrayBuffer);
|
|
1637
|
+
const decoder = new TextDecoder('latin1');
|
|
1638
|
+
const fullText = decoder.decode(bytes);
|
|
1639
|
+
|
|
1640
|
+
// 1. Mapeo de objetos directos PDF
|
|
1641
|
+
const allObjects = new Map();
|
|
1642
|
+
const objOffsets = new Map();
|
|
1643
|
+
const objRegex = /(\d+)\s+(\d+)\s+obj/g;
|
|
1644
|
+
let m;
|
|
1645
|
+
while ((m = objRegex.exec(fullText)) !== null) {
|
|
1646
|
+
objOffsets.set(m[1], m.index);
|
|
1647
|
+
}
|
|
1648
|
+
|
|
1649
|
+
for (const [num, offset] of objOffsets.entries()) {
|
|
1650
|
+
const endObj = fullText.indexOf('endobj', offset);
|
|
1651
|
+
if (endObj !== -1) {
|
|
1652
|
+
allObjects.set(num, fullText.substring(offset, endObj + 6));
|
|
1653
|
+
}
|
|
1654
|
+
}
|
|
1655
|
+
|
|
1656
|
+
// 2. Descomprimir todos los flujos de objetos comprimidos (/Type /ObjStm) antes de extraer CMaps
|
|
1657
|
+
for (const [num, body] of Array.from(allObjects.entries())) {
|
|
1658
|
+
if (body.includes('/Type/ObjStm') || body.includes('/Type /ObjStm')) {
|
|
1659
|
+
const sIdx = body.indexOf('stream');
|
|
1660
|
+
const eIdx = body.indexOf('endstream', sIdx);
|
|
1661
|
+
if (sIdx !== -1 && eIdx !== -1) {
|
|
1662
|
+
let dStart = sIdx + 6;
|
|
1663
|
+
if (body.charCodeAt(dStart) === 13) dStart++;
|
|
1664
|
+
if (body.charCodeAt(dStart) === 10) dStart++;
|
|
1665
|
+
let rEnd = eIdx;
|
|
1666
|
+
if (rEnd > dStart && (body.charCodeAt(rEnd - 1) === 10 || body.charCodeAt(rEnd - 1) === 13)) rEnd--;
|
|
1667
|
+
if (rEnd > dStart && (body.charCodeAt(rEnd - 1) === 10 || body.charCodeAt(rEnd - 1) === 13)) rEnd--;
|
|
1668
|
+
|
|
1669
|
+
const offset = objOffsets.get(num) || 0;
|
|
1670
|
+
const rawStream = bytes.subarray(offset + dStart, offset + rEnd);
|
|
1671
|
+
try {
|
|
1672
|
+
const decomp = await decompressDeflateData(rawStream);
|
|
1673
|
+
if (decomp) {
|
|
1674
|
+
const decStr = decoder.decode(decomp);
|
|
1675
|
+
const nMatch = body.match(/\/N\s+(\d+)/);
|
|
1676
|
+
const firstMatch = body.match(/\/First\s+(\d+)/);
|
|
1677
|
+
const n = nMatch ? parseInt(nMatch[1]) : 0;
|
|
1678
|
+
const first = firstMatch ? parseInt(firstMatch[1]) : 0;
|
|
1679
|
+
const header = decStr.substring(0, first).trim().split(/\s+/);
|
|
1680
|
+
for (let i = 0; i < header.length; i += 2) {
|
|
1681
|
+
const oNum = header[i];
|
|
1682
|
+
const oOffset = parseInt(header[i + 1]);
|
|
1683
|
+
const nextOffset = (i + 3 < header.length) ? parseInt(header[i + 3]) : decStr.length - first;
|
|
1684
|
+
const oBody = decStr.substring(first + oOffset, first + nextOffset);
|
|
1685
|
+
allObjects.set(oNum, oBody);
|
|
1686
|
+
}
|
|
1687
|
+
}
|
|
1688
|
+
} catch (e) {}
|
|
1689
|
+
}
|
|
1690
|
+
}
|
|
1691
|
+
}
|
|
1692
|
+
|
|
1693
|
+
// 3. Extracción de contexto de seguridad y CMaps / ToUnicode
|
|
1694
|
+
const security = createPdfSecurityContext(allObjects, fullText);
|
|
1695
|
+
const cmap = await parseCMaps(allObjects, fullText, bytes, objOffsets, security);
|
|
1696
|
+
|
|
1697
|
+
// 3b. Extracción de imágenes XObject (/Subtype /Image)
|
|
1698
|
+
const imagesByObjNum = extractImagesFromPdfObjects(allObjects, objOffsets, bytes, security);
|
|
1699
|
+
const assignedImages = new Set();
|
|
1700
|
+
const allExtractedImages = [];
|
|
1701
|
+
|
|
1702
|
+
// 4. Resolución del catálogo y árbol jerárquico de páginas (Page Tree)
|
|
1703
|
+
let catalogObjNum = null;
|
|
1704
|
+
for (const [num, body] of allObjects.entries()) {
|
|
1705
|
+
if (/\/Type\s*\/Catalog\b/.test(body)) {
|
|
1706
|
+
catalogObjNum = num;
|
|
1707
|
+
break;
|
|
1708
|
+
}
|
|
1709
|
+
}
|
|
1710
|
+
|
|
1711
|
+
const catalogBody = catalogObjNum ? (allObjects.get(catalogObjNum) || '') : '';
|
|
1712
|
+
const pagesMatch = catalogBody.match(/\/Pages\s+(\d+)\s+\d+\s+R/);
|
|
1713
|
+
|
|
1714
|
+
const pagesList = [];
|
|
1715
|
+
function traverse(nodeObjNum) {
|
|
1716
|
+
const body = allObjects.get(String(nodeObjNum));
|
|
1717
|
+
if (!body) return;
|
|
1718
|
+
if (/\/Type\s*\/Page\b/i.test(body) && !/\/Type\s*\/Pages\b/i.test(body)) {
|
|
1719
|
+
pagesList.push(nodeObjNum);
|
|
1720
|
+
return;
|
|
1721
|
+
}
|
|
1722
|
+
const kidsMatch = body.match(/\/Kids\s*\[([\s\S]*?)\]/);
|
|
1723
|
+
if (kidsMatch) {
|
|
1724
|
+
const refs = kidsMatch[1].match(/(\d+)\s+\d+\s+R/g) || [];
|
|
1725
|
+
for (const ref of refs) {
|
|
1726
|
+
const kidNum = ref.match(/^(\d+)/)[1];
|
|
1727
|
+
traverse(kidNum);
|
|
1728
|
+
}
|
|
1729
|
+
}
|
|
1730
|
+
}
|
|
1731
|
+
if (pagesMatch) traverse(pagesMatch[1]);
|
|
1732
|
+
|
|
1733
|
+
let pages = [];
|
|
1734
|
+
let pageNum = 1;
|
|
1735
|
+
|
|
1736
|
+
// A. Extracción secuencial basada en el Page Tree
|
|
1737
|
+
if (pagesList.length > 0) {
|
|
1738
|
+
for (const pageObjNum of pagesList) {
|
|
1739
|
+
const body = allObjects.get(String(pageObjNum));
|
|
1740
|
+
if (!body) continue;
|
|
1741
|
+
|
|
1742
|
+
const contentsMatch = body.match(/\/Contents\s+(?:\[([\s\S]*?)\]|(\d+)\s+\d+\s+R)/);
|
|
1743
|
+
let contentObjs = [];
|
|
1744
|
+
if (contentsMatch) {
|
|
1745
|
+
if (contentsMatch[1]) {
|
|
1746
|
+
contentObjs = (contentsMatch[1].match(/(\d+)\s+\d+\s+R/g) || []).map(r => r.match(/^(\d+)/)[1]);
|
|
1747
|
+
} else if (contentsMatch[2]) {
|
|
1748
|
+
contentObjs = [contentsMatch[2]];
|
|
1749
|
+
}
|
|
1750
|
+
}
|
|
1751
|
+
|
|
1752
|
+
let pageItems = [];
|
|
1753
|
+
|
|
1754
|
+
// Extraer contenido de streams de texto de la página
|
|
1755
|
+
for (const cNum of contentObjs) {
|
|
1756
|
+
const cBody = allObjects.get(String(cNum));
|
|
1757
|
+
if (!cBody) continue;
|
|
1758
|
+
const streamIdx = cBody.indexOf('stream');
|
|
1759
|
+
const endStreamIdx = cBody.indexOf('endstream', streamIdx);
|
|
1760
|
+
if (streamIdx !== -1 && endStreamIdx !== -1) {
|
|
1761
|
+
let dataStart = streamIdx + 6;
|
|
1762
|
+
if (cBody.charCodeAt(dataStart) === 13) dataStart++;
|
|
1763
|
+
if (cBody.charCodeAt(dataStart) === 10) dataStart++;
|
|
1764
|
+
let rawEnd = endStreamIdx;
|
|
1765
|
+
if (rawEnd > dataStart && (cBody.charCodeAt(rawEnd - 1) === 10 || cBody.charCodeAt(rawEnd - 1) === 13)) rawEnd--;
|
|
1766
|
+
if (rawEnd > dataStart && (cBody.charCodeAt(rawEnd - 1) === 10 || cBody.charCodeAt(rawEnd - 1) === 13)) rawEnd--;
|
|
1767
|
+
|
|
1768
|
+
const offset = objOffsets.get(String(cNum));
|
|
1769
|
+
const rawBytes = offset !== undefined
|
|
1770
|
+
? bytes.subarray(offset + dataStart, offset + rawEnd)
|
|
1771
|
+
: new Uint8Array(Array.from(cBody.substring(dataStart, rawEnd), ch => ch.charCodeAt(0)));
|
|
1772
|
+
|
|
1773
|
+
try {
|
|
1774
|
+
let streamString = '';
|
|
1775
|
+
const generation = Number(cBody.match(/^\s*\d+\s+(\d+)\s+obj/)?.[1] || 0);
|
|
1776
|
+
const streamBytes = security?.encrypted
|
|
1777
|
+
? decryptPdfObjectBytes(rawBytes, security, Number(cNum), generation)
|
|
1778
|
+
: rawBytes;
|
|
1779
|
+
const decompressed = await decompressDeflateData(streamBytes);
|
|
1780
|
+
if (decompressed) {
|
|
1781
|
+
streamString = decoder.decode(decompressed);
|
|
1782
|
+
} else {
|
|
1783
|
+
streamString = decoder.decode(streamBytes);
|
|
1784
|
+
}
|
|
1785
|
+
|
|
1786
|
+
if (streamString) {
|
|
1787
|
+
const parsed = parsePdfStreamText(streamString, cmap);
|
|
1788
|
+
if (parsed && parsed.length > 0) {
|
|
1789
|
+
pageItems.push(parsed);
|
|
1790
|
+
}
|
|
1791
|
+
}
|
|
1792
|
+
} catch (e) {}
|
|
1793
|
+
}
|
|
1794
|
+
}
|
|
1795
|
+
|
|
1796
|
+
// Detectar recursos /XObject directos e indirectos de la página
|
|
1797
|
+
const resMatch = body.match(/\/Resources\s*(?:<<([\s\S]*?)>>|(\d+)\s+\d+\s+R)/);
|
|
1798
|
+
let xobjDict = '';
|
|
1799
|
+
if (resMatch) {
|
|
1800
|
+
if (resMatch[1]) {
|
|
1801
|
+
const xMatch = resMatch[1].match(/\/XObject\s*(?:<<([\s\S]*?)>>|(\d+)\s+\d+\s+R)/);
|
|
1802
|
+
if (xMatch) {
|
|
1803
|
+
if (xMatch[1]) xobjDict = xMatch[1];
|
|
1804
|
+
else if (xMatch[2]) xobjDict = allObjects.get(xMatch[2]) || '';
|
|
1805
|
+
}
|
|
1806
|
+
} else if (resMatch[2]) {
|
|
1807
|
+
const resBody = allObjects.get(resMatch[2]) || '';
|
|
1808
|
+
const subXobjMatch = resBody.match(/\/XObject\s*(?:<<([\s\S]*?)>>|(\d+)\s+\d+\s+R)/);
|
|
1809
|
+
if (subXobjMatch) {
|
|
1810
|
+
if (subXobjMatch[1]) xobjDict = subXobjMatch[1];
|
|
1811
|
+
else if (subXobjMatch[2]) xobjDict = allObjects.get(subXobjMatch[2]) || '';
|
|
1812
|
+
}
|
|
1813
|
+
}
|
|
1814
|
+
}
|
|
1815
|
+
|
|
1816
|
+
// Asociar imágenes de esta página
|
|
1817
|
+
const pageImages = [];
|
|
1818
|
+
let combinedRefs = null;
|
|
1819
|
+
for (const [imgObjNum, imgData] of imagesByObjNum.entries()) {
|
|
1820
|
+
if (assignedImages.has(imgObjNum)) continue;
|
|
1821
|
+
if (combinedRefs === null) {
|
|
1822
|
+
combinedRefs = body + ' ' + xobjDict;
|
|
1823
|
+
for (const cNum of contentObjs) {
|
|
1824
|
+
const cBody = allObjects.get(String(cNum));
|
|
1825
|
+
if (cBody) combinedRefs += ' ' + cBody;
|
|
1826
|
+
}
|
|
1827
|
+
}
|
|
1828
|
+
if (combinedRefs.includes(imgObjNum + ' 0 R')) {
|
|
1829
|
+
assignedImages.add(imgObjNum);
|
|
1830
|
+
imgData.page = pageNum;
|
|
1831
|
+
imgData.label = `Diagrama / Imagen (Pág. ${pageNum})`;
|
|
1832
|
+
pageImages.push(imgData);
|
|
1833
|
+
allExtractedImages.push(imgData);
|
|
1834
|
+
}
|
|
1835
|
+
}
|
|
1836
|
+
|
|
1837
|
+
for (const img of pageImages) {
|
|
1838
|
+
pageItems.push(`\n\n`);
|
|
1839
|
+
}
|
|
1840
|
+
|
|
1841
|
+
const pageText = pageItems.join('\n\n').replace(/[ \t]+/g, ' ').trim();
|
|
1842
|
+
if (pageText.length > 0) {
|
|
1843
|
+
pages.push(`--- Página ${pageNum} ---\n${pageText}`);
|
|
1844
|
+
}
|
|
1845
|
+
pageNum++;
|
|
1846
|
+
}
|
|
1847
|
+
}
|
|
1848
|
+
|
|
1849
|
+
// Si quedaron imágenes no asociadas directamente al árbol de páginas, asignarlas
|
|
1850
|
+
for (const [imgObjNum, imgData] of imagesByObjNum.entries()) {
|
|
1851
|
+
if (!assignedImages.has(imgObjNum)) {
|
|
1852
|
+
assignedImages.add(imgObjNum);
|
|
1853
|
+
imgData.page = 1;
|
|
1854
|
+
imgData.label = `Diagrama / Imagen (Pág. 1)`;
|
|
1855
|
+
allExtractedImages.push(imgData);
|
|
1856
|
+
if (pages.length > 0) {
|
|
1857
|
+
pages[0] += `\n\n\n\n`;
|
|
1858
|
+
}
|
|
1859
|
+
}
|
|
1860
|
+
}
|
|
1861
|
+
|
|
1862
|
+
// B. Fallback a escaneo lineal si el árbol de páginas no produjo resultados
|
|
1863
|
+
if (pages.length === 0) {
|
|
1864
|
+
let currentPageItems = [];
|
|
1865
|
+
let pos = 0;
|
|
1866
|
+
const len = fullText.length;
|
|
1867
|
+
pageNum = 1;
|
|
1868
|
+
|
|
1869
|
+
while (pos < len) {
|
|
1870
|
+
const streamIdx = fullText.indexOf('stream', pos);
|
|
1871
|
+
if (streamIdx === -1) break;
|
|
1872
|
+
|
|
1873
|
+
const prevChar = streamIdx > 0 ? fullText.charCodeAt(streamIdx - 1) : 32;
|
|
1874
|
+
if (prevChar <= 32 || prevChar === 62 || prevChar === 47) {
|
|
1875
|
+
let dataStart = streamIdx + 6;
|
|
1876
|
+
if (dataStart < len && fullText.charCodeAt(dataStart) === 13) dataStart++;
|
|
1877
|
+
if (dataStart < len && fullText.charCodeAt(dataStart) === 10) dataStart++;
|
|
1878
|
+
|
|
1879
|
+
const endStreamIdx = fullText.indexOf('endstream', dataStart);
|
|
1880
|
+
if (endStreamIdx !== -1) {
|
|
1881
|
+
const dictStart = Math.max(0, streamIdx - 400);
|
|
1882
|
+
const dictSlice = fullText.substring(dictStart, streamIdx);
|
|
1883
|
+
const isDCT = dictSlice.includes('DCTDecode');
|
|
1884
|
+
const isFlate = dictSlice.includes('FlateDecode');
|
|
1885
|
+
const isFontOrMeta = dictSlice.includes('/Font') || dictSlice.includes('/Metadata') || dictSlice.includes('/ICCBased');
|
|
1886
|
+
let rawEnd = endStreamIdx;
|
|
1887
|
+
if (rawEnd > dataStart && (fullText.charCodeAt(rawEnd - 1) === 10 || fullText.charCodeAt(rawEnd - 1) === 13)) rawEnd--;
|
|
1888
|
+
if (rawEnd > dataStart && (fullText.charCodeAt(rawEnd - 1) === 10 || fullText.charCodeAt(rawEnd - 1) === 13)) rawEnd--;
|
|
1889
|
+
|
|
1890
|
+
const byteOffset = dataStart;
|
|
1891
|
+
const byteLen = rawEnd - dataStart;
|
|
1892
|
+
const rawStreamBytes = bytes.subarray(byteOffset, byteOffset + byteLen);
|
|
1893
|
+
|
|
1894
|
+
if (!isFontOrMeta && !isDCT) {
|
|
1895
|
+
let streamString = '';
|
|
1896
|
+
const objNumMatch = dictSlice.match(/(\d+)\s+(\d+)\s+obj[^\w]*$/);
|
|
1897
|
+
const objNum = objNumMatch ? Number(objNumMatch[1]) : 0;
|
|
1898
|
+
const generation = objNumMatch ? Number(objNumMatch[2]) : 0;
|
|
1899
|
+
const streamBytes = (security?.encrypted && objNum)
|
|
1900
|
+
? decryptPdfObjectBytes(rawStreamBytes, security, objNum, generation)
|
|
1901
|
+
: rawStreamBytes;
|
|
1902
|
+
if (isFlate) {
|
|
1903
|
+
const decompressed = await decompressDeflateData(streamBytes);
|
|
1904
|
+
if (decompressed) {
|
|
1905
|
+
streamString = decoder.decode(decompressed);
|
|
1906
|
+
}
|
|
1907
|
+
} else if (!isDCT) {
|
|
1908
|
+
streamString = decoder.decode(streamBytes);
|
|
1909
|
+
}
|
|
1910
|
+
|
|
1911
|
+
if (streamString) {
|
|
1912
|
+
const parsed = parsePdfStreamText(streamString, cmap);
|
|
1913
|
+
if (parsed && parsed.trim().length > 0) {
|
|
1914
|
+
currentPageItems.push(parsed.trim());
|
|
1915
|
+
}
|
|
1916
|
+
}
|
|
1917
|
+
}
|
|
1918
|
+
|
|
1919
|
+
if (currentPageItems.length >= 4) {
|
|
1920
|
+
const pageText = currentPageItems.join('\n\n').trim();
|
|
1921
|
+
if (pageText.length > 0) {
|
|
1922
|
+
pages.push(`--- Página ${pageNum} ---\n${pageText}`);
|
|
1923
|
+
pageNum++;
|
|
1924
|
+
currentPageItems = [];
|
|
1925
|
+
}
|
|
1926
|
+
}
|
|
1927
|
+
|
|
1928
|
+
pos = endStreamIdx + 9;
|
|
1929
|
+
continue;
|
|
1930
|
+
}
|
|
1931
|
+
}
|
|
1932
|
+
pos = streamIdx + 6;
|
|
1933
|
+
}
|
|
1934
|
+
|
|
1935
|
+
if (currentPageItems.length > 0) {
|
|
1936
|
+
const pageText = currentPageItems.join('\n\n').trim();
|
|
1937
|
+
if (pageText.length > 0) {
|
|
1938
|
+
pages.push(`--- Página ${pageNum} ---\n${pageText}`);
|
|
1939
|
+
}
|
|
1940
|
+
}
|
|
1941
|
+
}
|
|
1942
|
+
|
|
1943
|
+
// Fallback si no se pudieron dividir por páginas
|
|
1944
|
+
if (pages.length === 0) {
|
|
1945
|
+
const directText = parsePdfStreamText(fullText);
|
|
1946
|
+
if (directText && directText.trim().length > 0) {
|
|
1947
|
+
pages.push(`--- Página 1 ---\n${directText.trim()}`);
|
|
1948
|
+
}
|
|
1949
|
+
}
|
|
1950
|
+
|
|
1951
|
+
let finalCleanText = pages.join('\n\n').trim();
|
|
1952
|
+
|
|
1953
|
+
if (finalCleanText) {
|
|
1954
|
+
finalCleanText = collapseVerticallySplitGlyphs(finalCleanText);
|
|
1955
|
+
finalCleanText = decodePdfShiftedText(finalCleanText);
|
|
1956
|
+
}
|
|
1957
|
+
|
|
1958
|
+
if (!finalCleanText) {
|
|
1959
|
+
const extractionWarning = '[Documento PDF adjunto: No se pudo extraer texto seleccionable. Es posible que el PDF contenga únicamente imágenes escaneadas o esté protegido por contraseña.]';
|
|
1960
|
+
if (allExtractedImages.length > 0) {
|
|
1961
|
+
const imageReferences = allExtractedImages
|
|
1962
|
+
.map(image => ``)
|
|
1963
|
+
.join('\n\n');
|
|
1964
|
+
finalCleanText = `--- Página 1 ---\n${extractionWarning}\n\n[Se recuperaron ${allExtractedImages.length} imagen${allExtractedImages.length === 1 ? '' : 'es'} incrustada${allExtractedImages.length === 1 ? '' : 's'} del PDF.]\n\n${imageReferences}`;
|
|
1965
|
+
} else {
|
|
1966
|
+
finalCleanText = extractionWarning;
|
|
1967
|
+
}
|
|
1968
|
+
}
|
|
1969
|
+
|
|
1970
|
+
return {
|
|
1971
|
+
text: finalCleanText,
|
|
1972
|
+
images: allExtractedImages
|
|
1973
|
+
};
|
|
1974
|
+
}
|
|
1975
|
+
|
|
1976
|
+
/** Extrae únicamente el texto plano de un PDF sin dependencias externas. */
|
|
1977
|
+
async function extractTextFromPdf(arrayBuffer) {
|
|
1978
|
+
const res = await extractPdfInternal(arrayBuffer);
|
|
1979
|
+
return res.text;
|
|
1980
|
+
}
|
|
1981
|
+
|
|
1982
|
+
/** Extrae texto estructurado e imágenes incrustadas de un PDF. */
|
|
1983
|
+
async function parsePdfDocument(arrayBuffer) {
|
|
1984
|
+
return await extractPdfInternal(arrayBuffer);
|
|
1985
|
+
}
|
|
1986
|
+
|
|
1987
|
+
/**
|
|
1988
|
+
* Lee y procesa cualquier tipo de archivo (PDF, texto, código, imagen).
|
|
1989
|
+
*/
|
|
1990
|
+
async function parseFile(file) {
|
|
1991
|
+
const isPdf = file.name.toLowerCase().endsWith('.pdf') || file.type === 'application/pdf';
|
|
1992
|
+
const isImage = file.type.startsWith('image/') || /\.(png|jpe?g|webp|gif|svg)$/i.test(file.name);
|
|
1993
|
+
|
|
1994
|
+
if (isPdf) {
|
|
1995
|
+
const arrayBuffer = await file.arrayBuffer();
|
|
1996
|
+
const extractedText = await extractTextFromPdf(arrayBuffer);
|
|
1997
|
+
return {
|
|
1998
|
+
name: file.name,
|
|
1999
|
+
size: file.size,
|
|
2000
|
+
type: 'pdf',
|
|
2001
|
+
content: String(extractedText),
|
|
2002
|
+
preview: `${file.name} (${formatBytes(file.size)})`
|
|
2003
|
+
};
|
|
2004
|
+
}
|
|
2005
|
+
|
|
2006
|
+
if (isImage) {
|
|
2007
|
+
const base64 = await readFileAsDataUrl(file);
|
|
2008
|
+
const mime = file.type || (file.name.toLowerCase().endsWith('.png') ? 'image/png' : file.name.toLowerCase().endsWith('.webp') ? 'image/webp' : file.name.toLowerCase().endsWith('.gif') ? 'image/gif' : file.name.toLowerCase().endsWith('.svg') ? 'image/svg+xml' : 'image/jpeg');
|
|
2009
|
+
return {
|
|
2010
|
+
name: file.name,
|
|
2011
|
+
size: file.size,
|
|
2012
|
+
type: 'image',
|
|
2013
|
+
mimeType: mime,
|
|
2014
|
+
content: `[Imagen adjunta: ${file.name} (${formatBytes(file.size)})]`,
|
|
2015
|
+
dataUrl: base64,
|
|
2016
|
+
preview: `${file.name} (${formatBytes(file.size)})`
|
|
2017
|
+
};
|
|
2018
|
+
}
|
|
2019
|
+
|
|
2020
|
+
// Archivos de texto o código
|
|
2021
|
+
const text = await readFileAsText(file);
|
|
2022
|
+
return {
|
|
2023
|
+
name: file.name,
|
|
2024
|
+
size: file.size,
|
|
2025
|
+
type: 'text',
|
|
2026
|
+
content: text,
|
|
2027
|
+
preview: `${file.name} (${formatBytes(file.size)})`
|
|
2028
|
+
};
|
|
2029
|
+
}
|
|
2030
|
+
|
|
2031
|
+
function readFileAsText(file) {
|
|
2032
|
+
return new Promise((resolve, reject) => {
|
|
2033
|
+
const reader = new FileReader();
|
|
2034
|
+
reader.onload = () => resolve(reader.result);
|
|
2035
|
+
reader.onerror = () => reject(reader.error);
|
|
2036
|
+
reader.readAsText(file);
|
|
2037
|
+
});
|
|
2038
|
+
}
|
|
2039
|
+
|
|
2040
|
+
function readFileAsDataUrl(file) {
|
|
2041
|
+
return new Promise((resolve, reject) => {
|
|
2042
|
+
const reader = new FileReader();
|
|
2043
|
+
reader.onload = () => resolve(reader.result);
|
|
2044
|
+
reader.onerror = () => reject(reader.error);
|
|
2045
|
+
reader.readAsDataURL(file);
|
|
2046
|
+
});
|
|
2047
|
+
}
|
|
2048
|
+
|
|
2049
|
+
/**
|
|
2050
|
+
* Convierte un Data URL JPEG en espacio de color CMYK / YCCK a un Data URL JPEG sRGB bajo demanda.
|
|
2051
|
+
*/
|
|
2052
|
+
function convertCmykDataUrlToRgb(dataUrl) {
|
|
2053
|
+
if (!dataUrl || typeof dataUrl !== 'string' || !dataUrl.startsWith('data:image/jpeg;base64,')) {
|
|
2054
|
+
return dataUrl;
|
|
2055
|
+
}
|
|
2056
|
+
try {
|
|
2057
|
+
const b64 = dataUrl.substring('data:image/jpeg;base64,'.length);
|
|
2058
|
+
let bytes;
|
|
2059
|
+
if (typeof Buffer !== 'undefined') {
|
|
2060
|
+
bytes = new Uint8Array(Buffer.from(b64, 'base64'));
|
|
2061
|
+
} else if (typeof atob === 'function') {
|
|
2062
|
+
const bin = atob(b64);
|
|
2063
|
+
bytes = new Uint8Array(bin.length);
|
|
2064
|
+
for (let i = 0; i < bin.length; i++) bytes[i] = bin.charCodeAt(i);
|
|
2065
|
+
} else {
|
|
2066
|
+
return dataUrl;
|
|
2067
|
+
}
|
|
2068
|
+
const converted = convertCmykJpegToRgbDataUrl(bytes);
|
|
2069
|
+
return converted || dataUrl;
|
|
2070
|
+
} catch (_) {
|
|
2071
|
+
return dataUrl;
|
|
2072
|
+
}
|
|
2073
|
+
}
|
|
2074
|
+
|
|
2075
|
+
return {
|
|
2076
|
+
formatBytes,
|
|
2077
|
+
parseFile,
|
|
2078
|
+
extractTextFromPdf,
|
|
2079
|
+
parsePdfDocument,
|
|
2080
|
+
convertCmykJpegToRgbDataUrl,
|
|
2081
|
+
convertCmykDataUrlToRgb,
|
|
2082
|
+
decodePdfShiftedText,
|
|
2083
|
+
unshiftAsciiString,
|
|
2084
|
+
collapseSpacedLettersAndNumbers,
|
|
2085
|
+
collapseVerticallySplitGlyphs,
|
|
2086
|
+
parsePdfStreamText
|
|
2087
|
+
};
|
|
2088
|
+
});
|