tide-commander 1.199.6 → 1.201.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/dist/assets/{AudioPlayer-DbawH_kK.js → AudioPlayer-jjttJ6SM.js} +1 -1
- package/dist/assets/{BossLogsModal-DGR32RLh.js → BossLogsModal-DF0ar9g6.js} +1 -1
- package/dist/assets/{BossSpawnModal-1OOnwHa_.js → BossSpawnModal-rJfm0Ijw.js} +1 -1
- package/dist/assets/{ControlsModal-DFHBGDDH.js → ControlsModal-D4JLnO36.js} +1 -1
- package/dist/assets/{DockerLogsModal-CvQM5TOx.js → DockerLogsModal-ChSWCsD6.js} +1 -1
- package/dist/assets/{EmbeddedEditor-DGdihbaE.js → EmbeddedEditor-ChdfmuIg.js} +1 -1
- package/dist/assets/{FcStdViewer-BgT5FFuV.js → FcStdViewer-CdhcfLu8.js} +1 -1
- package/dist/assets/{GcodeViewer-CLAc1os2.js → GcodeViewer-CG5xftum.js} +1 -1
- package/dist/assets/{GlbViewer-DW3-EKE8.js → GlbViewer-DvmBSNPm.js} +1 -1
- package/dist/assets/{GmailOAuthSetup-ChWPdUvL.js → GmailOAuthSetup-yhSITtw7.js} +1 -1
- package/dist/assets/{GoogleOAuthSetup-QrN5TW8r.js → GoogleOAuthSetup-ugg8hhwa.js} +1 -1
- package/dist/assets/{IframeModal-DFnjMx4z.js → IframeModal-Bjo_-SSc.js} +1 -1
- package/dist/assets/{IntegrationsPanel-CxC744zM.js → IntegrationsPanel-JbfEFuTE.js} +2 -2
- package/dist/assets/{LogViewerModal-DAMjRGWX.js → LogViewerModal-CGGtQwwn.js} +1 -1
- package/dist/assets/{MonitoringModal-DjD33Q4Q.js → MonitoringModal-D7VxK2C8.js} +1 -1
- package/dist/assets/{PM2LogsModal-CpdUfSIi.js → PM2LogsModal-C-Ec4p65.js} +1 -1
- package/dist/assets/{RestoreArchivedAreaModal-Bjo5oNBo.js → RestoreArchivedAreaModal-BBZpQw8I.js} +1 -1
- package/dist/assets/{Scene2DCanvas-SVUlGMo1.js → Scene2DCanvas-Cpzgqdq1.js} +1 -1
- package/dist/assets/{SceneManager-BMNKqJTO.js → SceneManager-B9bEzlSi.js} +1 -1
- package/dist/assets/{SkillsPanel-C8onjf1H.js → SkillsPanel-crpQci9P.js} +1 -1
- package/dist/assets/{SlackMultiInstanceSetup-CgV-oGON.js → SlackMultiInstanceSetup-C1qkAM8r.js} +1 -1
- package/dist/assets/{SpawnModal-LXqc4WCm.js → SpawnModal-CIuJWY6N.js} +1 -1
- package/dist/assets/{StatisticsModal-ESAvPsy0.js → StatisticsModal-B14YGD66.js} +1 -1
- package/dist/assets/{StlViewer-NVExCnF3.js → StlViewer-D5EtQxO4.js} +1 -1
- package/dist/assets/{SubordinateAssignmentModal-DpdSOnqQ.js → SubordinateAssignmentModal-_8NWXgEl.js} +1 -1
- package/dist/assets/{ThreeAreaSelector-DAHbFYpb.js → ThreeAreaSelector-Bnb65bJt.js} +1 -1
- package/dist/assets/{ThreeViewShortcuts-BoxZeMmM.js → ThreeViewShortcuts-AI4Mj9zr.js} +1 -1
- package/dist/assets/{TriggerManagerPanel-CyaZWM4g.js → TriggerManagerPanel-aFrbWG0D.js} +1 -1
- package/dist/assets/{VideoPlayer-JEXTLcck.js → VideoPlayer-BSVjpcQa.js} +1 -1
- package/dist/assets/{WorkflowEditorPanel-DgqZRVud.js → WorkflowEditorPanel-CgFXTubk.js} +1 -1
- package/dist/assets/{abnfDiagram-VRR7QNED-Ntlrp2B1.js → abnfDiagram-VRR7QNED-LbVlypD7.js} +1 -1
- package/dist/assets/{arc-D3D4dl3j.js → arc-DDqZ78fN.js} +1 -1
- package/dist/assets/{architectureDiagram-ZJ3FMSHR-CbVTw6bL.js → architectureDiagram-ZJ3FMSHR-BLfa7hLc.js} +1 -1
- package/dist/assets/{area-logos-CYXdsKnv.js → area-logos-ml30JJUi.js} +1 -1
- package/dist/assets/{blockDiagram-677ZJIJ3-BI-pDXhB.js → blockDiagram-677ZJIJ3-CHWx5n03.js} +1 -1
- package/dist/assets/{c4Diagram-LMCZKHZV-74-UMGJT.js → c4Diagram-LMCZKHZV-DvUxK9_Q.js} +1 -1
- package/dist/assets/channel-CxofHie1.js +1 -0
- package/dist/assets/{chunk-2Q5K7J3B-D3vAVBE2.js → chunk-2Q5K7J3B-BFtzwIJU.js} +1 -1
- package/dist/assets/{chunk-32BRIVSS-BxxreJ5z.js → chunk-32BRIVSS-CPbVoclL.js} +1 -1
- package/dist/assets/{chunk-5VM5RSS4-D_bLkz8y.js → chunk-5VM5RSS4-dTYGN5E5.js} +1 -1
- package/dist/assets/{chunk-EX3LRPZG-CgG_gI5j.js → chunk-EX3LRPZG-CtqfsWRW.js} +1 -1
- package/dist/assets/{chunk-JWPE2WC7-CCD624Q4.js → chunk-JWPE2WC7-Bk6bqjv9.js} +1 -1
- package/dist/assets/{chunk-MOJQB5TN-BXc7SqUw.js → chunk-MOJQB5TN-UYfo6GD_.js} +1 -1
- package/dist/assets/{chunk-RYQCIY6F-DdE-eq8Q.js → chunk-RYQCIY6F-C8_6Iuf_.js} +1 -1
- package/dist/assets/{chunk-V7JOEXUC-Co4r2qOX.js → chunk-V7JOEXUC-CGdZK5My.js} +1 -1
- package/dist/assets/{chunk-VR4S4FIN-D823l1GL.js → chunk-VR4S4FIN-C-HH5QTv.js} +1 -1
- package/dist/assets/{chunk-XXDRQBXY-DNhI5Onf.js → chunk-XXDRQBXY-B2n8y5qo.js} +1 -1
- package/dist/assets/classDiagram-OUVF2IWQ-15eANucY.js +1 -0
- package/dist/assets/classDiagram-v2-EOCWNBFH-15eANucY.js +1 -0
- package/dist/assets/{cose-bilkent-JH36ORCC-DwwU-Vme.js → cose-bilkent-JH36ORCC-AB1HxIBy.js} +1 -1
- package/dist/assets/{cynefin-VYW2F7L2-FPusX1Pr.js → cynefin-VYW2F7L2-DTS_6pNv.js} +1 -1
- package/dist/assets/{cynefinDiagram-TSTJHNR4-D1ZQ8xBH.js → cynefinDiagram-TSTJHNR4-0oJZvAZy.js} +1 -1
- package/dist/assets/{dagre-VKFMJZFB-B6vs-yvj.js → dagre-VKFMJZFB-mOn_d4U5.js} +1 -1
- package/dist/assets/{diagram-FQU43EPY-EIEpqhAS.js → diagram-FQU43EPY-DqHifujZ.js} +1 -1
- package/dist/assets/{diagram-G47NLZAW-BgK2guk8.js → diagram-G47NLZAW-BZ7eUs_p.js} +1 -1
- package/dist/assets/{diagram-NH7WQ7WH-CpdwzkT0.js → diagram-NH7WQ7WH-C3r43VXW.js} +1 -1
- package/dist/assets/{diagram-OA4YK3LP-B1AdLJYG.js → diagram-OA4YK3LP-D2RSBPAX.js} +1 -1
- package/dist/assets/{diagram-WEI45ONY-BTLwXODW.js → diagram-WEI45ONY-B9Nkjrlq.js} +1 -1
- package/dist/assets/{ebnfDiagram-CCIWWBDH-BaFZwINa.js → ebnfDiagram-CCIWWBDH-B1RDvi6g.js} +1 -1
- package/dist/assets/{erDiagram-Q63AITRT-BHctMV_t.js → erDiagram-Q63AITRT-B38eKAG5.js} +1 -1
- package/dist/assets/{flowDiagram-23GEKE2U-FdEXIJDE.js → flowDiagram-23GEKE2U-C9txCjHm.js} +1 -1
- package/dist/assets/{ganttDiagram-NO4QXBWP-BzSKiNew.js → ganttDiagram-NO4QXBWP-BGNoppCl.js} +1 -1
- package/dist/assets/{gitGraphDiagram-IHSO6WYX-CCt1embf.js → gitGraphDiagram-IHSO6WYX-qrryYqaT.js} +1 -1
- package/dist/assets/{index-ChWT0VGM.js → index-B4sdd8Y2.js} +3 -3
- package/dist/assets/{index-YtoNwX7s.js → index-BXNPTpxx.js} +3 -3
- package/dist/assets/{index-YaKQQk9_.js → index-BdQoxlnS.js} +2 -2
- package/dist/assets/index-BoiyspB0.js +2 -0
- package/dist/assets/{index-UVAMV_H7.js → index-CKVTIr6G.js} +1 -1
- package/dist/assets/{index-DiILxoMf.js → index-CYxytlNP.js} +1 -1
- package/dist/assets/{index-CbRNk_Zg.js → index-ClvYehMs.js} +1 -1
- package/dist/assets/{index-vPlxnz-B.js → index-Drtp5UPa.js} +1 -1
- package/dist/assets/{index-rrOuVd7S.js → index-tE736Avl.js} +1 -1
- package/dist/assets/{infoDiagram-FWYZ7A6U-10zwrYDm.js → infoDiagram-FWYZ7A6U-DilAR8go.js} +1 -1
- package/dist/assets/{ishikawaDiagram-FXEZZL3T-ByRrZfDV.js → ishikawaDiagram-FXEZZL3T-B0cTGbt1.js} +1 -1
- package/dist/assets/{journeyDiagram-5HDEW3XC-CmYM751e.js → journeyDiagram-5HDEW3XC-eHbpGFin.js} +1 -1
- package/dist/assets/{kanban-definition-HUTT4EX6-Dh0ggxp3.js → kanban-definition-HUTT4EX6-BaSQRYl7.js} +1 -1
- package/dist/assets/{linear-CLAQtQPs.js → linear-BoUClmXF.js} +1 -1
- package/dist/assets/main-BdVckSM5.css +1 -0
- package/dist/assets/main-DuaHyH5t.js +341 -0
- package/dist/assets/{mermaid.core-CH7DoQlW.js → mermaid.core-CGrC54u1.js} +4 -4
- package/dist/assets/{mindmap-definition-LN4V7U3C-Bfw5UqxH.js → mindmap-definition-LN4V7U3C-BQWnb1yP.js} +1 -1
- package/dist/assets/{pegDiagram-2B236MQR-LbY5CO4b.js → pegDiagram-2B236MQR-Bs_4Go2e.js} +1 -1
- package/dist/assets/{pieDiagram-ENE6RG2P-D14z8IIZ.js → pieDiagram-ENE6RG2P-LCcCZeX5.js} +1 -1
- package/dist/assets/{quadrantDiagram-ABIIQ3AL-CdRVHJGf.js → quadrantDiagram-ABIIQ3AL-CdlG1WsA.js} +1 -1
- package/dist/assets/{railroadDiagram-RFXS5EU6-Csv_sb84.js → railroadDiagram-RFXS5EU6-CAcrzOwC.js} +1 -1
- package/dist/assets/{requirementDiagram-TGXJPOKE-CZFOgd5I.js → requirementDiagram-TGXJPOKE-zgaCBhns.js} +1 -1
- package/dist/assets/{sankeyDiagram-HTMAVEWB-BvBQTJvz.js → sankeyDiagram-HTMAVEWB-Cvn05yAi.js} +1 -1
- package/dist/assets/{sequenceDiagram-DBY2YBRQ-BVegUvqs.js → sequenceDiagram-DBY2YBRQ-C7oW7B56.js} +1 -1
- package/dist/assets/{sizeCapture-X5ZJPWSS-FtRVl68V.js → sizeCapture-X5ZJPWSS-Z61MsHMO.js} +1 -1
- package/dist/assets/{stateDiagram-2N3HPSRC-DjZEX_y5.js → stateDiagram-2N3HPSRC-CHjZ2LvZ.js} +1 -1
- package/dist/assets/stateDiagram-v2-6OUMAXLB-ePlfPVZw.js +1 -0
- package/dist/assets/{swimlanes-5IMT3BWC-D7iHNp7G.js → swimlanes-5IMT3BWC-BOxFXKKq.js} +2 -2
- package/dist/assets/swimlanesDiagram-G3AALYLV-ClgJ2Bw0.js +8 -0
- package/dist/assets/{timeline-definition-FHXFAJF6-DlAm5TB7.js → timeline-definition-FHXFAJF6-7Hw_U3c1.js} +1 -1
- package/dist/assets/{vennDiagram-L72KCM5P-B2jd5Egr.js → vennDiagram-L72KCM5P-DXOydhCz.js} +1 -1
- package/dist/assets/{wardleyDiagram-EHGQE667-BpSVMhG8.js → wardleyDiagram-EHGQE667-CHEopdXj.js} +1 -1
- package/dist/assets/{web-A7HWVS8k.js → web-BNIDHx-H.js} +1 -1
- package/dist/assets/{web-C9OGdTUE.js → web-DAWG2rda.js} +1 -1
- package/dist/assets/{web-DMfUbDpe.js → web-vmnQRY1J.js} +1 -1
- package/dist/assets/{xychartDiagram-FW5EYKEG-DUXSUD59.js → xychartDiagram-FW5EYKEG-BpNrwdIN.js} +1 -1
- package/dist/build-info.json +1 -1
- package/dist/index.html +2 -2
- package/dist/locales/en/terminal.json +26 -1
- package/dist/locales/es/terminal.json +2 -1
- package/dist/src/packages/server/routes/exec.js +64 -6
- package/dist/src/packages/server/routes/files.js +132 -1
- package/dist/src/packages/server/services/cfb-reader.js +153 -0
- package/dist/src/packages/server/services/document-doc.js +235 -0
- package/dist/src/packages/server/services/document-parse.js +1159 -0
- package/dist/src/packages/server/services/spreadsheet-biff.js +3 -135
- package/dist/src/packages/shared/document-types.js +22 -0
- package/package.json +1 -1
- package/dist/assets/channel-D9UzWg43.js +0 -1
- package/dist/assets/classDiagram-OUVF2IWQ-DEsVlshj.js +0 -1
- package/dist/assets/classDiagram-v2-EOCWNBFH-DEsVlshj.js +0 -1
- package/dist/assets/index-Ci4guETG.js +0 -2
- package/dist/assets/main-Dzc0Sg9A.css +0 -1
- package/dist/assets/main-fiQD9nPL.js +0 -334
- package/dist/assets/stateDiagram-v2-6OUMAXLB-Ct3Yc5aR.js +0 -1
- package/dist/assets/swimlanesDiagram-G3AALYLV-LdlH8jLU.js +0 -8
|
@@ -14,7 +14,8 @@ import { loadAreas } from '../data/index.js';
|
|
|
14
14
|
import { detectRunnerType, mightBeTestFile, mightBeVitestFile, mightBePhpTestFile } from '../services/test-runner-service.js';
|
|
15
15
|
import { DEFAULT_FILE_SEARCH_EXCLUDE_DIRS, parseExcludeDirNames } from '../../shared/file-search.js';
|
|
16
16
|
import { detectArchiveFormat, listArchive } from '../services/archive-listing.js';
|
|
17
|
-
import { readSpreadsheet, detectSpreadsheetKind, UnsupportedSpreadsheetError, SPREADSHEET_DEFAULT_MAX_ROWS, SPREADSHEET_DEFAULT_MAX_COLS, SPREADSHEET_HARD_MAX_ROWS, SPREADSHEET_HARD_MAX_COLS } from '../services/spreadsheet-parse.js';
|
|
17
|
+
import { readSpreadsheet, detectSpreadsheetKind, UnsupportedSpreadsheetError, SPREADSHEET_DEFAULT_MAX_ROWS, SPREADSHEET_DEFAULT_MAX_COLS, SPREADSHEET_HARD_MAX_ROWS, SPREADSHEET_HARD_MAX_COLS, zipSourceFromFd } from '../services/spreadsheet-parse.js';
|
|
18
|
+
import { readDocument, detectDocumentKind, UnsupportedDocumentError, DOCUMENT_DEFAULT_MAX_BLOCKS, DOCUMENT_HARD_MAX_BLOCKS } from '../services/document-parse.js';
|
|
18
19
|
import { searchFilesGlobal, searchFileContentsGlobal, FILE_SEARCH_MIN_QUERY, FILE_CONTENT_SEARCH_MIN_QUERY, } from '../services/global-file-search.js';
|
|
19
20
|
const log = logger.files;
|
|
20
21
|
// Get or create temp directory for tide-commander uploads. Exported so other
|
|
@@ -971,6 +972,136 @@ router.get('/spreadsheet', async (req, res) => {
|
|
|
971
972
|
res.status(422).json({ error: err.message });
|
|
972
973
|
}
|
|
973
974
|
});
|
|
975
|
+
/** Parsed documents, most-recently-used last (Map preserves order). */
|
|
976
|
+
const documentCache = new Map();
|
|
977
|
+
const DOCUMENT_CACHE_ENTRIES = 12;
|
|
978
|
+
// GET /api/files/document - Parse a word-processing document (docx/docm/odt/
|
|
979
|
+
// doc/rtf — see document-parse.ts) into renderable blocks for the file
|
|
980
|
+
// viewers. `blocks` caps the payload; the response says how many exist.
|
|
981
|
+
router.get('/document', async (req, res) => {
|
|
982
|
+
try {
|
|
983
|
+
const resolution = findFileWithFallbacks(req.query.path, req.query.baseDir);
|
|
984
|
+
if (!resolution.ok) {
|
|
985
|
+
const body = { error: resolution.error };
|
|
986
|
+
if (resolution.requested)
|
|
987
|
+
body.path = resolution.requested;
|
|
988
|
+
if (resolution.tried)
|
|
989
|
+
body.triedRoots = resolution.tried;
|
|
990
|
+
res.status(resolution.status).json(body);
|
|
991
|
+
return;
|
|
992
|
+
}
|
|
993
|
+
const filePath = resolution.path;
|
|
994
|
+
const stats = fs.statSync(filePath);
|
|
995
|
+
if (stats.isDirectory()) {
|
|
996
|
+
res.status(400).json({ error: 'Path is a directory', path: filePath });
|
|
997
|
+
return;
|
|
998
|
+
}
|
|
999
|
+
const filename = path.basename(filePath);
|
|
1000
|
+
if (!detectDocumentKind(filename)) {
|
|
1001
|
+
res.status(415).json({ error: 'Not a supported document type', path: filePath });
|
|
1002
|
+
return;
|
|
1003
|
+
}
|
|
1004
|
+
const rawBlocks = typeof req.query.blocks === 'string' ? parseInt(req.query.blocks, 10) : NaN;
|
|
1005
|
+
const maxBlocks = Number.isFinite(rawBlocks) && rawBlocks > 0
|
|
1006
|
+
? Math.min(rawBlocks, DOCUMENT_HARD_MAX_BLOCKS)
|
|
1007
|
+
: DOCUMENT_DEFAULT_MAX_BLOCKS;
|
|
1008
|
+
const cacheKey = `${filePath}|${stats.mtimeMs}|${stats.size}|${maxBlocks}`;
|
|
1009
|
+
const cached = documentCache.get(cacheKey);
|
|
1010
|
+
if (cached) {
|
|
1011
|
+
documentCache.delete(cacheKey);
|
|
1012
|
+
documentCache.set(cacheKey, cached);
|
|
1013
|
+
res.json(cached);
|
|
1014
|
+
return;
|
|
1015
|
+
}
|
|
1016
|
+
const parsed = await readDocument(filePath, { maxBlocks });
|
|
1017
|
+
const body = {
|
|
1018
|
+
path: filePath,
|
|
1019
|
+
filename,
|
|
1020
|
+
extension: path.extname(filePath).toLowerCase(),
|
|
1021
|
+
size: stats.size,
|
|
1022
|
+
modified: stats.mtime,
|
|
1023
|
+
format: parsed.format,
|
|
1024
|
+
title: parsed.title,
|
|
1025
|
+
author: parsed.author,
|
|
1026
|
+
blocks: parsed.blocks,
|
|
1027
|
+
footnotes: parsed.footnotes,
|
|
1028
|
+
header: parsed.header,
|
|
1029
|
+
footer: parsed.footer,
|
|
1030
|
+
wordCount: parsed.wordCount,
|
|
1031
|
+
blockCount: parsed.blockCount,
|
|
1032
|
+
truncated: parsed.truncated,
|
|
1033
|
+
plainTextOnly: parsed.plainTextOnly,
|
|
1034
|
+
maxBlocks,
|
|
1035
|
+
};
|
|
1036
|
+
documentCache.set(cacheKey, body);
|
|
1037
|
+
while (documentCache.size > DOCUMENT_CACHE_ENTRIES) {
|
|
1038
|
+
const oldest = documentCache.keys().next().value;
|
|
1039
|
+
if (oldest === undefined)
|
|
1040
|
+
break;
|
|
1041
|
+
documentCache.delete(oldest);
|
|
1042
|
+
}
|
|
1043
|
+
res.json(body);
|
|
1044
|
+
}
|
|
1045
|
+
catch (err) {
|
|
1046
|
+
if (err instanceof UnsupportedDocumentError) {
|
|
1047
|
+
res.status(415).json({ error: err.message, extension: err.extension, unsupported: true });
|
|
1048
|
+
return;
|
|
1049
|
+
}
|
|
1050
|
+
log.error(' Failed to parse document:', err);
|
|
1051
|
+
res.status(422).json({ error: err.message });
|
|
1052
|
+
}
|
|
1053
|
+
});
|
|
1054
|
+
/** Image types that may be served out of a document container. */
|
|
1055
|
+
const DOCUMENT_MEDIA_TYPES = {
|
|
1056
|
+
'.png': 'image/png', '.jpg': 'image/jpeg', '.jpeg': 'image/jpeg', '.gif': 'image/gif',
|
|
1057
|
+
'.bmp': 'image/bmp', '.webp': 'image/webp', '.svg': 'image/svg+xml', '.tif': 'image/tiff',
|
|
1058
|
+
'.tiff': 'image/tiff', '.emf': 'image/emf', '.wmf': 'image/wmf', '.ico': 'image/x-icon',
|
|
1059
|
+
};
|
|
1060
|
+
// GET /api/files/document-media - Serve ONE embedded image out of a document
|
|
1061
|
+
// container (docx/odt zip entry), so the viewer can render pictures inline
|
|
1062
|
+
// without ever extracting the archive to disk.
|
|
1063
|
+
router.get('/document-media', async (req, res) => {
|
|
1064
|
+
let fd;
|
|
1065
|
+
try {
|
|
1066
|
+
const entry = typeof req.query.entry === 'string' ? req.query.entry : '';
|
|
1067
|
+
// Only media entries of the container — never an arbitrary zip member.
|
|
1068
|
+
if (!/^(?:word\/media\/|media\/|Pictures\/)[^/]+$/i.test(entry)) {
|
|
1069
|
+
res.status(400).json({ error: 'Invalid media entry' });
|
|
1070
|
+
return;
|
|
1071
|
+
}
|
|
1072
|
+
const resolution = findFileWithFallbacks(req.query.path, req.query.baseDir);
|
|
1073
|
+
if (!resolution.ok) {
|
|
1074
|
+
res.status(resolution.status).json({ error: resolution.error });
|
|
1075
|
+
return;
|
|
1076
|
+
}
|
|
1077
|
+
const filePath = resolution.path;
|
|
1078
|
+
const stats = fs.statSync(filePath);
|
|
1079
|
+
if (stats.isDirectory() || !detectDocumentKind(path.basename(filePath))) {
|
|
1080
|
+
res.status(415).json({ error: 'Not a supported document type' });
|
|
1081
|
+
return;
|
|
1082
|
+
}
|
|
1083
|
+
fd = await fs.promises.open(filePath, 'r');
|
|
1084
|
+
const src = await zipSourceFromFd(fd, stats.size);
|
|
1085
|
+
const data = await src.readMember(entry);
|
|
1086
|
+
if (!data) {
|
|
1087
|
+
res.status(404).json({ error: 'Media entry not found in document' });
|
|
1088
|
+
return;
|
|
1089
|
+
}
|
|
1090
|
+
const ext = path.extname(entry).toLowerCase();
|
|
1091
|
+
res.setHeader('Content-Type', DOCUMENT_MEDIA_TYPES[ext] ?? 'application/octet-stream');
|
|
1092
|
+
res.setHeader('Cache-Control', 'private, max-age=3600');
|
|
1093
|
+
res.setHeader('Content-Length', String(data.length));
|
|
1094
|
+
res.end(data);
|
|
1095
|
+
}
|
|
1096
|
+
catch (err) {
|
|
1097
|
+
log.error(' Failed to read document media:', err);
|
|
1098
|
+
if (!res.headersSent)
|
|
1099
|
+
res.status(422).json({ error: err.message });
|
|
1100
|
+
}
|
|
1101
|
+
finally {
|
|
1102
|
+
await fd?.close();
|
|
1103
|
+
}
|
|
1104
|
+
});
|
|
974
1105
|
// GET /api/files/binary - Read binary file (for images, PDFs, downloads)
|
|
975
1106
|
router.get('/binary', async (req, res) => {
|
|
976
1107
|
try {
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OLE2 / Compound File Binary (CFB) container reader — dependency-free.
|
|
3
|
+
*
|
|
4
|
+
* The envelope of every legacy Office binary: `.xls` keeps its BIFF records in
|
|
5
|
+
* a "Workbook" stream, `.doc` keeps text and tables in "WordDocument" +
|
|
6
|
+
* "0Table"/"1Table", `.ppt` in "PowerPoint Document". This module exposes the
|
|
7
|
+
* directory and the streams; the format-specific readers live next to it
|
|
8
|
+
* (spreadsheet-biff.ts, document-doc.ts).
|
|
9
|
+
*/
|
|
10
|
+
const CFB_MAGIC = Buffer.from([0xd0, 0xcf, 0x11, 0xe0, 0xa1, 0xb1, 0x1a, 0xe1]);
|
|
11
|
+
const ENDOFCHAIN = 0xfffffffe;
|
|
12
|
+
const FREESECT = 0xffffffff;
|
|
13
|
+
const MAX_STREAM_BYTES = 256 * 1024 * 1024;
|
|
14
|
+
export function isCfbBuffer(buf) {
|
|
15
|
+
return buf.length >= 8 && buf.subarray(0, 8).equals(CFB_MAGIC);
|
|
16
|
+
}
|
|
17
|
+
export function readCfb(buf) {
|
|
18
|
+
if (!isCfbBuffer(buf))
|
|
19
|
+
throw new Error('not an OLE2 compound file');
|
|
20
|
+
const sectorShift = buf.readUInt16LE(0x1e);
|
|
21
|
+
const miniShift = buf.readUInt16LE(0x20);
|
|
22
|
+
const sectorSize = 1 << sectorShift;
|
|
23
|
+
const miniSectorSize = 1 << miniShift;
|
|
24
|
+
const numFatSectors = buf.readUInt32LE(0x2c);
|
|
25
|
+
const firstDirSector = buf.readUInt32LE(0x30);
|
|
26
|
+
const miniCutoff = buf.readUInt32LE(0x38);
|
|
27
|
+
const firstMiniFatSector = buf.readUInt32LE(0x3c);
|
|
28
|
+
const numMiniFatSectors = buf.readUInt32LE(0x40);
|
|
29
|
+
const firstDifatSector = buf.readUInt32LE(0x44);
|
|
30
|
+
const numDifatSectors = buf.readUInt32LE(0x48);
|
|
31
|
+
const entriesPerSector = sectorSize / 4;
|
|
32
|
+
const sectorOffset = (n) => (n + 1) * sectorSize;
|
|
33
|
+
const readSector = (n) => {
|
|
34
|
+
const off = sectorOffset(n);
|
|
35
|
+
if (off + sectorSize > buf.length) {
|
|
36
|
+
// Truncated files: pad the last sector instead of failing outright.
|
|
37
|
+
const out = Buffer.alloc(sectorSize);
|
|
38
|
+
if (off < buf.length)
|
|
39
|
+
buf.copy(out, 0, off);
|
|
40
|
+
return out;
|
|
41
|
+
}
|
|
42
|
+
return buf.subarray(off, off + sectorSize);
|
|
43
|
+
};
|
|
44
|
+
// DIFAT → list of FAT sectors.
|
|
45
|
+
const fatSectors = [];
|
|
46
|
+
for (let i = 0; i < 109 && fatSectors.length < numFatSectors; i++) {
|
|
47
|
+
const s = buf.readUInt32LE(0x4c + i * 4);
|
|
48
|
+
if (s === FREESECT || s === ENDOFCHAIN)
|
|
49
|
+
break;
|
|
50
|
+
fatSectors.push(s);
|
|
51
|
+
}
|
|
52
|
+
let difat = firstDifatSector;
|
|
53
|
+
let guard = 0;
|
|
54
|
+
while (difat !== ENDOFCHAIN && difat !== FREESECT && guard++ < numDifatSectors + 1) {
|
|
55
|
+
const sec = readSector(difat);
|
|
56
|
+
for (let i = 0; i < entriesPerSector - 1 && fatSectors.length < numFatSectors; i++) {
|
|
57
|
+
const s = sec.readUInt32LE(i * 4);
|
|
58
|
+
if (s === FREESECT || s === ENDOFCHAIN)
|
|
59
|
+
break;
|
|
60
|
+
fatSectors.push(s);
|
|
61
|
+
}
|
|
62
|
+
difat = sec.readUInt32LE((entriesPerSector - 1) * 4);
|
|
63
|
+
}
|
|
64
|
+
const fat = new Uint32Array(fatSectors.length * entriesPerSector);
|
|
65
|
+
fatSectors.forEach((s, i) => {
|
|
66
|
+
const sec = readSector(s);
|
|
67
|
+
for (let j = 0; j < entriesPerSector; j++)
|
|
68
|
+
fat[i * entriesPerSector + j] = sec.readUInt32LE(j * 4);
|
|
69
|
+
});
|
|
70
|
+
const readChain = (start, sizeHint) => {
|
|
71
|
+
const parts = [];
|
|
72
|
+
let s = start;
|
|
73
|
+
let total = 0;
|
|
74
|
+
let steps = 0;
|
|
75
|
+
const maxSteps = fat.length + 2;
|
|
76
|
+
while (s !== ENDOFCHAIN && s !== FREESECT && s < 0xfffffffa && steps++ < maxSteps) {
|
|
77
|
+
parts.push(readSector(s));
|
|
78
|
+
total += sectorSize;
|
|
79
|
+
if (total > MAX_STREAM_BYTES)
|
|
80
|
+
throw new Error('compound file stream too large');
|
|
81
|
+
if (sizeHint !== undefined && total >= sizeHint)
|
|
82
|
+
break;
|
|
83
|
+
s = s < fat.length ? fat[s] : ENDOFCHAIN;
|
|
84
|
+
}
|
|
85
|
+
const joined = Buffer.concat(parts);
|
|
86
|
+
return sizeHint !== undefined ? joined.subarray(0, Math.min(sizeHint, joined.length)) : joined;
|
|
87
|
+
};
|
|
88
|
+
// Directory entries.
|
|
89
|
+
const dir = readChain(firstDirSector);
|
|
90
|
+
const entries = [];
|
|
91
|
+
for (let off = 0; off + 128 <= dir.length; off += 128) {
|
|
92
|
+
const nameLen = dir.readUInt16LE(off + 0x40);
|
|
93
|
+
const type = dir[off + 0x42];
|
|
94
|
+
if (type === 0 && nameLen === 0)
|
|
95
|
+
continue;
|
|
96
|
+
const name = dir.subarray(off, off + Math.max(0, Math.min(64, nameLen) - 2)).toString('utf16le');
|
|
97
|
+
const startSector = dir.readUInt32LE(off + 0x74);
|
|
98
|
+
// 64-bit size field; anything beyond 2^32 is nonsense for our purposes.
|
|
99
|
+
const size = dir.readUInt32LE(off + 0x78);
|
|
100
|
+
entries.push({ name, type, startSector, size });
|
|
101
|
+
}
|
|
102
|
+
const root = entries.find((e) => e.type === 5) ?? entries[0];
|
|
103
|
+
// Mini FAT + mini stream (root entry's chain).
|
|
104
|
+
let miniFat = new Uint32Array(0);
|
|
105
|
+
let miniStream = Buffer.alloc(0);
|
|
106
|
+
if (root && numMiniFatSectors > 0 && firstMiniFatSector !== ENDOFCHAIN) {
|
|
107
|
+
const mf = readChain(firstMiniFatSector);
|
|
108
|
+
miniFat = new Uint32Array(Math.floor(mf.length / 4));
|
|
109
|
+
for (let i = 0; i < miniFat.length; i++)
|
|
110
|
+
miniFat[i] = mf.readUInt32LE(i * 4);
|
|
111
|
+
miniStream = readChain(root.startSector, root.size);
|
|
112
|
+
}
|
|
113
|
+
return { sectorSize, miniSectorSize, fat, miniFat, miniStream, miniCutoff, entries };
|
|
114
|
+
}
|
|
115
|
+
export function readCfbStream(cfb, entry, buf) {
|
|
116
|
+
if (entry.size < cfb.miniCutoff) {
|
|
117
|
+
const parts = [];
|
|
118
|
+
let s = entry.startSector;
|
|
119
|
+
let total = 0;
|
|
120
|
+
let steps = 0;
|
|
121
|
+
while (s !== ENDOFCHAIN && s !== FREESECT && s < 0xfffffffa && steps++ < cfb.miniFat.length + 2 && total < entry.size) {
|
|
122
|
+
const off = s * cfb.miniSectorSize;
|
|
123
|
+
parts.push(cfb.miniStream.subarray(off, off + cfb.miniSectorSize));
|
|
124
|
+
total += cfb.miniSectorSize;
|
|
125
|
+
s = s < cfb.miniFat.length ? cfb.miniFat[s] : ENDOFCHAIN;
|
|
126
|
+
}
|
|
127
|
+
return Buffer.concat(parts).subarray(0, entry.size);
|
|
128
|
+
}
|
|
129
|
+
const parts = [];
|
|
130
|
+
let s = entry.startSector;
|
|
131
|
+
let total = 0;
|
|
132
|
+
let steps = 0;
|
|
133
|
+
while (s !== ENDOFCHAIN && s !== FREESECT && s < 0xfffffffa && steps++ < cfb.fat.length + 2 && total < entry.size) {
|
|
134
|
+
const off = (s + 1) * cfb.sectorSize;
|
|
135
|
+
const sec = off + cfb.sectorSize <= buf.length ? buf.subarray(off, off + cfb.sectorSize) : Buffer.concat([buf.subarray(off), Buffer.alloc(Math.max(0, off + cfb.sectorSize - buf.length))]);
|
|
136
|
+
parts.push(sec);
|
|
137
|
+
total += cfb.sectorSize;
|
|
138
|
+
if (total > MAX_STREAM_BYTES)
|
|
139
|
+
throw new Error('compound file stream too large');
|
|
140
|
+
s = s < cfb.fat.length ? cfb.fat[s] : ENDOFCHAIN;
|
|
141
|
+
}
|
|
142
|
+
return Buffer.concat(parts).subarray(0, entry.size);
|
|
143
|
+
}
|
|
144
|
+
/** Read a stream by (case-insensitive) name; null when absent. */
|
|
145
|
+
export function readCfbStreamByName(buf, cfb, name) {
|
|
146
|
+
const target = name.toLowerCase();
|
|
147
|
+
const entry = cfb.entries.find((e) => e.type === 2 && e.name.toLowerCase() === target);
|
|
148
|
+
return entry ? readCfbStream(cfb, entry, buf) : null;
|
|
149
|
+
}
|
|
150
|
+
/** Names of the streams in the container (for diagnostics / sniffing). */
|
|
151
|
+
export function cfbStreamNames(cfb) {
|
|
152
|
+
return cfb.entries.filter((e) => e.type === 2).map((e) => e.name);
|
|
153
|
+
}
|
|
@@ -0,0 +1,235 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Legacy Word (.doc, Word 97-2003) reader — dependency-free, text + structure.
|
|
3
|
+
*
|
|
4
|
+
* A .doc is an OLE2 compound file (see cfb-reader.ts). The "WordDocument"
|
|
5
|
+
* stream starts with the FIB (File Information Block), whose `fcClx` /`lcbClx`
|
|
6
|
+
* point into the table stream ("1Table" or "0Table", chosen by the FIB's
|
|
7
|
+
* `fWhichTblStm` flag). The Clx ends with a Pcdt: a PlcPcd — an array of
|
|
8
|
+
* character positions followed by 8-byte piece descriptors. Each piece says
|
|
9
|
+
* where its text lives in WordDocument and whether it is UTF-16 or CP1252
|
|
10
|
+
* (bit 30 of the fc field, with the offset halved when set).
|
|
11
|
+
*
|
|
12
|
+
* What we recover: the document text split into paragraphs (Word separates
|
|
13
|
+
* them with CR = 0x0D), with the field/footnote/header sentinels and the
|
|
14
|
+
* cell/row marks turned into readable structure. Formatting (bold, styles)
|
|
15
|
+
* lives in the CHPX/PAPX binary tables and is deliberately NOT parsed — the
|
|
16
|
+
* viewer marks these documents `plainTextOnly` and suggests converting to
|
|
17
|
+
* .docx for full fidelity.
|
|
18
|
+
*/
|
|
19
|
+
import { cfbStreamNames, isCfbBuffer, readCfb, readCfbStreamByName } from './cfb-reader.js';
|
|
20
|
+
const FIB_MAGIC_WORD = 0xa5ec; // wIdent for Word 97+ (0xA5DC/0xA5DB are Word 6/95)
|
|
21
|
+
/** Parse the PlcPcd (piece table) out of the Clx blob. */
|
|
22
|
+
export function parsePieceTable(clx) {
|
|
23
|
+
// Clx = Prc* Pcdt. Skip Prc entries (0x01 + 2-byte cbGrpprl + data).
|
|
24
|
+
let pos = 0;
|
|
25
|
+
while (pos < clx.length && clx[pos] === 0x01) {
|
|
26
|
+
if (pos + 3 > clx.length)
|
|
27
|
+
return [];
|
|
28
|
+
const cb = clx.readUInt16LE(pos + 1);
|
|
29
|
+
pos += 3 + cb;
|
|
30
|
+
}
|
|
31
|
+
if (pos >= clx.length || clx[pos] !== 0x02)
|
|
32
|
+
return [];
|
|
33
|
+
if (pos + 5 > clx.length)
|
|
34
|
+
return [];
|
|
35
|
+
const lcb = clx.readUInt32LE(pos + 1);
|
|
36
|
+
const plc = clx.subarray(pos + 5, pos + 5 + lcb);
|
|
37
|
+
// PlcPcd: (n+1) CPs (4 bytes each) then n PCDs (8 bytes each).
|
|
38
|
+
const n = Math.floor((plc.length - 4) / 12);
|
|
39
|
+
if (n <= 0)
|
|
40
|
+
return [];
|
|
41
|
+
const pieces = [];
|
|
42
|
+
for (let i = 0; i < n; i++) {
|
|
43
|
+
const cpStart = plc.readUInt32LE(i * 4);
|
|
44
|
+
const cpEnd = plc.readUInt32LE((i + 1) * 4);
|
|
45
|
+
const pcdOff = (n + 1) * 4 + i * 8;
|
|
46
|
+
if (pcdOff + 8 > plc.length)
|
|
47
|
+
break;
|
|
48
|
+
const fcValue = plc.readUInt32LE(pcdOff + 2);
|
|
49
|
+
const compressed = (fcValue & 0x40000000) !== 0;
|
|
50
|
+
const fc = compressed ? (fcValue & 0x3fffffff) / 2 : (fcValue & 0x3fffffff);
|
|
51
|
+
pieces.push({ cpStart, cpEnd, fc: Math.floor(fc), compressed });
|
|
52
|
+
}
|
|
53
|
+
return pieces;
|
|
54
|
+
}
|
|
55
|
+
/** CP1252 high range (0x80-0x9F) — the rest of latin1 maps 1:1. */
|
|
56
|
+
const CP1252_HIGH = '€‚ƒ„…†‡ˆ‰Š‹ŒŽ‘’“”•–—˜™š›œžŸ';
|
|
57
|
+
function decodeCp1252(bytes) {
|
|
58
|
+
let out = '';
|
|
59
|
+
for (const b of bytes)
|
|
60
|
+
out += b >= 0x80 && b <= 0x9f ? CP1252_HIGH[b - 0x80] : String.fromCharCode(b);
|
|
61
|
+
return out;
|
|
62
|
+
}
|
|
63
|
+
/** Concatenate the text of every piece, in CP order. */
|
|
64
|
+
export function readPieceText(wordDocument, pieces) {
|
|
65
|
+
const parts = [];
|
|
66
|
+
for (const p of pieces) {
|
|
67
|
+
const chars = Math.max(0, p.cpEnd - p.cpStart);
|
|
68
|
+
if (chars === 0)
|
|
69
|
+
continue;
|
|
70
|
+
const bytes = p.compressed ? chars : chars * 2;
|
|
71
|
+
if (p.fc < 0 || p.fc >= wordDocument.length)
|
|
72
|
+
continue;
|
|
73
|
+
const slice = wordDocument.subarray(p.fc, Math.min(wordDocument.length, p.fc + bytes));
|
|
74
|
+
parts.push(p.compressed ? decodeCp1252(slice) : slice.toString('utf16le'));
|
|
75
|
+
}
|
|
76
|
+
return parts.join('');
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* Word control characters that must not reach the reader:
|
|
80
|
+
* 0x01 picture placeholder · 0x02 footnote/annotation ref · 0x05 comment
|
|
81
|
+
* 0x08 drawn object · 0x13/0x14/0x15 field begin/separator/end · 0x0C page
|
|
82
|
+
* break · 0x0E column break · 0x1E non-breaking hyphen · 0x1F optional hyphen.
|
|
83
|
+
* Field instructions (between 0x13 and 0x14) are metadata (HYPERLINK "…",
|
|
84
|
+
* PAGE, TOC) — dropped; the field RESULT (0x14…0x15) is kept.
|
|
85
|
+
*/
|
|
86
|
+
export function docTextToParagraphs(text, maxBlocks) {
|
|
87
|
+
const blocks = [];
|
|
88
|
+
let blockCount = 0;
|
|
89
|
+
let truncated = false;
|
|
90
|
+
let current = '';
|
|
91
|
+
let inFieldInstruction = false;
|
|
92
|
+
let cellFlush = false;
|
|
93
|
+
const push = () => {
|
|
94
|
+
const cleaned = current.replace(/[\u0000-\u0008\u000e-\u001f]/g, '').replace(/\u00a0/g, ' ').trimEnd();
|
|
95
|
+
current = '';
|
|
96
|
+
if (cleaned.trim() === '')
|
|
97
|
+
return;
|
|
98
|
+
blockCount++;
|
|
99
|
+
if (blocks.length >= maxBlocks) {
|
|
100
|
+
truncated = true;
|
|
101
|
+
return;
|
|
102
|
+
}
|
|
103
|
+
const para = { type: 'paragraph', runs: [{ text: cleaned }] };
|
|
104
|
+
blocks.push(para);
|
|
105
|
+
};
|
|
106
|
+
for (let i = 0; i < text.length; i++) {
|
|
107
|
+
const code = text.charCodeAt(i);
|
|
108
|
+
switch (code) {
|
|
109
|
+
case 0x13:
|
|
110
|
+
inFieldInstruction = true;
|
|
111
|
+
continue; // field begin
|
|
112
|
+
case 0x14:
|
|
113
|
+
inFieldInstruction = false;
|
|
114
|
+
continue; // field separator
|
|
115
|
+
case 0x15: continue; // field end
|
|
116
|
+
case 0x0d:
|
|
117
|
+
case 0x0a: // paragraph end
|
|
118
|
+
push();
|
|
119
|
+
continue;
|
|
120
|
+
case 0x07:
|
|
121
|
+
// Every cell ends with 0x07; the ROW ends with one more right after
|
|
122
|
+
// the last cell's — so two in a row mean "end of line", one means
|
|
123
|
+
// "next cell" (rendered as a tab, like the .doc → text convention).
|
|
124
|
+
if (cellFlush) {
|
|
125
|
+
push();
|
|
126
|
+
cellFlush = false;
|
|
127
|
+
}
|
|
128
|
+
else {
|
|
129
|
+
current += '\t';
|
|
130
|
+
cellFlush = true;
|
|
131
|
+
}
|
|
132
|
+
continue;
|
|
133
|
+
case 0x0b:
|
|
134
|
+
current += '\n';
|
|
135
|
+
continue; // line break
|
|
136
|
+
case 0x0c:
|
|
137
|
+
push();
|
|
138
|
+
continue; // page break
|
|
139
|
+
case 0x1e:
|
|
140
|
+
current += '-';
|
|
141
|
+
continue; // non-breaking hyphen
|
|
142
|
+
case 0x1f: continue; // optional hyphen
|
|
143
|
+
default:
|
|
144
|
+
if (inFieldInstruction)
|
|
145
|
+
continue;
|
|
146
|
+
if (code < 0x20 && code !== 0x09)
|
|
147
|
+
continue;
|
|
148
|
+
current += text[i];
|
|
149
|
+
}
|
|
150
|
+
// Only a SECOND consecutive 0x07 is a row mark, so the flag survives only
|
|
151
|
+
// until the next character.
|
|
152
|
+
cellFlush = false;
|
|
153
|
+
}
|
|
154
|
+
push();
|
|
155
|
+
return { blocks, blockCount, truncated };
|
|
156
|
+
}
|
|
157
|
+
/** Parse a Word 97-2003 .doc buffer into plain paragraphs. */
|
|
158
|
+
export function parseDocBuffer(buf, maxBlocks) {
|
|
159
|
+
if (!isCfbBuffer(buf))
|
|
160
|
+
throw new Error('not an OLE2 compound file');
|
|
161
|
+
const cfb = readCfb(buf);
|
|
162
|
+
const wordDocument = readCfbStreamByName(buf, cfb, 'WordDocument');
|
|
163
|
+
if (!wordDocument) {
|
|
164
|
+
const names = cfbStreamNames(cfb).slice(0, 8).join(', ');
|
|
165
|
+
throw new Error(`OLE2 file without a WordDocument stream (streams: ${names || 'none'}) — not a Word document`);
|
|
166
|
+
}
|
|
167
|
+
if (wordDocument.length < 0x0200)
|
|
168
|
+
throw new Error('WordDocument stream is too small to hold a FIB');
|
|
169
|
+
const wIdent = wordDocument.readUInt16LE(0);
|
|
170
|
+
if (wIdent !== FIB_MAGIC_WORD) {
|
|
171
|
+
throw new Error(wIdent === 0xa5db || wIdent === 0xa5dc
|
|
172
|
+
? 'This is a Word 6/95 document — too old for the built-in viewer; open it in LibreOffice and save as .docx.'
|
|
173
|
+
: `unrecognized Word file header (0x${wIdent.toString(16)})`);
|
|
174
|
+
}
|
|
175
|
+
// FIB base: flags at 0x0A, fWhichTblStm is bit 9 (0x0200).
|
|
176
|
+
const flags = wordDocument.readUInt16LE(0x0a);
|
|
177
|
+
const tableName = (flags & 0x0200) ? '1Table' : '0Table';
|
|
178
|
+
const table = readCfbStreamByName(buf, cfb, tableName) ?? readCfbStreamByName(buf, cfb, tableName === '1Table' ? '0Table' : '1Table');
|
|
179
|
+
if (!table)
|
|
180
|
+
throw new Error('Word document without a table stream (0Table/1Table)');
|
|
181
|
+
// The FibRgFcLcb array starts after the FIB base (0x0020) + csw (2 + 2*csw)
|
|
182
|
+
// + cslw (2 + 4*cslw) + cbRgFcLcb (2). Its entries are (fc, lcb) u32 pairs;
|
|
183
|
+
// fcClx/lcbClx is entry 33 of FibRgFcLcb97 (unchanged in the 2000-2007
|
|
184
|
+
// extensions, which only append).
|
|
185
|
+
let pos = 0x0020;
|
|
186
|
+
const csw = wordDocument.readUInt16LE(pos);
|
|
187
|
+
pos += 2 + csw * 2;
|
|
188
|
+
const cslw = wordDocument.readUInt16LE(pos);
|
|
189
|
+
pos += 2 + cslw * 4;
|
|
190
|
+
const cbRgFcLcb = wordDocument.readUInt16LE(pos);
|
|
191
|
+
pos += 2;
|
|
192
|
+
const rgFcLcb = wordDocument.subarray(pos, pos + cbRgFcLcb * 8);
|
|
193
|
+
const CLX_INDEX = 33;
|
|
194
|
+
if (rgFcLcb.length < (CLX_INDEX + 1) * 8)
|
|
195
|
+
throw new Error('Word FIB is truncated (no Clx pointer)');
|
|
196
|
+
let fcClx = rgFcLcb.readUInt32LE(CLX_INDEX * 8);
|
|
197
|
+
let lcbClx = rgFcLcb.readUInt32LE(CLX_INDEX * 8 + 4);
|
|
198
|
+
let pieces = lcbClx > 0 && fcClx + lcbClx <= table.length
|
|
199
|
+
? parsePieceTable(table.subarray(fcClx, fcClx + lcbClx))
|
|
200
|
+
: [];
|
|
201
|
+
if (pieces.length === 0) {
|
|
202
|
+
// Writers that lay the FIB out differently (or files repaired by other
|
|
203
|
+
// tools) still keep a Clx in the table stream: find the entry that points
|
|
204
|
+
// at one and yields a usable piece table.
|
|
205
|
+
for (let i = 0; i < cbRgFcLcb && pieces.length === 0; i++) {
|
|
206
|
+
const fc = rgFcLcb.readUInt32LE(i * 8);
|
|
207
|
+
const lcb = rgFcLcb.readUInt32LE(i * 8 + 4);
|
|
208
|
+
if (lcb <= 8 || fc + lcb > table.length)
|
|
209
|
+
continue;
|
|
210
|
+
if (table[fc] !== 0x01 && table[fc] !== 0x02)
|
|
211
|
+
continue;
|
|
212
|
+
const candidate = parsePieceTable(table.subarray(fc, fc + lcb));
|
|
213
|
+
if (candidate.length > 0) {
|
|
214
|
+
pieces = candidate;
|
|
215
|
+
fcClx = fc;
|
|
216
|
+
lcbClx = lcb;
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
if (pieces.length === 0)
|
|
221
|
+
throw new Error('Word document has no readable piece table');
|
|
222
|
+
const text = readPieceText(wordDocument, pieces);
|
|
223
|
+
const { blocks, blockCount, truncated } = docTextToParagraphs(text, maxBlocks);
|
|
224
|
+
let wordCount = 0;
|
|
225
|
+
for (const b of blocks) {
|
|
226
|
+
if (b.type !== 'paragraph')
|
|
227
|
+
continue;
|
|
228
|
+
for (const r of b.runs) {
|
|
229
|
+
const t = r.text.trim();
|
|
230
|
+
if (t)
|
|
231
|
+
wordCount += t.split(/\s+/).length;
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
return { format: 'doc', blocks, wordCount, blockCount, truncated, plainTextOnly: true };
|
|
235
|
+
}
|