officeparser 5.2.2 → 6.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +411 -163
- package/dist/OfficeParser.d.ts +90 -0
- package/dist/OfficeParser.js +217 -0
- package/dist/index.d.ts +51 -0
- package/dist/index.js +108 -0
- package/dist/officeparser.browser.js +165 -0
- package/dist/officeparser.browser.js.map +7 -0
- package/dist/parsers/ExcelParser.d.ts +33 -0
- package/dist/parsers/ExcelParser.js +643 -0
- package/dist/parsers/OpenOfficeParser.d.ts +32 -0
- package/dist/parsers/OpenOfficeParser.js +1399 -0
- package/dist/parsers/PdfParser.d.ts +68 -0
- package/dist/parsers/PdfParser.js +847 -0
- package/dist/parsers/PowerPointParser.d.ts +33 -0
- package/dist/parsers/PowerPointParser.js +778 -0
- package/dist/parsers/RtfParser.d.ts +164 -0
- package/dist/parsers/RtfParser.js +1641 -0
- package/dist/parsers/WordParser.d.ts +79 -0
- package/dist/parsers/WordParser.js +787 -0
- package/dist/types.d.ts +615 -0
- package/dist/types.js +2 -0
- package/dist/utils/chartUtils.d.ts +7 -0
- package/dist/utils/chartUtils.js +255 -0
- package/dist/utils/errorUtils.d.ts +58 -0
- package/dist/utils/errorUtils.js +120 -0
- package/dist/utils/imageUtils.d.ts +67 -0
- package/dist/utils/imageUtils.js +133 -0
- package/dist/utils/ocrUtils.d.ts +39 -0
- package/dist/utils/ocrUtils.js +61 -0
- package/dist/utils/xmlUtils.d.ts +83 -0
- package/dist/utils/xmlUtils.js +158 -0
- package/dist/utils/zipUtils.d.ts +74 -0
- package/dist/utils/zipUtils.js +112 -0
- package/package.json +44 -17
- package/officeParser.js +0 -790
- package/typings/officeParser.d.ts +0 -33
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Word Document (DOCX) Parser
|
|
3
|
+
*
|
|
4
|
+
* **DOCX Format Overview:**
|
|
5
|
+
* DOCX is the default format for Microsoft Word documents since Office 2007.
|
|
6
|
+
* It's based on the Office Open XML (OOXML) standard (ECMA-376, ISO/IEC 29500).
|
|
7
|
+
*
|
|
8
|
+
* **File Structure:**
|
|
9
|
+
* DOCX files are ZIP archives containing:
|
|
10
|
+
* - `word/document.xml` - Main document content
|
|
11
|
+
* - `word/styles.xml` - Style definitions
|
|
12
|
+
* - `word/numbering.xml` - List numbering definitions
|
|
13
|
+
* - `word/footnotes.xml` - Footnotes content
|
|
14
|
+
* - `word/media/*` - Embedded images and media
|
|
15
|
+
* - `docProps/core.xml` - Document metadata
|
|
16
|
+
* - `[Content_Types].xml` - MIME type mappings
|
|
17
|
+
*
|
|
18
|
+
* **XML Structure (word/document.xml):**
|
|
19
|
+
* ```xml
|
|
20
|
+
* <w:document>
|
|
21
|
+
* <w:body>
|
|
22
|
+
* <w:p> <!-- Paragraph -->
|
|
23
|
+
* <w:pPr> <!-- Paragraph properties -->
|
|
24
|
+
* <w:pStyle w:val="Heading1"/>
|
|
25
|
+
* </w:pPr>
|
|
26
|
+
* <w:r> <!-- Run (text with same formatting) -->
|
|
27
|
+
* <w:rPr> <!-- Run properties -->
|
|
28
|
+
* <w:b/> <!-- Bold -->
|
|
29
|
+
* <w:sz w:val="24"/> <!-- Font size (half-points) -->
|
|
30
|
+
* </w:rPr>
|
|
31
|
+
* <w:t>Hello</w:t> <!-- Text -->
|
|
32
|
+
* </w:r>
|
|
33
|
+
* </w:p>
|
|
34
|
+
* </w:body>
|
|
35
|
+
* </w:document>
|
|
36
|
+
* ```
|
|
37
|
+
*
|
|
38
|
+
* **Key OOXML Elements:**
|
|
39
|
+
* - `<w:p>` - Paragraph
|
|
40
|
+
* - `<w:r>` - Run (contiguous text with same formatting)
|
|
41
|
+
* - `<w:t>` - Text content
|
|
42
|
+
* - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
|
|
43
|
+
* - `<w:pStyle>` - Paragraph style (for headings)
|
|
44
|
+
* - `<w:numPr>` - List numbering properties
|
|
45
|
+
* - `<w:tbl>` - Table
|
|
46
|
+
* - `<w:drawing>` - Drawing/image
|
|
47
|
+
*
|
|
48
|
+
* **Parsing Approach:**
|
|
49
|
+
* 1. Extract ZIP contents
|
|
50
|
+
* 2. Parse word/document.xml for structure and text
|
|
51
|
+
* 3. Extract formatting from run properties (rPr)
|
|
52
|
+
* 4. Identify headings via paragraph styles
|
|
53
|
+
* 5. Extract footnotes from word/footnotes.xml
|
|
54
|
+
* 6. Process embedded images from word/media/*
|
|
55
|
+
* 7. Parse metadata from docProps/core.xml
|
|
56
|
+
*
|
|
57
|
+
* @module WordParser
|
|
58
|
+
* @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
|
|
59
|
+
* @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
|
|
60
|
+
*/
|
|
61
|
+
/// <reference types="node" />
|
|
62
|
+
import { OfficeParserAST, OfficeParserConfig } from '../types';
|
|
63
|
+
/**
|
|
64
|
+
* Parses a Word document (.docx) and extracts content, formatting, and metadata.
|
|
65
|
+
*
|
|
66
|
+
* The parsing process:
|
|
67
|
+
* 1. Unzip the DOCX file
|
|
68
|
+
* 2. Parse word/document.xml to extract paragraphs and runs
|
|
69
|
+
* 3. Extract text formatting from run properties
|
|
70
|
+
* 4. Identify headings from paragraph styles
|
|
71
|
+
* 5. Process lists from numbering properties
|
|
72
|
+
* 6. Extract images and optionally perform OCR
|
|
73
|
+
* 7. Parse document metadata
|
|
74
|
+
*
|
|
75
|
+
* @param buffer - The DOCX file as a Buffer
|
|
76
|
+
* @param config - Parser configuration options
|
|
77
|
+
* @returns A promise resolving to the parsed AST
|
|
78
|
+
*/
|
|
79
|
+
export declare const parseWord: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;
|