officeparser 5.2.1 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,79 @@
1
+ /**
2
+ * Word Document (DOCX) Parser
3
+ *
4
+ * **DOCX Format Overview:**
5
+ * DOCX is the default format for Microsoft Word documents since Office 2007.
6
+ * It's based on the Office Open XML (OOXML) standard (ECMA-376, ISO/IEC 29500).
7
+ *
8
+ * **File Structure:**
9
+ * DOCX files are ZIP archives containing:
10
+ * - `word/document.xml` - Main document content
11
+ * - `word/styles.xml` - Style definitions
12
+ * - `word/numbering.xml` - List numbering definitions
13
+ * - `word/footnotes.xml` - Footnotes content
14
+ * - `word/media/*` - Embedded images and media
15
+ * - `docProps/core.xml` - Document metadata
16
+ * - `[Content_Types].xml` - MIME type mappings
17
+ *
18
+ * **XML Structure (word/document.xml):**
19
+ * ```xml
20
+ * <w:document>
21
+ * <w:body>
22
+ * <w:p> <!-- Paragraph -->
23
+ * <w:pPr> <!-- Paragraph properties -->
24
+ * <w:pStyle w:val="Heading1"/>
25
+ * </w:pPr>
26
+ * <w:r> <!-- Run (text with same formatting) -->
27
+ * <w:rPr> <!-- Run properties -->
28
+ * <w:b/> <!-- Bold -->
29
+ * <w:sz w:val="24"/> <!-- Font size (half-points) -->
30
+ * </w:rPr>
31
+ * <w:t>Hello</w:t> <!-- Text -->
32
+ * </w:r>
33
+ * </w:p>
34
+ * </w:body>
35
+ * </w:document>
36
+ * ```
37
+ *
38
+ * **Key OOXML Elements:**
39
+ * - `<w:p>` - Paragraph
40
+ * - `<w:r>` - Run (contiguous text with same formatting)
41
+ * - `<w:t>` - Text content
42
+ * - `<w:b>`, `<w:i>`, `<w:u>` - Bold, italic, underline
43
+ * - `<w:pStyle>` - Paragraph style (for headings)
44
+ * - `<w:numPr>` - List numbering properties
45
+ * - `<w:tbl>` - Table
46
+ * - `<w:drawing>` - Drawing/image
47
+ *
48
+ * **Parsing Approach:**
49
+ * 1. Extract ZIP contents
50
+ * 2. Parse word/document.xml for structure and text
51
+ * 3. Extract formatting from run properties (rPr)
52
+ * 4. Identify headings via paragraph styles
53
+ * 5. Extract footnotes from word/footnotes.xml
54
+ * 6. Process embedded images from word/media/*
55
+ * 7. Parse metadata from docProps/core.xml
56
+ *
57
+ * @module WordParser
58
+ * @see https://www.ecma-international.org/publications-and-standards/standards/ecma-376/ OOXML Standard
59
+ * @see https://learn.microsoft.com/en-us/openspecs/office_standards/ms-docx/ [MS-DOCX] Specification
60
+ */
61
+ /// <reference types="node" />
62
+ import { OfficeParserAST, OfficeParserConfig } from '../types';
63
+ /**
64
+ * Parses a Word document (.docx) and extracts content, formatting, and metadata.
65
+ *
66
+ * The parsing process:
67
+ * 1. Unzip the DOCX file
68
+ * 2. Parse word/document.xml to extract paragraphs and runs
69
+ * 3. Extract text formatting from run properties
70
+ * 4. Identify headings from paragraph styles
71
+ * 5. Process lists from numbering properties
72
+ * 6. Extract images and optionally perform OCR
73
+ * 7. Parse document metadata
74
+ *
75
+ * @param buffer - The DOCX file as a Buffer
76
+ * @param config - Parser configuration options
77
+ * @returns A promise resolving to the parsed AST
78
+ */
79
+ export declare const parseWord: (buffer: Buffer, config: OfficeParserConfig) => Promise<OfficeParserAST>;