officeparser 6.1.0 → 7.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +284 -86
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +55 -28
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +107 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +878 -5
  32. package/dist/officeparser.browser.iife.js +703 -49
  33. package/dist/officeparser.browser.mjs +703 -49
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +237 -128
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +132 -123
  48. package/dist/parsers/RtfParser.d.ts +22 -2
  49. package/dist/parsers/RtfParser.js +1398 -1282
  50. package/dist/parsers/WordParser.d.ts +3 -2
  51. package/dist/parsers/WordParser.js +333 -115
  52. package/dist/sbom.cdx.json +103 -103
  53. package/dist/types.d.ts +833 -5
  54. package/dist/types.js +71 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.js +56 -2
  60. package/dist/utils/errorUtils.d.ts +17 -29
  61. package/dist/utils/errorUtils.js +109 -52
  62. package/dist/utils/moduleLoader.js +15 -9
  63. package/dist/utils/ocrUtils.js +2 -1
  64. package/dist/utils/sheetUtils.d.ts +7 -0
  65. package/dist/utils/sheetUtils.js +35 -0
  66. package/dist/utils/styleMapper.d.ts +36 -0
  67. package/dist/utils/styleMapper.js +224 -0
  68. package/dist/utils/xmlUtils.d.ts +0 -8
  69. package/dist/utils/xmlUtils.js +2 -1
  70. package/package.json +28 -9
@@ -7,46 +7,62 @@
7
7
  * consistent error reporting across all parsers and the main entry point.
8
8
  */
9
9
  Object.defineProperty(exports, "__esModule", { value: true });
10
- exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.OfficeErrorType = void 0;
10
+ exports.logWarning = exports.getWrappedError = exports.getOfficeError = exports.getWarningMessage = void 0;
11
+ const types_js_1 = require("../types.js");
11
12
  /** Error header prefix for all error messages */
12
13
  const ERRORHEADER = "[OfficeParser]: ";
13
- /**
14
- * Standard error types for OfficeParser.
15
- * Use these to identify the kind of error being reported.
16
- */
17
- var OfficeErrorType;
18
- (function (OfficeErrorType) {
19
- /** Unsupported file extension */
20
- OfficeErrorType["EXTENSION_UNSUPPORTED"] = "EXTENSION_UNSUPPORTED";
21
- /** File appears to be corrupted or malformed */
22
- OfficeErrorType["FILE_CORRUPTED"] = "FILE_CORRUPTED";
23
- /** File could not be found at the specified path */
24
- OfficeErrorType["FILE_DOES_NOT_EXIST"] = "FILE_DOES_NOT_EXIST";
25
- /** Specified location/directory is not reachable or is a directory */
26
- OfficeErrorType["LOCATION_NOT_FOUND"] = "LOCATION_NOT_FOUND";
27
- /** Arguments passed to the function are missing or invalid */
28
- OfficeErrorType["IMPROPER_ARGUMENTS"] = "IMPROPER_ARGUMENTS";
29
- /** Error occurred while reading or processing file buffers */
30
- OfficeErrorType["IMPROPER_BUFFERS"] = "IMPROPER_BUFFERS";
31
- /** Input type is not a supported type (string, Buffer, ArrayBuffer) */
32
- OfficeErrorType["INVALID_INPUT"] = "INVALID_INPUT";
33
- /** PDF worker source is missing (required in browser) */
34
- OfficeErrorType["PDF_WORKER_MISSING"] = "PDF_WORKER_MISSING";
35
- })(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
36
14
  /**
37
15
  * Lookup table for error messages.
38
16
  * Some entries are functions that take parameters to build dynamic messages.
39
17
  */
40
18
  const ERROR_MESSAGES = {
41
- [OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
42
- [OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
43
- [OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
44
- [OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
45
- [OfficeErrorType.IMPROPER_ARGUMENTS]: `Improper arguments`,
46
- [OfficeErrorType.IMPROPER_BUFFERS]: `Error occured while reading the file buffers`,
47
- [OfficeErrorType.INVALID_INPUT]: `Invalid input type: Expected a Buffer or a valid file path`,
48
- [OfficeErrorType.PDF_WORKER_MISSING]: `Missing PDF worker configuration. PDF parsing in browser environments requires a worker source. Please provide "pdfWorkerSrc" in your configuration.`
19
+ [types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
20
+ [types_js_1.OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
21
+ [types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
22
+ [types_js_1.OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
23
+ [types_js_1.OfficeErrorType.IMPROPER_ARGUMENTS]: `Improper arguments`,
24
+ [types_js_1.OfficeErrorType.IMPROPER_BUFFERS]: `Error occured while reading the file buffers. If you are passing text-based formats like md, html, or csv as a Buffer, you must provide the 'fileType' hint in the configuration as they lack magic bytes for auto-detection.`,
25
+ [types_js_1.OfficeErrorType.INVALID_INPUT]: `Invalid input type: Expected a Buffer or a valid file path`,
26
+ [types_js_1.OfficeErrorType.PDF_WORKER_MISSING]: `Missing PDF worker configuration. PDF parsing in browser environments requires a worker source. Please provide "pdfWorkerSrc" in your configuration.`,
27
+ [types_js_1.OfficeErrorType.FEATURE_NOT_SUPPORTED_IN_BROWSER]: (feature) => `'${feature}' is not supported in the browser. Browser users must pass file content as Buffer or ArrayBuffer directly.`,
28
+ [types_js_1.OfficeErrorType.INVALID_STYLE_MAPPING]: (mapping) => `Invalid style mapping string: ${mapping}`,
29
+ [types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
30
+ [types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
31
+ [types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`
49
32
  };
33
+ /**
34
+ * Lookup table for warning messages.
35
+ */
36
+ const WARNING_MESSAGES = {
37
+ [types_js_1.OfficeWarningType.PERFORMANCE_TIP]: (tip) => `⚡️ Performance Tip: ${tip}`,
38
+ [types_js_1.OfficeWarningType.OCR_FAILED]: (name) => `OCR failed for ${name}:`,
39
+ [types_js_1.OfficeWarningType.CHART_DATA_EXTRACTION_FAILED]: (path) => `Failed to extract chart data from ${path}:`,
40
+ [types_js_1.OfficeWarningType.PDF_WORKER_FALLBACK]: `Could not auto-resolve local worker path, falling back to CDN:`,
41
+ [types_js_1.OfficeWarningType.ATTACHMENT_EXTRACTION_FAILED]: `Error extracting embedded attachments:`,
42
+ [types_js_1.OfficeWarningType.PAGE_LOAD_FAILED]: (page) => `Error loading page ${page}:`,
43
+ [types_js_1.OfficeWarningType.DEPENDENCY_LOAD_FAILED]: (dep) => `Failed to load dependency ${dep}:`,
44
+ [types_js_1.OfficeWarningType.IMAGE_EXTRACTION_FAILED]: (context) => `Error extracting images ${context}:`,
45
+ [types_js_1.OfficeWarningType.ANNOTATION_EXTRACTION_FAILED]: (page) => `Error extracting annotations from page ${page}:`,
46
+ [types_js_1.OfficeWarningType.IMAGE_PROCESSING_FAILED]: `Failed to extract from ImageBitmap:`,
47
+ [types_js_1.OfficeWarningType.BROWSER_GENERATION_LIMITATION]: (msg) => msg,
48
+ [types_js_1.OfficeWarningType.SHEET_RANGE_NOT_FOUND]: (range) => `No sheets found matching the range: ${range}`,
49
+ [types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH]: (info) => `File content type mismatch: Detected '${info.detected}' but expected/provided '${info.expected}'. Parsing will proceed with '${info.expected}' as requested.`,
50
+ [types_js_1.OfficeWarningType.EMPTY_CHUNK_GENERATED]: (strategy) => `No chunks generated for document. Check if the document content is compatible with the '${strategy}' strategy.`,
51
+ [types_js_1.OfficeWarningType.WHITESPACE_NODE_SKIPPED]: (nodeType) => `Skipped whitespace-only node of type: ${nodeType}`
52
+ };
53
+ /**
54
+ * Creates a formatted warning message for a specific warning type.
55
+ *
56
+ * @param type - The type of warning
57
+ * @param info - Optional additional information
58
+ * @returns The formatted warning message string
59
+ */
60
+ const getWarningMessage = (type, info) => {
61
+ const msg = WARNING_MESSAGES[type];
62
+ const message = typeof msg === 'function' ? msg(info) : msg;
63
+ return message;
64
+ };
65
+ exports.getWarningMessage = getWarningMessage;
50
66
  /**
51
67
  * Creates a formatted error message for a specific error type.
52
68
  *
@@ -59,19 +75,41 @@ const createOfficeError = (type, info) => {
59
75
  const message = typeof msg === 'function' ? msg(info) : msg;
60
76
  return message;
61
77
  };
78
+ /**
79
+ * Core reporting logic for all issues.
80
+ * Ensures consistent logging and callback execution.
81
+ */
82
+ const reportIssue = (issue, config) => {
83
+ if (config?.onWarning) {
84
+ config.onWarning(issue);
85
+ }
86
+ else if (!config || config.outputErrorToConsole) {
87
+ const formatted = ERRORHEADER + issue.message;
88
+ if (issue.type === 'error') {
89
+ console.error(formatted, issue.details || '');
90
+ }
91
+ else {
92
+ console.warn(formatted, issue.details || '');
93
+ }
94
+ }
95
+ };
62
96
  /**
63
97
  * Creates, optionally logs to console, and returns a formatted OfficeParser error.
64
98
  *
65
99
  * @param type - The type of error
66
- * @param config - Parser configuration (checks outputErrorToConsole)
100
+ * @param config - Optional parser configuration (checks outputErrorToConsole)
67
101
  * @param info - Optional additional information
68
102
  * @returns The Error object to be thrown
69
103
  */
70
104
  const getOfficeError = (type, config, info) => {
71
105
  const message = createOfficeError(type, info);
72
- if (config.outputErrorToConsole) {
73
- console.error(ERRORHEADER + message);
74
- }
106
+ const issue = {
107
+ type: 'error',
108
+ code: type,
109
+ message,
110
+ details: info
111
+ };
112
+ reportIssue(issue, config);
75
113
  return new Error(ERRORHEADER + message);
76
114
  };
77
115
  exports.getOfficeError = getOfficeError;
@@ -86,35 +124,54 @@ exports.getOfficeError = getOfficeError;
86
124
  */
87
125
  const getWrappedError = (error, config, filePath) => {
88
126
  let message = error.message || error;
127
+ let code = types_js_1.OfficeErrorType.FILE_CORRUPTED; // Default for wrapped errors
89
128
  // Detect file corruption from common library error messages
90
129
  if (filePath && (message.includes('end of central directory record') ||
91
130
  message.includes('invalid XML') ||
92
131
  message.includes('Failed to open zip file') ||
93
132
  message.includes('invalid distance too far back'))) {
94
- message = createOfficeError(OfficeErrorType.FILE_CORRUPTED, filePath);
95
- }
96
- if (config.outputErrorToConsole) {
97
- console.error(ERRORHEADER + message);
133
+ message = createOfficeError(types_js_1.OfficeErrorType.FILE_CORRUPTED, filePath);
98
134
  }
135
+ const issue = {
136
+ type: 'error',
137
+ code: types_js_1.OfficeErrorType.FILE_CORRUPTED,
138
+ message,
139
+ details: filePath ? { filePath, originalError: error } : error
140
+ };
141
+ reportIssue(issue, config);
99
142
  return new Error(ERRORHEADER + message);
100
143
  };
101
144
  exports.getWrappedError = getWrappedError;
102
145
  /**
103
- * Conditionally logs a warning message to the console.
104
- * Used for non-fatal errors that shouldn't stop the parsing process.
146
+ * Centralized logging utility for non-fatal warnings or issues.
147
+ * Routes messages to config.onWarning if provided, or console.warn/error
148
+ * if config.outputErrorToConsole is true.
105
149
  *
106
- * @param message - The warning message
107
- * @param config - Parser configuration
108
- * @param error - Optional original error object for more context
150
+ * @param messageOrType - The warning message or warning type
151
+ * @param config - Optional parser configuration
152
+ * @param info - Optional additional information for dynamic messages or context
153
+ * @param error - Optional original error object
109
154
  */
110
- const logWarning = (message, config, error) => {
111
- if (config.outputErrorToConsole) {
112
- if (error) {
113
- console.warn(ERRORHEADER + message, error);
114
- }
115
- else {
116
- console.warn(ERRORHEADER + message);
155
+ const logWarning = (type, config, info, error) => {
156
+ let message;
157
+ let details = info;
158
+ const msg = WARNING_MESSAGES[type];
159
+ if (typeof msg === 'function') {
160
+ message = msg(info);
161
+ details = error || info;
162
+ }
163
+ else {
164
+ message = msg;
165
+ if (info instanceof Error && !error) {
166
+ details = info;
117
167
  }
118
168
  }
169
+ const issue = {
170
+ type: 'warning',
171
+ code: type,
172
+ message,
173
+ details
174
+ };
175
+ reportIssue(issue, config);
119
176
  };
120
177
  exports.logWarning = logWarning;
@@ -12,16 +12,22 @@ Object.defineProperty(exports, "__esModule", { value: true });
12
12
  exports.loadFileType = loadFileType;
13
13
  exports.loadPdfJs = loadPdfJs;
14
14
  const envUtils_js_1 = require("./envUtils.js");
15
- /**
16
- * Dynamically loads an ESM module in a Node.js CJS context.
17
- *
18
- * @param specifier - The module specifier to load
19
- * @returns The loaded module
20
- */
21
15
  async function loadNodeEsmModule(specifier) {
22
- // We use 'new Function' to bypass static analysis of tsc and some bundlers.
23
- // This ensures that Node.js sees a real 'import()' at runtime, even in a CJS file.
24
- return new Function('s', 'return import(s)')(specifier);
16
+ // In Node.js, we resolve the specifier to an absolute file URL.
17
+ // This ensures that the dynamic import() call (executed via new Function)
18
+ // always finds the correct module regardless of the caller's context.
19
+ // This is especially important in Node 18 for sub-paths of packages.
20
+ try {
21
+ const { pathToFileURL } = await import('url');
22
+ // @ts-ignore - require.resolve is available in Node.js
23
+ const absolutePath = require.resolve(specifier);
24
+ const fileUrl = pathToFileURL(absolutePath).href;
25
+ return new Function('s', 'return import(s)')(fileUrl);
26
+ }
27
+ catch (e) {
28
+ // Fallback for cases where require.resolve might fail (e.g. non-file specifiers)
29
+ return new Function('s', 'return import(s)')(specifier);
30
+ }
25
31
  }
26
32
  /**
27
33
  * Specialized loader for file-type
@@ -12,6 +12,7 @@
12
12
  */
13
13
  Object.defineProperty(exports, "__esModule", { value: true });
14
14
  exports.terminateOcr = exports.performOcr = void 0;
15
+ const envUtils_js_1 = require("./envUtils.js");
15
16
  /**
16
17
  * Manages a pool of Tesseract workers with "Smart Affinity".
17
18
  *
@@ -202,7 +203,7 @@ const performOcr = async (image, config) => {
202
203
  let inputImage = image;
203
204
  // In browser environment, convert Buffer to Blob for better compatibility
204
205
  // @ts-ignore
205
- if (typeof window !== 'undefined' && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
206
+ if (envUtils_js_1.isBrowser && typeof Blob !== 'undefined' && Buffer.isBuffer(image)) {
206
207
  inputImage = new Blob([image], { type: 'image/bmp' });
207
208
  }
208
209
  return await OcrSchedulerManager.getInstance().recognize(inputImage, config);
@@ -0,0 +1,7 @@
1
+ /**
2
+ * Parses a range string (e.g., "1", "1-3", "1,2", "1,3-5, 7") into an array of numbers.
3
+ *
4
+ * @param rangeStr - The range string to parse
5
+ * @returns An array of unique, sorted numbers (1-based indices)
6
+ */
7
+ export declare function parseRangeString(rangeStr: string): number[];
@@ -0,0 +1,35 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.parseRangeString = parseRangeString;
4
+ /**
5
+ * Parses a range string (e.g., "1", "1-3", "1,2", "1,3-5, 7") into an array of numbers.
6
+ *
7
+ * @param rangeStr - The range string to parse
8
+ * @returns An array of unique, sorted numbers (1-based indices)
9
+ */
10
+ function parseRangeString(rangeStr) {
11
+ const result = new Set();
12
+ const segments = rangeStr.split(',');
13
+ for (const segment of segments) {
14
+ const trimmed = segment.trim();
15
+ if (trimmed.includes('-')) {
16
+ const [startStr, endStr] = trimmed.split('-');
17
+ const start = parseInt(startStr, 10);
18
+ const end = parseInt(endStr, 10);
19
+ if (!isNaN(start) && !isNaN(end)) {
20
+ const actualStart = Math.min(start, end);
21
+ const actualEnd = Math.max(start, end);
22
+ for (let i = actualStart; i <= actualEnd; i++) {
23
+ result.add(i);
24
+ }
25
+ }
26
+ }
27
+ else {
28
+ const val = parseInt(trimmed, 10);
29
+ if (!isNaN(val)) {
30
+ result.add(val);
31
+ }
32
+ }
33
+ }
34
+ return Array.from(result).sort((a, b) => a - b);
35
+ }
@@ -0,0 +1,36 @@
1
+ import { OfficeContentNode, StructuredStyleMapping } from '../types.js';
2
+ export interface StyleMapping {
3
+ selector: {
4
+ nodeType?: string;
5
+ attributes: Record<string, {
6
+ value: string | number | boolean;
7
+ operator: '=' | '~=';
8
+ compiled?: RegExp;
9
+ }>;
10
+ };
11
+ output: {
12
+ tag: string;
13
+ classes: string[];
14
+ attributes: Record<string, string>;
15
+ fresh: boolean;
16
+ };
17
+ }
18
+ /**
19
+ * Parser and matcher for the style mapping DSL.
20
+ * Supports a structured JSON format and a legacy string DSL.
21
+ */
22
+ export declare class StyleMapper {
23
+ private mappings;
24
+ constructor(mappings?: string[] | StructuredStyleMapping[] | Record<string, any>, ignoreDefaults?: boolean);
25
+ /**
26
+ * Finds the best matching mapping for a node.
27
+ */
28
+ getMapping(node: OfficeContentNode): StyleMapping['output'] | undefined;
29
+ private matches;
30
+ private getNodeAttribute;
31
+ private convertStructuredMapping;
32
+ /**
33
+ * Parses a mapping string like "p[style-name='Heading 1'] => h1.title:fresh"
34
+ */
35
+ private parseMappingString;
36
+ }
@@ -0,0 +1,224 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.StyleMapper = void 0;
4
+ const types_js_1 = require("../types.js");
5
+ const errorUtils_js_1 = require("./errorUtils.js");
6
+ const DEFAULT_MAPPINGS = [
7
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 1' } }, output: { tag: 'h1' } },
8
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 2' } }, output: { tag: 'h2' } },
9
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 3' } }, output: { tag: 'h3' } },
10
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 4' } }, output: { tag: 'h4' } },
11
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 5' } }, output: { tag: 'h5' } },
12
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Heading 6' } }, output: { tag: 'h6' } },
13
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Title' } }, output: { tag: 'h1', classes: ['title'] } },
14
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Subtitle' } }, output: { tag: 'p', classes: ['subtitle'] } },
15
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Quote' } }, output: { tag: 'blockquote' } },
16
+ { selector: { nodeType: 'paragraph', attributes: { 'style-name': 'Intense Quote' } }, output: { tag: 'blockquote', classes: ['intense'] } },
17
+ ];
18
+ /**
19
+ * Parser and matcher for the style mapping DSL.
20
+ * Supports a structured JSON format and a legacy string DSL.
21
+ */
22
+ class StyleMapper {
23
+ mappings = [];
24
+ constructor(mappings, ignoreDefaults = false) {
25
+ // 1. Add user mappings (they take precedence)
26
+ if (mappings) {
27
+ if (Array.isArray(mappings)) {
28
+ for (const m of mappings) {
29
+ if (typeof m === 'string') {
30
+ this.mappings.push(this.parseMappingString(m));
31
+ }
32
+ else {
33
+ this.mappings.push(this.convertStructuredMapping(m));
34
+ }
35
+ }
36
+ }
37
+ else {
38
+ // Support legacy object format: { 'Heading 1': { tag: 'h1', class: 'title' } }
39
+ for (const [styleName, target] of Object.entries(mappings)) {
40
+ this.mappings.push({
41
+ selector: {
42
+ attributes: { style: { value: styleName, operator: '=' } }
43
+ },
44
+ output: {
45
+ tag: target.tag || 'div',
46
+ classes: target.class ? target.class.split(' ') : [],
47
+ attributes: {},
48
+ fresh: false
49
+ }
50
+ });
51
+ }
52
+ }
53
+ }
54
+ // 2. Add default mappings if not ignored
55
+ if (!ignoreDefaults) {
56
+ this.mappings.push(...DEFAULT_MAPPINGS.map(m => this.convertStructuredMapping(m)));
57
+ }
58
+ }
59
+ /**
60
+ * Finds the best matching mapping for a node.
61
+ */
62
+ getMapping(node) {
63
+ for (const mapping of this.mappings) {
64
+ if (this.matches(node, mapping.selector)) {
65
+ return mapping.output;
66
+ }
67
+ }
68
+ return undefined;
69
+ }
70
+ matches(node, selector) {
71
+ // Match node type if specified
72
+ if (selector.nodeType && node.type !== selector.nodeType) {
73
+ return false;
74
+ }
75
+ // Match attributes (style, level, etc.)
76
+ for (const [attr, { value, operator, compiled }] of Object.entries(selector.attributes)) {
77
+ const actualValue = this.getNodeAttribute(node, attr);
78
+ if (actualValue === undefined)
79
+ return false;
80
+ if (operator === '=') {
81
+ if (String(actualValue) !== String(value))
82
+ return false;
83
+ }
84
+ else if (operator === '~=') {
85
+ const regex = compiled || new RegExp(String(value));
86
+ if (!regex.test(String(actualValue)))
87
+ return false;
88
+ }
89
+ }
90
+ return true;
91
+ }
92
+ getNodeAttribute(node, attr) {
93
+ // Special case for style (alias style-name for mammoth.js compatibility)
94
+ if (attr === 'style' || attr === 'style-name') {
95
+ return node.metadata?.style || node.formatting?.font;
96
+ }
97
+ // Metadata attributes
98
+ if (node.metadata && attr in node.metadata) {
99
+ return node.metadata[attr];
100
+ }
101
+ // Formatting attributes
102
+ if (node.formatting && attr in node.formatting) {
103
+ return node.formatting[attr];
104
+ }
105
+ return undefined;
106
+ }
107
+ convertStructuredMapping(m) {
108
+ const attributes = {};
109
+ if (m.selector.attributes) {
110
+ for (const [key, val] of Object.entries(m.selector.attributes)) {
111
+ if (typeof val === 'object' && val !== null && 'value' in val) {
112
+ const operator = val.operator || '=';
113
+ attributes[key] = {
114
+ value: val.value,
115
+ operator,
116
+ compiled: operator === '~=' ? new RegExp(String(val.value)) : undefined
117
+ };
118
+ }
119
+ else {
120
+ attributes[key] = {
121
+ value: val,
122
+ operator: '='
123
+ };
124
+ }
125
+ }
126
+ }
127
+ return {
128
+ selector: {
129
+ nodeType: m.selector.nodeType,
130
+ attributes
131
+ },
132
+ output: {
133
+ tag: m.output.tag,
134
+ classes: m.output.classes || [],
135
+ attributes: m.output.attributes || {},
136
+ fresh: m.output.fresh || false
137
+ }
138
+ };
139
+ }
140
+ /**
141
+ * Parses a mapping string like "p[style-name='Heading 1'] => h1.title:fresh"
142
+ */
143
+ parseMappingString(mapping) {
144
+ const lastIndex = mapping.lastIndexOf('=>');
145
+ if (lastIndex === -1) {
146
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_STYLE_MAPPING, undefined, mapping);
147
+ }
148
+ const selectorStr = mapping.substring(0, lastIndex).trim();
149
+ const outputStr = mapping.substring(lastIndex + 2).trim();
150
+ // Parse Selector
151
+ const selectorMatch = selectorStr.match(/^([a-z]+)?(?:\[(.+?)\])?$/);
152
+ if (!selectorMatch) {
153
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_SELECTOR, undefined, selectorStr);
154
+ }
155
+ const typeMap = {
156
+ 'p': 'paragraph',
157
+ 'h': 'heading',
158
+ 't': 'table',
159
+ 'tr': 'row',
160
+ 'td': 'cell',
161
+ 'li': 'list',
162
+ 'img': 'image'
163
+ };
164
+ const nodeType = selectorMatch[1] ? (typeMap[selectorMatch[1]] || selectorMatch[1]) : undefined;
165
+ const attrStr = selectorMatch[2];
166
+ const attributes = {};
167
+ if (attrStr) {
168
+ // Improved attribute parsing to handle commas inside quotes
169
+ const attrParts = [];
170
+ let currentPart = '';
171
+ let inQuotes = false;
172
+ for (let i = 0; i < attrStr.length; i++) {
173
+ const char = attrStr[i];
174
+ if (char === "'" || char === '"')
175
+ inQuotes = !inQuotes;
176
+ if (char === ',' && !inQuotes) {
177
+ attrParts.push(currentPart.trim());
178
+ currentPart = '';
179
+ }
180
+ else {
181
+ currentPart += char;
182
+ }
183
+ }
184
+ if (currentPart)
185
+ attrParts.push(currentPart.trim());
186
+ for (const part of attrParts) {
187
+ const m = part.match(/^([\w-]+)\s*(=|~=)\s*(?:(["'])(.*?)\3|(.+))$/);
188
+ if (m) {
189
+ const operator = m[2];
190
+ const value = m[4] !== undefined ? m[4] : m[5];
191
+ attributes[m[1]] = {
192
+ operator,
193
+ value,
194
+ compiled: operator === '~=' ? new RegExp(value) : undefined
195
+ };
196
+ }
197
+ }
198
+ }
199
+ // Parse Output
200
+ const outputParts = outputStr.split(':');
201
+ const fresh = outputParts.includes('fresh');
202
+ const mainOutput = outputParts[0];
203
+ const outputMatch = mainOutput.match(/^([a-z0-9]+)?((?:\.[\w-]+)*)(?:\[(.+?)\])?$/);
204
+ if (!outputMatch) {
205
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING, undefined, mainOutput);
206
+ }
207
+ const tag = outputMatch[1] || 'div';
208
+ const classes = outputMatch[2] ? outputMatch[2].split('.').filter(Boolean) : [];
209
+ const outAttrs = {};
210
+ if (outputMatch[3]) {
211
+ const outAttrParts = outputMatch[3].split(',').map(a => a.trim());
212
+ for (const part of outAttrParts) {
213
+ const m = part.match(/^([\w-]+)\s*=\s*(?:(["'])(.*?)\2|(.+))$/);
214
+ if (m)
215
+ outAttrs[m[1]] = m[3] !== undefined ? m[3] : m[4];
216
+ }
217
+ }
218
+ return {
219
+ selector: { nodeType, attributes },
220
+ output: { tag, classes, attributes: outAttrs, fresh }
221
+ };
222
+ }
223
+ }
224
+ exports.StyleMapper = StyleMapper;
@@ -42,14 +42,6 @@ export declare const parseXmlString: (xml: string, options?: {
42
42
  * ```
43
43
  */
44
44
  export declare const getElementsByTagName: (element: Element | Document, tagName: string) => Element[];
45
- /**
46
- * Serializes a DOM Node (Document, Element, etc.) back into an XML string.
47
- * This is cross-platform and works in both Node.js and Browser environments.
48
- *
49
- * @param node - The DOM node to serialize
50
- * @param options - Serialization options
51
- * @returns The XML string representation
52
- */
53
45
  export declare const serializeXml: (node: Node, options?: {
54
46
  preserveWhitespace?: boolean;
55
47
  }) => string;
@@ -72,13 +72,14 @@ exports.getElementsByTagName = getElementsByTagName;
72
72
  * @param options - Serialization options
73
73
  * @returns The XML string representation
74
74
  */
75
+ const serializer = new xmldom_1.XMLSerializer();
75
76
  const serializeXml = (node, options = {}) => {
76
77
  // Note: xmldom's XMLSerializer doesn't natively support a 'pretty' or 'preserve'
77
78
  // flag in a way that matches all user expectations, but it defaults to
78
79
  // preserving structure. Formatting (indentation) is usually handled by the
79
80
  // parser's initial whitespace handling.
80
81
  // @ts-ignore - xmldom's Node is compatible with the global Node interface
81
- return new xmldom_1.XMLSerializer().serializeToString(node);
82
+ return serializer.serializeToString(node);
82
83
  };
83
84
  exports.serializeXml = serializeXml;
84
85
  /**