officeparser 6.1.1 → 7.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +301 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +74 -31
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +828 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +783 -5
- package/dist/types.js +73 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.d.ts +8 -3
- package/dist/utils/envUtils.js +117 -34
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +110 -52
- package/dist/utils/moduleLoader.js +19 -11
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +26 -7
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.BaseGenerator = void 0;
|
|
4
|
+
const configUtils_js_1 = require("../utils/configUtils.js");
|
|
5
|
+
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
6
|
+
const styleMapper_js_1 = require("../utils/styleMapper.js");
|
|
7
|
+
/**
|
|
8
|
+
* Base class for all document generators.
|
|
9
|
+
* Provides common traversal logic and configuration handling.
|
|
10
|
+
*/
|
|
11
|
+
class BaseGenerator {
|
|
12
|
+
destination;
|
|
13
|
+
config;
|
|
14
|
+
ast;
|
|
15
|
+
messages = [];
|
|
16
|
+
styleMapper;
|
|
17
|
+
constructor(destination, ast, config) {
|
|
18
|
+
this.destination = destination;
|
|
19
|
+
this.config = (0, configUtils_js_1.resolveGeneratorConfig)(destination, ast.config, config);
|
|
20
|
+
this.ast = ast;
|
|
21
|
+
this.styleMapper = new styleMapper_js_1.StyleMapper(this.config.styleMap, this.config.ignoreDefaultStyleMap);
|
|
22
|
+
}
|
|
23
|
+
/**
|
|
24
|
+
* Retrieves the semantic mapping for a node, respecting the includeFormatting flag.
|
|
25
|
+
* Per design requirements: Style mapping is bypassed if formatting is disabled.
|
|
26
|
+
*/
|
|
27
|
+
getSemanticMapping(node) {
|
|
28
|
+
if (this.config.includeFormatting === false) {
|
|
29
|
+
return undefined;
|
|
30
|
+
}
|
|
31
|
+
return this.styleMapper.getMapping(node);
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* Centralized logic for handling the onNode callback.
|
|
35
|
+
* Evaluates the callback and returns a result that tells the generator how to proceed.
|
|
36
|
+
*
|
|
37
|
+
* @returns
|
|
38
|
+
* - `string`: Use this as the node's output, skip default processing.
|
|
39
|
+
* - `false`: Skip this node and its subtree.
|
|
40
|
+
* - `void`: Proceed with default processing.
|
|
41
|
+
*/
|
|
42
|
+
async handleOnNode(node) {
|
|
43
|
+
const result = await this.config.onNode(node);
|
|
44
|
+
if (result === false)
|
|
45
|
+
return false;
|
|
46
|
+
if (typeof result === 'string')
|
|
47
|
+
return result;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Recursively processes nodes and builds output.
|
|
51
|
+
*
|
|
52
|
+
* @param node - The current node being processed
|
|
53
|
+
* @param processor - A function that takes a node and its children's output and returns the node's output string.
|
|
54
|
+
* @returns The generated string for this node and its subtree.
|
|
55
|
+
*/
|
|
56
|
+
async processNodeRecursive(node, processor) {
|
|
57
|
+
const override = await this.handleOnNode(node);
|
|
58
|
+
if (override === false)
|
|
59
|
+
return '';
|
|
60
|
+
if (typeof override === 'string')
|
|
61
|
+
return override;
|
|
62
|
+
let childrenOutput = '';
|
|
63
|
+
if (node.children) {
|
|
64
|
+
for (const child of node.children) {
|
|
65
|
+
childrenOutput += await this.processNodeRecursive(child, processor);
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
return await processor(node, childrenOutput);
|
|
69
|
+
}
|
|
70
|
+
/**
|
|
71
|
+
* Helper to generate a unique ID (slug) from text.
|
|
72
|
+
*/
|
|
73
|
+
slugify(text) {
|
|
74
|
+
return text
|
|
75
|
+
.toLowerCase()
|
|
76
|
+
.replace(/[^\w\s-]/g, '')
|
|
77
|
+
.replace(/[\s_-]+/g, '-')
|
|
78
|
+
.replace(/^-+|-+$/g, '');
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Recursively extracts plain text from a node and its children.
|
|
82
|
+
*/
|
|
83
|
+
getNodeText(node) {
|
|
84
|
+
if (node.text)
|
|
85
|
+
return node.text;
|
|
86
|
+
if (node.children) {
|
|
87
|
+
return node.children.map(c => this.getNodeText(c)).join('');
|
|
88
|
+
}
|
|
89
|
+
return '';
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Reports a warning to the user and collects it for the final result.
|
|
93
|
+
*/
|
|
94
|
+
warn(type, info, node) {
|
|
95
|
+
const message = (0, errorUtils_js_1.getWarningMessage)(type, info);
|
|
96
|
+
const issue = {
|
|
97
|
+
type: 'warning',
|
|
98
|
+
code: type,
|
|
99
|
+
message,
|
|
100
|
+
node,
|
|
101
|
+
details: info
|
|
102
|
+
};
|
|
103
|
+
this.messages.push(issue);
|
|
104
|
+
this.config.onWarning(issue);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
exports.BaseGenerator = BaseGenerator;
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
import { ConversionResult, GeneratorConfig, OfficeParserAST } from '../types.js';
|
|
2
|
+
import { BaseGenerator } from './BaseGenerator.js';
|
|
3
|
+
/**
|
|
4
|
+
* Generates a list of OfficeChunk objects from an AST for use in RAG pipelines.
|
|
5
|
+
* Supports three strategies: 'fixed-size', 'document-structure', and 'semantic'.
|
|
6
|
+
*/
|
|
7
|
+
export declare class ChunkingGenerator extends BaseGenerator<'chunks'> {
|
|
8
|
+
/** The resolved chunking config (with defaults applied). */
|
|
9
|
+
private chunkConfig;
|
|
10
|
+
/** Whether the user provided an explicit sentence boundary regex. */
|
|
11
|
+
private isCustomRegex;
|
|
12
|
+
constructor(ast: OfficeParserAST, config?: GeneratorConfig<'chunks'>);
|
|
13
|
+
/**
|
|
14
|
+
* Merges the user's chunking config with the appropriate defaults for the chosen strategy.
|
|
15
|
+
*/
|
|
16
|
+
private resolveChunkingConfig;
|
|
17
|
+
/**
|
|
18
|
+
* Main entry point. Routes to the correct strategy implementation.
|
|
19
|
+
* Note: ConversionResult.value is a JSON string of OfficeChunk[] for the 'chunks' destination.
|
|
20
|
+
*/
|
|
21
|
+
generate(): Promise<ConversionResult<'chunks'>>;
|
|
22
|
+
/**
|
|
23
|
+
* Splits the full document text into fixed-size chunks with optional overlap.
|
|
24
|
+
* Attempts to split on natural separators before hard-cutting.
|
|
25
|
+
*/
|
|
26
|
+
private generateFixedSize;
|
|
27
|
+
/**
|
|
28
|
+
* Recursively tries separators to split text into chunks of at most `chunkSize`,
|
|
29
|
+
* with `chunkOverlap` characters of overlap between consecutive chunks.
|
|
30
|
+
*/
|
|
31
|
+
private splitTextRecursively;
|
|
32
|
+
/**
|
|
33
|
+
* Walks the AST and splits at the designated structural boundaries (slide, page, heading, paragraph).
|
|
34
|
+
*/
|
|
35
|
+
private generateDocumentStructure;
|
|
36
|
+
private processNodeForStructure;
|
|
37
|
+
private isStructuralBoundary;
|
|
38
|
+
/**
|
|
39
|
+
* Handles table chunking with the configured tableSplitStrategy.
|
|
40
|
+
* 'row': keeps header row attached to every chunk.
|
|
41
|
+
* 'flatten': converts table to text and splits normally.
|
|
42
|
+
*/
|
|
43
|
+
private processTableNode;
|
|
44
|
+
/**
|
|
45
|
+
* Renders a list of row nodes as a pipe-separated text string.
|
|
46
|
+
*/
|
|
47
|
+
private renderRowsAsText;
|
|
48
|
+
/**
|
|
49
|
+
* Splits document into semantically coherent chunks using cosine similarity
|
|
50
|
+
* between sentence embeddings. A new chunk begins when similarity drops
|
|
51
|
+
* below `similarityThreshold`.
|
|
52
|
+
*/
|
|
53
|
+
private generateSemantic;
|
|
54
|
+
/**
|
|
55
|
+
* Extracts all text sentences from the AST with their contextual metadata.
|
|
56
|
+
*/
|
|
57
|
+
private extractSentences;
|
|
58
|
+
/**
|
|
59
|
+
* Builds a flat text string from the entire document and a map of
|
|
60
|
+
* character offsets to AST node metadata for position-based metadata lookups.
|
|
61
|
+
*/
|
|
62
|
+
private buildFlatTextWithPositions;
|
|
63
|
+
/**
|
|
64
|
+
* Finds the closest AST metadata for a given character position.
|
|
65
|
+
*/
|
|
66
|
+
private enrichMetadataFromPosition;
|
|
67
|
+
/**
|
|
68
|
+
* Applies final post-processing: strips whitespace, sets sourceType.
|
|
69
|
+
*/
|
|
70
|
+
private finalizeChunks;
|
|
71
|
+
/**
|
|
72
|
+
* Helper to process embeddings in sequential batches to avoid API rate limits and memory issues.
|
|
73
|
+
*/
|
|
74
|
+
private batchEmbeddings;
|
|
75
|
+
private cosineSimilarity;
|
|
76
|
+
private averageEmbeddings;
|
|
77
|
+
/**
|
|
78
|
+
* Robustly splits text into sentences, respecting abbreviations and non-Western punctuation.
|
|
79
|
+
*/
|
|
80
|
+
private splitIntoSentences;
|
|
81
|
+
}
|