@fin.cx/einvoice 5.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist_ts/00_commitinfo_data.d.ts +8 -0
- package/dist_ts/00_commitinfo_data.js +9 -0
- package/dist_ts/classes.decoder.d.ts +40 -0
- package/dist_ts/classes.decoder.js +320 -0
- package/dist_ts/classes.encoder.d.ts +51 -0
- package/dist_ts/classes.encoder.js +293 -0
- package/dist_ts/classes.xinvoice.d.ts +136 -0
- package/dist_ts/classes.xinvoice.js +352 -0
- package/dist_ts/einvoice.d.ts +217 -0
- package/dist_ts/einvoice.js +505 -0
- package/dist_ts/errors.d.ts +121 -0
- package/dist_ts/errors.js +241 -0
- package/dist_ts/formats/base/base.decoder.d.ts +32 -0
- package/dist_ts/formats/base/base.decoder.js +52 -0
- package/dist_ts/formats/base/base.encoder.d.ts +13 -0
- package/dist_ts/formats/base/base.encoder.js +7 -0
- package/dist_ts/formats/base/base.validator.d.ts +44 -0
- package/dist_ts/formats/base/base.validator.js +39 -0
- package/dist_ts/formats/base.decoder.d.ts +19 -0
- package/dist_ts/formats/base.decoder.js +130 -0
- package/dist_ts/formats/base.validator.d.ts +44 -0
- package/dist_ts/formats/base.validator.js +39 -0
- package/dist_ts/formats/cii/cii.decoder.d.ts +61 -0
- package/dist_ts/formats/cii/cii.decoder.js +111 -0
- package/dist_ts/formats/cii/cii.encoder.d.ts +43 -0
- package/dist_ts/formats/cii/cii.encoder.js +48 -0
- package/dist_ts/formats/cii/cii.types.d.ts +31 -0
- package/dist_ts/formats/cii/cii.types.js +40 -0
- package/dist_ts/formats/cii/cii.validator.d.ts +56 -0
- package/dist_ts/formats/cii/cii.validator.js +146 -0
- package/dist_ts/formats/cii/facturx/facturx.decoder.d.ts +43 -0
- package/dist_ts/formats/cii/facturx/facturx.decoder.js +207 -0
- package/dist_ts/formats/cii/facturx/facturx.encoder.d.ts +82 -0
- package/dist_ts/formats/cii/facturx/facturx.encoder.js +400 -0
- package/dist_ts/formats/cii/facturx/facturx.types.d.ts +10 -0
- package/dist_ts/formats/cii/facturx/facturx.types.js +15 -0
- package/dist_ts/formats/cii/facturx/facturx.validator.d.ts +32 -0
- package/dist_ts/formats/cii/facturx/facturx.validator.js +138 -0
- package/dist_ts/formats/cii/zugferd/zugferd.decoder.d.ts +43 -0
- package/dist_ts/formats/cii/zugferd/zugferd.decoder.js +218 -0
- package/dist_ts/formats/cii/zugferd/zugferd.encoder.d.ts +97 -0
- package/dist_ts/formats/cii/zugferd/zugferd.encoder.js +550 -0
- package/dist_ts/formats/cii/zugferd/zugferd.types.d.ts +10 -0
- package/dist_ts/formats/cii/zugferd/zugferd.types.js +15 -0
- package/dist_ts/formats/cii/zugferd/zugferd.v1.decoder.d.ts +49 -0
- package/dist_ts/formats/cii/zugferd/zugferd.v1.decoder.js +229 -0
- package/dist_ts/formats/cii/zugferd/zugferd.validator.d.ts +16 -0
- package/dist_ts/formats/cii/zugferd/zugferd.validator.js +29 -0
- package/dist_ts/formats/decoder.factory.d.ts +15 -0
- package/dist_ts/formats/decoder.factory.js +46 -0
- package/dist_ts/formats/factories/decoder.factory.d.ts +13 -0
- package/dist_ts/formats/factories/decoder.factory.js +56 -0
- package/dist_ts/formats/factories/encoder.factory.d.ts +14 -0
- package/dist_ts/formats/factories/encoder.factory.js +40 -0
- package/dist_ts/formats/factories/validator.factory.d.ts +12 -0
- package/dist_ts/formats/factories/validator.factory.js +110 -0
- package/dist_ts/formats/factorx.decoder.d.ts +18 -0
- package/dist_ts/formats/factorx.decoder.js +178 -0
- package/dist_ts/formats/factorx.encoder.d.ts +51 -0
- package/dist_ts/formats/factorx.encoder.js +293 -0
- package/dist_ts/formats/facturx.decoder.d.ts +18 -0
- package/dist_ts/formats/facturx.decoder.js +210 -0
- package/dist_ts/formats/facturx.encoder.d.ts +51 -0
- package/dist_ts/formats/facturx.encoder.js +293 -0
- package/dist_ts/formats/facturx.validator.d.ts +67 -0
- package/dist_ts/formats/facturx.validator.js +266 -0
- package/dist_ts/formats/pdf/extractors/associated.extractor.d.ts +14 -0
- package/dist_ts/formats/pdf/extractors/associated.extractor.js +68 -0
- package/dist_ts/formats/pdf/extractors/base.extractor.d.ts +92 -0
- package/dist_ts/formats/pdf/extractors/base.extractor.js +319 -0
- package/dist_ts/formats/pdf/extractors/index.d.ts +4 -0
- package/dist_ts/formats/pdf/extractors/index.js +5 -0
- package/dist_ts/formats/pdf/extractors/standard.extractor.d.ts +13 -0
- package/dist_ts/formats/pdf/extractors/standard.extractor.js +74 -0
- package/dist_ts/formats/pdf/extractors/text.extractor.d.ts +45 -0
- package/dist_ts/formats/pdf/extractors/text.extractor.js +151 -0
- package/dist_ts/formats/pdf/pdf.embedder.d.ts +69 -0
- package/dist_ts/formats/pdf/pdf.embedder.js +183 -0
- package/dist_ts/formats/pdf/pdf.extractor.d.ts +49 -0
- package/dist_ts/formats/pdf/pdf.extractor.js +98 -0
- package/dist_ts/formats/pdf/robust-pdf.extractor.d.ts +40 -0
- package/dist_ts/formats/pdf/robust-pdf.extractor.js +324 -0
- package/dist_ts/formats/ubl/en16931.ubl.validator.d.ts +18 -0
- package/dist_ts/formats/ubl/en16931.ubl.validator.js +169 -0
- package/dist_ts/formats/ubl/generic/ubl.encoder.d.ts +151 -0
- package/dist_ts/formats/ubl/generic/ubl.encoder.js +892 -0
- package/dist_ts/formats/ubl/ubl.decoder.d.ts +61 -0
- package/dist_ts/formats/ubl/ubl.decoder.js +95 -0
- package/dist_ts/formats/ubl/ubl.encoder.d.ts +38 -0
- package/dist_ts/formats/ubl/ubl.encoder.js +46 -0
- package/dist_ts/formats/ubl/ubl.types.d.ts +16 -0
- package/dist_ts/formats/ubl/ubl.types.js +21 -0
- package/dist_ts/formats/ubl/ubl.validator.d.ts +50 -0
- package/dist_ts/formats/ubl/ubl.validator.js +112 -0
- package/dist_ts/formats/ubl/xrechnung/xrechnung.decoder.d.ts +34 -0
- package/dist_ts/formats/ubl/xrechnung/xrechnung.decoder.js +424 -0
- package/dist_ts/formats/ubl/xrechnung/xrechnung.encoder.d.ts +75 -0
- package/dist_ts/formats/ubl/xrechnung/xrechnung.encoder.js +555 -0
- package/dist_ts/formats/ubl/xrechnung.validator.d.ts +15 -0
- package/dist_ts/formats/ubl/xrechnung.validator.js +107 -0
- package/dist_ts/formats/ubl.validator.d.ts +82 -0
- package/dist_ts/formats/ubl.validator.js +306 -0
- package/dist_ts/formats/utils/format.detector.d.ts +62 -0
- package/dist_ts/formats/utils/format.detector.js +249 -0
- package/dist_ts/formats/validation/en16931.validator.d.ts +23 -0
- package/dist_ts/formats/validation/en16931.validator.js +119 -0
- package/dist_ts/formats/validator.factory.d.ts +18 -0
- package/dist_ts/formats/validator.factory.js +78 -0
- package/dist_ts/formats/xinvoice.decoder.d.ts +28 -0
- package/dist_ts/formats/xinvoice.decoder.js +332 -0
- package/dist_ts/formats/xinvoice.encoder.d.ts +19 -0
- package/dist_ts/formats/xinvoice.encoder.js +282 -0
- package/dist_ts/index.d.ts +51 -0
- package/dist_ts/index.js +90 -0
- package/dist_ts/interfaces/common.d.ts +79 -0
- package/dist_ts/interfaces/common.js +24 -0
- package/dist_ts/interfaces.d.ts +87 -0
- package/dist_ts/interfaces.js +23 -0
- package/dist_ts/plugins.d.ts +14 -0
- package/dist_ts/plugins.js +31 -0
- package/npmextra.json +35 -0
- package/package.json +71 -0
- package/readme.hints.md +1107 -0
- package/readme.howtofixtests.md +38 -0
- package/readme.literature.md +1 -0
- package/readme.md +998 -0
- package/readme.plan.md +481 -0
- package/ts/00_commitinfo_data.ts +8 -0
- package/ts/einvoice.ts +603 -0
- package/ts/errors.ts +341 -0
- package/ts/formats/base/base.decoder.ts +68 -0
- package/ts/formats/base/base.encoder.ts +14 -0
- package/ts/formats/base/base.validator.ts +64 -0
- package/ts/formats/cii/cii.decoder.ts +139 -0
- package/ts/formats/cii/cii.encoder.ts +64 -0
- package/ts/formats/cii/cii.types.ts +44 -0
- package/ts/formats/cii/cii.validator.ts +171 -0
- package/ts/formats/cii/facturx/facturx.decoder.ts +243 -0
- package/ts/formats/cii/facturx/facturx.encoder.ts +483 -0
- package/ts/formats/cii/facturx/facturx.types.ts +18 -0
- package/ts/formats/cii/facturx/facturx.validator.ts +180 -0
- package/ts/formats/cii/zugferd/zugferd.decoder.ts +256 -0
- package/ts/formats/cii/zugferd/zugferd.encoder.ts +661 -0
- package/ts/formats/cii/zugferd/zugferd.types.ts +18 -0
- package/ts/formats/cii/zugferd/zugferd.v1.decoder.ts +267 -0
- package/ts/formats/cii/zugferd/zugferd.validator.ts +31 -0
- package/ts/formats/factories/decoder.factory.ts +62 -0
- package/ts/formats/factories/encoder.factory.ts +47 -0
- package/ts/formats/factories/validator.factory.ts +134 -0
- package/ts/formats/pdf/extractors/associated.extractor.ts +78 -0
- package/ts/formats/pdf/extractors/base.extractor.ts +355 -0
- package/ts/formats/pdf/extractors/index.ts +4 -0
- package/ts/formats/pdf/extractors/standard.extractor.ts +86 -0
- package/ts/formats/pdf/extractors/text.extractor.ts +162 -0
- package/ts/formats/pdf/pdf.embedder.ts +242 -0
- package/ts/formats/pdf/pdf.extractor.ts +141 -0
- package/ts/formats/ubl/en16931.ubl.validator.ts +216 -0
- package/ts/formats/ubl/generic/ubl.encoder.ts +1041 -0
- package/ts/formats/ubl/ubl.decoder.ts +121 -0
- package/ts/formats/ubl/ubl.encoder.ts +63 -0
- package/ts/formats/ubl/ubl.types.ts +22 -0
- package/ts/formats/ubl/ubl.validator.ts +133 -0
- package/ts/formats/ubl/xrechnung/xrechnung.decoder.ts +471 -0
- package/ts/formats/ubl/xrechnung/xrechnung.encoder.ts +619 -0
- package/ts/formats/ubl/xrechnung.validator.ts +185 -0
- package/ts/formats/utils/format.detector.ts +306 -0
- package/ts/formats/validation/en16931.validator.ts +135 -0
- package/ts/index.ts +164 -0
- package/ts/interfaces/common.ts +90 -0
- package/ts/interfaces.ts +98 -0
- package/ts/plugins.ts +61 -0
|
@@ -0,0 +1,355 @@
|
|
|
1
|
+
import { PDFDocument, PDFDict, PDFName, PDFRawStream, PDFArray, PDFString, pako } from '../../../plugins.js';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Base class for PDF XML extractors with common functionality
|
|
5
|
+
*/
|
|
6
|
+
export abstract class BaseXMLExtractor {
|
|
7
|
+
/**
|
|
8
|
+
* Known XML file names for different invoice formats
|
|
9
|
+
*/
|
|
10
|
+
protected readonly knownFileNames = [
|
|
11
|
+
'factur-x.xml',
|
|
12
|
+
'zugferd-invoice.xml',
|
|
13
|
+
'ZUGFeRD-invoice.xml',
|
|
14
|
+
'xrechnung.xml',
|
|
15
|
+
'ubl-invoice.xml',
|
|
16
|
+
'invoice.xml',
|
|
17
|
+
'metadata.xml'
|
|
18
|
+
];
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Known XML formats to validate extracted content
|
|
22
|
+
*/
|
|
23
|
+
protected readonly knownFormats = [
|
|
24
|
+
'CrossIndustryInvoice',
|
|
25
|
+
'CrossIndustryDocument',
|
|
26
|
+
'Invoice',
|
|
27
|
+
'CreditNote',
|
|
28
|
+
'ubl:Invoice',
|
|
29
|
+
'ubl:CreditNote',
|
|
30
|
+
'rsm:CrossIndustryInvoice',
|
|
31
|
+
'rsm:CrossIndustryDocument',
|
|
32
|
+
'ram:CrossIndustryDocument',
|
|
33
|
+
'urn:un:unece:uncefact',
|
|
34
|
+
'urn:ferd:CrossIndustryDocument',
|
|
35
|
+
'urn:zugferd',
|
|
36
|
+
'urn:factur-x',
|
|
37
|
+
'factur-x.eu',
|
|
38
|
+
'ZUGFeRD',
|
|
39
|
+
'FatturaElettronica'
|
|
40
|
+
];
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Known XML end tags for extracting content from strings
|
|
44
|
+
*/
|
|
45
|
+
protected readonly knownEndTags = [
|
|
46
|
+
'</CrossIndustryInvoice>',
|
|
47
|
+
'</CrossIndustryDocument>',
|
|
48
|
+
'</Invoice>',
|
|
49
|
+
'</CreditNote>',
|
|
50
|
+
'</rsm:CrossIndustryInvoice>',
|
|
51
|
+
'</rsm:CrossIndustryDocument>',
|
|
52
|
+
'</ram:CrossIndustryDocument>',
|
|
53
|
+
'</ubl:Invoice>',
|
|
54
|
+
'</ubl:CreditNote>',
|
|
55
|
+
'</FatturaElettronica>'
|
|
56
|
+
];
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Extract XML from a PDF buffer
|
|
60
|
+
* @param pdfBuffer PDF buffer
|
|
61
|
+
* @returns XML content or null if not found
|
|
62
|
+
*/
|
|
63
|
+
public abstract extractXml(pdfBuffer: Uint8Array | Buffer): Promise<string | null>;
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Check if an XML string is valid
|
|
67
|
+
* @param xmlString XML string to check
|
|
68
|
+
* @returns True if the XML is valid
|
|
69
|
+
*/
|
|
70
|
+
protected isValidXml(xmlString: string): boolean {
|
|
71
|
+
try {
|
|
72
|
+
// Basic checks for XML validity
|
|
73
|
+
if (!xmlString || typeof xmlString !== 'string') {
|
|
74
|
+
return false;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
// Check if it starts with XML declaration or a valid element
|
|
78
|
+
if (!xmlString.includes('<?xml') && !this.hasKnownXmlElement(xmlString)) {
|
|
79
|
+
return false;
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// Check if the XML string contains known invoice formats
|
|
83
|
+
const hasKnownFormat = this.hasKnownFormat(xmlString);
|
|
84
|
+
if (!hasKnownFormat) {
|
|
85
|
+
return false;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// Check if the XML string contains binary data or invalid characters
|
|
89
|
+
if (this.hasBinaryData(xmlString)) {
|
|
90
|
+
return false;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// Check if the XML string is too short
|
|
94
|
+
if (xmlString.length < 100) {
|
|
95
|
+
return false;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// Check if XML has a proper structure (contains both opening and closing tags)
|
|
99
|
+
if (!this.hasProperXmlStructure(xmlString)) {
|
|
100
|
+
return false;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
return true;
|
|
104
|
+
} catch (error) {
|
|
105
|
+
console.error('Error validating XML:', error);
|
|
106
|
+
return false;
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Check if the XML string contains a known element
|
|
112
|
+
* @param xmlString XML string to check
|
|
113
|
+
* @returns True if the XML contains a known element
|
|
114
|
+
*/
|
|
115
|
+
protected hasKnownXmlElement(xmlString: string): boolean {
|
|
116
|
+
for (const format of this.knownFormats) {
|
|
117
|
+
// Check for opening tag of format
|
|
118
|
+
if (xmlString.includes(`<${format}`)) {
|
|
119
|
+
return true;
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
return false;
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Check if the XML string contains a known format
|
|
127
|
+
* @param xmlString XML string to check
|
|
128
|
+
* @returns True if the XML contains a known format
|
|
129
|
+
*/
|
|
130
|
+
protected hasKnownFormat(xmlString: string): boolean {
|
|
131
|
+
for (const format of this.knownFormats) {
|
|
132
|
+
if (xmlString.includes(format)) {
|
|
133
|
+
return true;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
return false;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Check if the XML string has a proper structure
|
|
141
|
+
* @param xmlString XML string to check
|
|
142
|
+
* @returns True if the XML has a proper structure
|
|
143
|
+
*/
|
|
144
|
+
protected hasProperXmlStructure(xmlString: string): boolean {
|
|
145
|
+
// Check for at least one matching opening and closing tag
|
|
146
|
+
for (const endTag of this.knownEndTags) {
|
|
147
|
+
const startTag = endTag.replace('/', '');
|
|
148
|
+
if (xmlString.includes(startTag) && xmlString.includes(endTag)) {
|
|
149
|
+
return true;
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
// If no specific tag is found but it has a basic XML structure
|
|
154
|
+
return (
|
|
155
|
+
(xmlString.includes('<?xml') && xmlString.includes('?>')) ||
|
|
156
|
+
(xmlString.match(/<[^>]+>/) !== null && xmlString.match(/<\/[^>]+>/) !== null)
|
|
157
|
+
);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* Check if the XML string contains binary data
|
|
162
|
+
* @param xmlString XML string to check
|
|
163
|
+
* @returns True if the XML contains binary data
|
|
164
|
+
*/
|
|
165
|
+
protected hasBinaryData(xmlString: string): boolean {
|
|
166
|
+
// Check for common binary data indicators
|
|
167
|
+
const binaryChars = ['\u0000', '\u0001', '\u0002', '\u0003', '\u0004', '\u0005'];
|
|
168
|
+
const consecutiveNulls = '\u0000\u0000\u0000';
|
|
169
|
+
|
|
170
|
+
// Check for control characters that shouldn't be in XML
|
|
171
|
+
if (binaryChars.some(char => xmlString.includes(char))) {
|
|
172
|
+
return true;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
// Check for consecutive null bytes which indicate binary data
|
|
176
|
+
if (xmlString.includes(consecutiveNulls)) {
|
|
177
|
+
return true;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
// Check for high concentration of non-printable characters
|
|
181
|
+
const nonPrintableCount = (xmlString.match(/[\x00-\x08\x0B\x0C\x0E-\x1F]/g) || []).length;
|
|
182
|
+
if (nonPrintableCount > xmlString.length * 0.05) { // More than 5% non-printable
|
|
183
|
+
return true;
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
return false;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Extract XML from a string
|
|
191
|
+
* @param text Text to extract XML from
|
|
192
|
+
* @param startIndex Index to start extraction from
|
|
193
|
+
* @returns XML content or null if not found
|
|
194
|
+
*/
|
|
195
|
+
protected extractXmlFromString(text: string, startIndex: number = 0): string | null {
|
|
196
|
+
try {
|
|
197
|
+
// Find the start of the XML document
|
|
198
|
+
let xmlStartIndex = text.indexOf('<?xml', startIndex);
|
|
199
|
+
|
|
200
|
+
// If no XML declaration, try to find known elements
|
|
201
|
+
if (xmlStartIndex === -1) {
|
|
202
|
+
for (const format of this.knownFormats) {
|
|
203
|
+
const formatStartIndex = text.indexOf(`<${format.split(':').pop()}`, startIndex);
|
|
204
|
+
if (formatStartIndex !== -1) {
|
|
205
|
+
xmlStartIndex = formatStartIndex;
|
|
206
|
+
break;
|
|
207
|
+
}
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
// Still didn't find any start marker
|
|
211
|
+
if (xmlStartIndex === -1) {
|
|
212
|
+
return null;
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
// Try to find the end of the XML document
|
|
217
|
+
let xmlEndIndex = -1;
|
|
218
|
+
for (const endTag of this.knownEndTags) {
|
|
219
|
+
const endIndex = text.indexOf(endTag, xmlStartIndex);
|
|
220
|
+
if (endIndex !== -1) {
|
|
221
|
+
xmlEndIndex = endIndex + endTag.length;
|
|
222
|
+
break;
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
// If no known end tag found, try to use a heuristic approach
|
|
227
|
+
if (xmlEndIndex === -1) {
|
|
228
|
+
// Try to find the last closing tag
|
|
229
|
+
const lastClosingTagMatch = text.slice(xmlStartIndex).match(/<\/[^>]+>(?!.*<\/[^>]+>)/);
|
|
230
|
+
if (lastClosingTagMatch && lastClosingTagMatch.index !== undefined) {
|
|
231
|
+
xmlEndIndex = xmlStartIndex + lastClosingTagMatch.index + lastClosingTagMatch[0].length;
|
|
232
|
+
} else {
|
|
233
|
+
return null;
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
// Extract the XML content
|
|
238
|
+
const xmlContent = text.substring(xmlStartIndex, xmlEndIndex);
|
|
239
|
+
|
|
240
|
+
// Validate the extracted content
|
|
241
|
+
if (this.isValidXml(xmlContent)) {
|
|
242
|
+
return xmlContent;
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
return null;
|
|
246
|
+
} catch (error) {
|
|
247
|
+
console.error('Error extracting XML from string:', error);
|
|
248
|
+
return null;
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
/**
|
|
253
|
+
* Decompress and decode XML content from a PDF stream
|
|
254
|
+
* @param stream PDF stream containing XML data
|
|
255
|
+
* @param fileName Name of the file (for logging)
|
|
256
|
+
* @returns XML content or null if not valid
|
|
257
|
+
*/
|
|
258
|
+
protected async extractXmlFromStream(stream: PDFRawStream, fileName: string): Promise<string | null> {
|
|
259
|
+
try {
|
|
260
|
+
// Get the raw bytes from the stream
|
|
261
|
+
const rawBytes = stream.getContents();
|
|
262
|
+
|
|
263
|
+
// First try without decompression (in case the content is not compressed)
|
|
264
|
+
let xmlContent = this.tryDecodeBuffer(rawBytes);
|
|
265
|
+
if (xmlContent && this.isValidXml(xmlContent)) {
|
|
266
|
+
console.log(`Successfully extracted uncompressed XML from PDF file. File name: ${fileName}`);
|
|
267
|
+
return xmlContent;
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// Try with decompression
|
|
271
|
+
try {
|
|
272
|
+
const decompressedBytes = this.tryDecompress(rawBytes);
|
|
273
|
+
if (decompressedBytes) {
|
|
274
|
+
xmlContent = this.tryDecodeBuffer(decompressedBytes);
|
|
275
|
+
if (xmlContent && this.isValidXml(xmlContent)) {
|
|
276
|
+
console.log(`Successfully extracted decompressed XML from PDF file. File name: ${fileName}`);
|
|
277
|
+
return xmlContent;
|
|
278
|
+
}
|
|
279
|
+
}
|
|
280
|
+
} catch (decompressError) {
|
|
281
|
+
console.log(`Decompression failed for ${fileName}: ${decompressError}`);
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
return null;
|
|
285
|
+
} catch (error) {
|
|
286
|
+
console.error('Error extracting XML from stream:', error);
|
|
287
|
+
return null;
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* Try to decompress a buffer using different methods
|
|
293
|
+
* @param buffer Buffer to decompress
|
|
294
|
+
* @returns Decompressed buffer or null if decompression failed
|
|
295
|
+
*/
|
|
296
|
+
protected tryDecompress(buffer: Uint8Array): Uint8Array | null {
|
|
297
|
+
try {
|
|
298
|
+
// Try pako inflate (for deflate/zlib compression)
|
|
299
|
+
return pako.inflate(buffer);
|
|
300
|
+
} catch (error) {
|
|
301
|
+
// If pako fails, try other methods if needed
|
|
302
|
+
console.warn('Pako decompression failed, might be uncompressed or using a different algorithm');
|
|
303
|
+
return null;
|
|
304
|
+
}
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* Try to decode a buffer to a string using different encodings
|
|
309
|
+
* @param buffer Buffer to decode
|
|
310
|
+
* @returns Decoded string or null if decoding failed
|
|
311
|
+
*/
|
|
312
|
+
protected tryDecodeBuffer(buffer: Uint8Array): string | null {
|
|
313
|
+
try {
|
|
314
|
+
// Try UTF-8 first
|
|
315
|
+
let content = new TextDecoder('utf-8').decode(buffer);
|
|
316
|
+
if (this.isPlausibleXml(content)) {
|
|
317
|
+
return content;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
// Try ISO-8859-1 (Latin1)
|
|
321
|
+
content = this.decodeLatin1(buffer);
|
|
322
|
+
if (this.isPlausibleXml(content)) {
|
|
323
|
+
return content;
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
return null;
|
|
327
|
+
} catch (error) {
|
|
328
|
+
console.warn('Error decoding buffer:', error);
|
|
329
|
+
return null;
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
/**
|
|
334
|
+
* Decode a buffer using ISO-8859-1 (Latin1) encoding
|
|
335
|
+
* @param buffer Buffer to decode
|
|
336
|
+
* @returns Decoded string
|
|
337
|
+
*/
|
|
338
|
+
protected decodeLatin1(buffer: Uint8Array): string {
|
|
339
|
+
return Array.from(buffer)
|
|
340
|
+
.map(byte => String.fromCharCode(byte))
|
|
341
|
+
.join('');
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
/**
|
|
345
|
+
* Check if a string is plausibly XML (quick check before validation)
|
|
346
|
+
* @param content String to check
|
|
347
|
+
* @returns True if the string is plausibly XML
|
|
348
|
+
*/
|
|
349
|
+
protected isPlausibleXml(content: string): boolean {
|
|
350
|
+
return content.includes('<') &&
|
|
351
|
+
content.includes('>') &&
|
|
352
|
+
(content.includes('<?xml') ||
|
|
353
|
+
this.knownFormats.some(format => content.includes(format)));
|
|
354
|
+
}
|
|
355
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import { PDFDocument, PDFDict, PDFName, PDFRawStream, PDFArray, PDFString } from '../../../plugins.js';
|
|
2
|
+
import { BaseXMLExtractor } from './base.extractor.js';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Standard PDF XML extractor that extracts XML from embedded files
|
|
6
|
+
* Works with PDF/A-3 documents that follow the standard for embedding files
|
|
7
|
+
*/
|
|
8
|
+
export class StandardXMLExtractor extends BaseXMLExtractor {
|
|
9
|
+
/**
|
|
10
|
+
* Extract XML from a PDF buffer using standard PDF/A-3 embedded files
|
|
11
|
+
* @param pdfBuffer PDF buffer
|
|
12
|
+
* @returns XML content or null if not found
|
|
13
|
+
*/
|
|
14
|
+
public async extractXml(pdfBuffer: Uint8Array | Buffer): Promise<string | null> {
|
|
15
|
+
try {
|
|
16
|
+
const pdfDoc = await PDFDocument.load(pdfBuffer);
|
|
17
|
+
|
|
18
|
+
// Get the document's metadata dictionary
|
|
19
|
+
const namesDictObj = pdfDoc.catalog.lookup(PDFName.of('Names'));
|
|
20
|
+
if (!(namesDictObj instanceof PDFDict)) {
|
|
21
|
+
console.warn('No Names dictionary found in PDF! This PDF does not contain embedded files.');
|
|
22
|
+
return null;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
// Get the embedded files dictionary
|
|
26
|
+
const embeddedFilesDictObj = namesDictObj.lookup(PDFName.of('EmbeddedFiles'));
|
|
27
|
+
if (!(embeddedFilesDictObj instanceof PDFDict)) {
|
|
28
|
+
console.warn('No EmbeddedFiles dictionary found! This PDF does not contain embedded files.');
|
|
29
|
+
return null;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// Get the names array
|
|
33
|
+
const filesSpecObj = embeddedFilesDictObj.lookup(PDFName.of('Names'));
|
|
34
|
+
if (!(filesSpecObj instanceof PDFArray)) {
|
|
35
|
+
console.warn('No files specified in EmbeddedFiles dictionary!');
|
|
36
|
+
return null;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
// Try to find an XML file in the embedded files
|
|
40
|
+
for (let i = 0; i < filesSpecObj.size(); i += 2) {
|
|
41
|
+
const fileNameObj = filesSpecObj.lookup(i);
|
|
42
|
+
const fileSpecObj = filesSpecObj.lookup(i + 1);
|
|
43
|
+
|
|
44
|
+
if (!(fileNameObj instanceof PDFString) || !(fileSpecObj instanceof PDFDict)) {
|
|
45
|
+
continue;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// Get the filename as string
|
|
49
|
+
const fileName = fileNameObj.decodeText();
|
|
50
|
+
|
|
51
|
+
// Check if it's a known invoice XML file name
|
|
52
|
+
const isKnownFileName = this.knownFileNames.some(
|
|
53
|
+
knownName => fileName.toLowerCase() === knownName.toLowerCase()
|
|
54
|
+
);
|
|
55
|
+
|
|
56
|
+
// Check if it's any XML file or has invoice-related keywords
|
|
57
|
+
const isXmlFile = fileName.toLowerCase().endsWith('.xml') ||
|
|
58
|
+
fileName.toLowerCase().includes('zugferd') ||
|
|
59
|
+
fileName.toLowerCase().includes('factur-x') ||
|
|
60
|
+
fileName.toLowerCase().includes('xrechnung') ||
|
|
61
|
+
fileName.toLowerCase().includes('invoice');
|
|
62
|
+
|
|
63
|
+
if (isKnownFileName || isXmlFile) {
|
|
64
|
+
const efDictObj = fileSpecObj.lookup(PDFName.of('EF'));
|
|
65
|
+
if (!(efDictObj instanceof PDFDict)) {
|
|
66
|
+
continue;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
const fileStream = efDictObj.lookup(PDFName.of('F'));
|
|
70
|
+
if (fileStream instanceof PDFRawStream) {
|
|
71
|
+
const xmlContent = await this.extractXmlFromStream(fileStream, fileName);
|
|
72
|
+
if (xmlContent) {
|
|
73
|
+
return xmlContent;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
console.warn('No valid XML found in embedded files');
|
|
80
|
+
return null;
|
|
81
|
+
} catch (error) {
|
|
82
|
+
console.error('Error in standard extraction:', error);
|
|
83
|
+
return null;
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
import { BaseXMLExtractor } from './base.extractor.js';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Text-based XML extractor for PDF documents
|
|
5
|
+
* Extracts XML by searching for XML patterns in the PDF text
|
|
6
|
+
* Used as a fallback when other extraction methods fail
|
|
7
|
+
*/
|
|
8
|
+
export class TextXMLExtractor extends BaseXMLExtractor {
|
|
9
|
+
// Maximum chunk size to process at once (4MB)
|
|
10
|
+
private readonly CHUNK_SIZE = 4 * 1024 * 1024;
|
|
11
|
+
|
|
12
|
+
// Maximum number of chunks to check (effective 20MB search limit)
|
|
13
|
+
private readonly MAX_CHUNKS = 5;
|
|
14
|
+
|
|
15
|
+
// Common XML patterns to look for
|
|
16
|
+
private readonly XML_PATTERNS = [
|
|
17
|
+
'<?xml',
|
|
18
|
+
'<CrossIndustryInvoice',
|
|
19
|
+
'<CrossIndustryDocument',
|
|
20
|
+
'<Invoice',
|
|
21
|
+
'<CreditNote',
|
|
22
|
+
'<rsm:CrossIndustryInvoice',
|
|
23
|
+
'<rsm:CrossIndustryDocument',
|
|
24
|
+
'<ram:CrossIndustryDocument',
|
|
25
|
+
'<ubl:Invoice',
|
|
26
|
+
'<ubl:CreditNote',
|
|
27
|
+
'<FatturaElettronica'
|
|
28
|
+
];
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Extract XML from a PDF buffer by searching for XML patterns in the text
|
|
32
|
+
* Uses a chunked approach to handle large files efficiently
|
|
33
|
+
* @param pdfBuffer PDF buffer
|
|
34
|
+
* @returns XML content or null if not found
|
|
35
|
+
*/
|
|
36
|
+
public async extractXml(pdfBuffer: Uint8Array | Buffer): Promise<string | null> {
|
|
37
|
+
try {
|
|
38
|
+
console.log('Attempting text-based XML extraction from PDF...');
|
|
39
|
+
|
|
40
|
+
// Convert Buffer to Uint8Array if needed
|
|
41
|
+
const buffer = Buffer.isBuffer(pdfBuffer) ? new Uint8Array(pdfBuffer) : pdfBuffer;
|
|
42
|
+
|
|
43
|
+
// Try extracting XML using the chunked approach
|
|
44
|
+
return this.extractXmlFromBufferChunked(buffer);
|
|
45
|
+
} catch (error) {
|
|
46
|
+
console.error('Error in text-based extraction:', error);
|
|
47
|
+
return null;
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Extract XML from buffer using a chunked approach
|
|
53
|
+
* This helps avoid memory issues with large PDFs
|
|
54
|
+
* @param buffer Buffer to search in
|
|
55
|
+
* @returns XML content or null if not found
|
|
56
|
+
*/
|
|
57
|
+
private extractXmlFromBufferChunked(buffer: Uint8Array): string | null {
|
|
58
|
+
// Process the PDF in chunks
|
|
59
|
+
for (let chunkIndex = 0; chunkIndex < this.MAX_CHUNKS; chunkIndex++) {
|
|
60
|
+
const startPos = chunkIndex * this.CHUNK_SIZE;
|
|
61
|
+
if (startPos >= buffer.length) break;
|
|
62
|
+
|
|
63
|
+
const endPos = Math.min(startPos + this.CHUNK_SIZE, buffer.length);
|
|
64
|
+
const chunk = buffer.slice(startPos, endPos);
|
|
65
|
+
|
|
66
|
+
// Try to extract XML from this chunk
|
|
67
|
+
const chunkResult = this.processChunk(chunk, startPos);
|
|
68
|
+
if (chunkResult) {
|
|
69
|
+
return chunkResult;
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
console.warn('No valid XML found in any chunk of the PDF');
|
|
74
|
+
return null;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Process a single chunk of the PDF buffer
|
|
79
|
+
* @param chunk Chunk buffer to process
|
|
80
|
+
* @param chunkOffset Offset position of the chunk in the original buffer
|
|
81
|
+
* @returns XML content or null if not found
|
|
82
|
+
*/
|
|
83
|
+
private processChunk(chunk: Uint8Array, chunkOffset: number): string | null {
|
|
84
|
+
try {
|
|
85
|
+
// First try UTF-8 encoding for this chunk
|
|
86
|
+
const utf8String = this.decodeBufferToString(chunk, 'utf-8');
|
|
87
|
+
let xmlContent = this.searchForXmlInString(utf8String);
|
|
88
|
+
|
|
89
|
+
if (xmlContent) {
|
|
90
|
+
console.log(`Found XML content in chunk at offset ${chunkOffset} using UTF-8 encoding`);
|
|
91
|
+
return xmlContent;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// If UTF-8 fails, try Latin-1 (ISO-8859-1) which can handle binary better
|
|
95
|
+
const latin1String = this.decodeBufferToString(chunk, 'latin1');
|
|
96
|
+
xmlContent = this.searchForXmlInString(latin1String);
|
|
97
|
+
|
|
98
|
+
if (xmlContent) {
|
|
99
|
+
console.log(`Found XML content in chunk at offset ${chunkOffset} using Latin-1 encoding`);
|
|
100
|
+
return xmlContent;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
// No XML found in this chunk
|
|
104
|
+
return null;
|
|
105
|
+
} catch (error) {
|
|
106
|
+
console.warn(`Error processing chunk at offset ${chunkOffset}:`, error);
|
|
107
|
+
return null;
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Safely decode a buffer to string using the specified encoding
|
|
113
|
+
* @param buffer Buffer to decode
|
|
114
|
+
* @param encoding Encoding to use ('utf-8' or 'latin1')
|
|
115
|
+
* @returns Decoded string
|
|
116
|
+
*/
|
|
117
|
+
private decodeBufferToString(buffer: Uint8Array, encoding: 'utf-8' | 'latin1'): string {
|
|
118
|
+
try {
|
|
119
|
+
if (encoding === 'utf-8') {
|
|
120
|
+
return new TextDecoder('utf-8', { fatal: false }).decode(buffer);
|
|
121
|
+
} else {
|
|
122
|
+
// For Latin-1 we can use a direct mapping (bytes 0-255 map directly to code points 0-255)
|
|
123
|
+
// This is more reliable for binary data than TextDecoder for legacy encodings
|
|
124
|
+
return Array.from(buffer)
|
|
125
|
+
.map(byte => String.fromCharCode(byte))
|
|
126
|
+
.join('');
|
|
127
|
+
}
|
|
128
|
+
} catch (error) {
|
|
129
|
+
console.warn(`Error decoding buffer using ${encoding}:`, error);
|
|
130
|
+
// Return empty string on error to allow processing to continue
|
|
131
|
+
return '';
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Search for XML patterns in a string
|
|
137
|
+
* @param content String to search in
|
|
138
|
+
* @returns XML content or null if not found
|
|
139
|
+
*/
|
|
140
|
+
private searchForXmlInString(content: string): string | null {
|
|
141
|
+
if (!content) return null;
|
|
142
|
+
|
|
143
|
+
// Search for each XML pattern
|
|
144
|
+
for (const pattern of this.XML_PATTERNS) {
|
|
145
|
+
const patternIndex = content.indexOf(pattern);
|
|
146
|
+
if (patternIndex !== -1) {
|
|
147
|
+
console.log(`Found XML pattern "${pattern}" at position ${patternIndex}`);
|
|
148
|
+
|
|
149
|
+
// Try to extract the XML content starting from the pattern position
|
|
150
|
+
const xmlContent = this.extractXmlFromString(content, patternIndex);
|
|
151
|
+
|
|
152
|
+
// Validate the extracted content
|
|
153
|
+
if (xmlContent && this.isValidXml(xmlContent)) {
|
|
154
|
+
console.log('Successfully extracted and validated XML from text');
|
|
155
|
+
return xmlContent;
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
return null;
|
|
161
|
+
}
|
|
162
|
+
}
|