@fin.cx/einvoice 5.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (171) hide show
  1. package/dist_ts/00_commitinfo_data.d.ts +8 -0
  2. package/dist_ts/00_commitinfo_data.js +9 -0
  3. package/dist_ts/classes.decoder.d.ts +40 -0
  4. package/dist_ts/classes.decoder.js +320 -0
  5. package/dist_ts/classes.encoder.d.ts +51 -0
  6. package/dist_ts/classes.encoder.js +293 -0
  7. package/dist_ts/classes.xinvoice.d.ts +136 -0
  8. package/dist_ts/classes.xinvoice.js +352 -0
  9. package/dist_ts/einvoice.d.ts +217 -0
  10. package/dist_ts/einvoice.js +505 -0
  11. package/dist_ts/errors.d.ts +121 -0
  12. package/dist_ts/errors.js +241 -0
  13. package/dist_ts/formats/base/base.decoder.d.ts +32 -0
  14. package/dist_ts/formats/base/base.decoder.js +52 -0
  15. package/dist_ts/formats/base/base.encoder.d.ts +13 -0
  16. package/dist_ts/formats/base/base.encoder.js +7 -0
  17. package/dist_ts/formats/base/base.validator.d.ts +44 -0
  18. package/dist_ts/formats/base/base.validator.js +39 -0
  19. package/dist_ts/formats/base.decoder.d.ts +19 -0
  20. package/dist_ts/formats/base.decoder.js +130 -0
  21. package/dist_ts/formats/base.validator.d.ts +44 -0
  22. package/dist_ts/formats/base.validator.js +39 -0
  23. package/dist_ts/formats/cii/cii.decoder.d.ts +61 -0
  24. package/dist_ts/formats/cii/cii.decoder.js +111 -0
  25. package/dist_ts/formats/cii/cii.encoder.d.ts +43 -0
  26. package/dist_ts/formats/cii/cii.encoder.js +48 -0
  27. package/dist_ts/formats/cii/cii.types.d.ts +31 -0
  28. package/dist_ts/formats/cii/cii.types.js +40 -0
  29. package/dist_ts/formats/cii/cii.validator.d.ts +56 -0
  30. package/dist_ts/formats/cii/cii.validator.js +146 -0
  31. package/dist_ts/formats/cii/facturx/facturx.decoder.d.ts +43 -0
  32. package/dist_ts/formats/cii/facturx/facturx.decoder.js +207 -0
  33. package/dist_ts/formats/cii/facturx/facturx.encoder.d.ts +82 -0
  34. package/dist_ts/formats/cii/facturx/facturx.encoder.js +400 -0
  35. package/dist_ts/formats/cii/facturx/facturx.types.d.ts +10 -0
  36. package/dist_ts/formats/cii/facturx/facturx.types.js +15 -0
  37. package/dist_ts/formats/cii/facturx/facturx.validator.d.ts +32 -0
  38. package/dist_ts/formats/cii/facturx/facturx.validator.js +138 -0
  39. package/dist_ts/formats/cii/zugferd/zugferd.decoder.d.ts +43 -0
  40. package/dist_ts/formats/cii/zugferd/zugferd.decoder.js +218 -0
  41. package/dist_ts/formats/cii/zugferd/zugferd.encoder.d.ts +97 -0
  42. package/dist_ts/formats/cii/zugferd/zugferd.encoder.js +550 -0
  43. package/dist_ts/formats/cii/zugferd/zugferd.types.d.ts +10 -0
  44. package/dist_ts/formats/cii/zugferd/zugferd.types.js +15 -0
  45. package/dist_ts/formats/cii/zugferd/zugferd.v1.decoder.d.ts +49 -0
  46. package/dist_ts/formats/cii/zugferd/zugferd.v1.decoder.js +229 -0
  47. package/dist_ts/formats/cii/zugferd/zugferd.validator.d.ts +16 -0
  48. package/dist_ts/formats/cii/zugferd/zugferd.validator.js +29 -0
  49. package/dist_ts/formats/decoder.factory.d.ts +15 -0
  50. package/dist_ts/formats/decoder.factory.js +46 -0
  51. package/dist_ts/formats/factories/decoder.factory.d.ts +13 -0
  52. package/dist_ts/formats/factories/decoder.factory.js +56 -0
  53. package/dist_ts/formats/factories/encoder.factory.d.ts +14 -0
  54. package/dist_ts/formats/factories/encoder.factory.js +40 -0
  55. package/dist_ts/formats/factories/validator.factory.d.ts +12 -0
  56. package/dist_ts/formats/factories/validator.factory.js +110 -0
  57. package/dist_ts/formats/factorx.decoder.d.ts +18 -0
  58. package/dist_ts/formats/factorx.decoder.js +178 -0
  59. package/dist_ts/formats/factorx.encoder.d.ts +51 -0
  60. package/dist_ts/formats/factorx.encoder.js +293 -0
  61. package/dist_ts/formats/facturx.decoder.d.ts +18 -0
  62. package/dist_ts/formats/facturx.decoder.js +210 -0
  63. package/dist_ts/formats/facturx.encoder.d.ts +51 -0
  64. package/dist_ts/formats/facturx.encoder.js +293 -0
  65. package/dist_ts/formats/facturx.validator.d.ts +67 -0
  66. package/dist_ts/formats/facturx.validator.js +266 -0
  67. package/dist_ts/formats/pdf/extractors/associated.extractor.d.ts +14 -0
  68. package/dist_ts/formats/pdf/extractors/associated.extractor.js +68 -0
  69. package/dist_ts/formats/pdf/extractors/base.extractor.d.ts +92 -0
  70. package/dist_ts/formats/pdf/extractors/base.extractor.js +319 -0
  71. package/dist_ts/formats/pdf/extractors/index.d.ts +4 -0
  72. package/dist_ts/formats/pdf/extractors/index.js +5 -0
  73. package/dist_ts/formats/pdf/extractors/standard.extractor.d.ts +13 -0
  74. package/dist_ts/formats/pdf/extractors/standard.extractor.js +74 -0
  75. package/dist_ts/formats/pdf/extractors/text.extractor.d.ts +45 -0
  76. package/dist_ts/formats/pdf/extractors/text.extractor.js +151 -0
  77. package/dist_ts/formats/pdf/pdf.embedder.d.ts +69 -0
  78. package/dist_ts/formats/pdf/pdf.embedder.js +183 -0
  79. package/dist_ts/formats/pdf/pdf.extractor.d.ts +49 -0
  80. package/dist_ts/formats/pdf/pdf.extractor.js +98 -0
  81. package/dist_ts/formats/pdf/robust-pdf.extractor.d.ts +40 -0
  82. package/dist_ts/formats/pdf/robust-pdf.extractor.js +324 -0
  83. package/dist_ts/formats/ubl/en16931.ubl.validator.d.ts +18 -0
  84. package/dist_ts/formats/ubl/en16931.ubl.validator.js +169 -0
  85. package/dist_ts/formats/ubl/generic/ubl.encoder.d.ts +151 -0
  86. package/dist_ts/formats/ubl/generic/ubl.encoder.js +892 -0
  87. package/dist_ts/formats/ubl/ubl.decoder.d.ts +61 -0
  88. package/dist_ts/formats/ubl/ubl.decoder.js +95 -0
  89. package/dist_ts/formats/ubl/ubl.encoder.d.ts +38 -0
  90. package/dist_ts/formats/ubl/ubl.encoder.js +46 -0
  91. package/dist_ts/formats/ubl/ubl.types.d.ts +16 -0
  92. package/dist_ts/formats/ubl/ubl.types.js +21 -0
  93. package/dist_ts/formats/ubl/ubl.validator.d.ts +50 -0
  94. package/dist_ts/formats/ubl/ubl.validator.js +112 -0
  95. package/dist_ts/formats/ubl/xrechnung/xrechnung.decoder.d.ts +34 -0
  96. package/dist_ts/formats/ubl/xrechnung/xrechnung.decoder.js +424 -0
  97. package/dist_ts/formats/ubl/xrechnung/xrechnung.encoder.d.ts +75 -0
  98. package/dist_ts/formats/ubl/xrechnung/xrechnung.encoder.js +555 -0
  99. package/dist_ts/formats/ubl/xrechnung.validator.d.ts +15 -0
  100. package/dist_ts/formats/ubl/xrechnung.validator.js +107 -0
  101. package/dist_ts/formats/ubl.validator.d.ts +82 -0
  102. package/dist_ts/formats/ubl.validator.js +306 -0
  103. package/dist_ts/formats/utils/format.detector.d.ts +62 -0
  104. package/dist_ts/formats/utils/format.detector.js +249 -0
  105. package/dist_ts/formats/validation/en16931.validator.d.ts +23 -0
  106. package/dist_ts/formats/validation/en16931.validator.js +119 -0
  107. package/dist_ts/formats/validator.factory.d.ts +18 -0
  108. package/dist_ts/formats/validator.factory.js +78 -0
  109. package/dist_ts/formats/xinvoice.decoder.d.ts +28 -0
  110. package/dist_ts/formats/xinvoice.decoder.js +332 -0
  111. package/dist_ts/formats/xinvoice.encoder.d.ts +19 -0
  112. package/dist_ts/formats/xinvoice.encoder.js +282 -0
  113. package/dist_ts/index.d.ts +51 -0
  114. package/dist_ts/index.js +90 -0
  115. package/dist_ts/interfaces/common.d.ts +79 -0
  116. package/dist_ts/interfaces/common.js +24 -0
  117. package/dist_ts/interfaces.d.ts +87 -0
  118. package/dist_ts/interfaces.js +23 -0
  119. package/dist_ts/plugins.d.ts +14 -0
  120. package/dist_ts/plugins.js +31 -0
  121. package/npmextra.json +35 -0
  122. package/package.json +71 -0
  123. package/readme.hints.md +1107 -0
  124. package/readme.howtofixtests.md +38 -0
  125. package/readme.literature.md +1 -0
  126. package/readme.md +998 -0
  127. package/readme.plan.md +481 -0
  128. package/ts/00_commitinfo_data.ts +8 -0
  129. package/ts/einvoice.ts +603 -0
  130. package/ts/errors.ts +341 -0
  131. package/ts/formats/base/base.decoder.ts +68 -0
  132. package/ts/formats/base/base.encoder.ts +14 -0
  133. package/ts/formats/base/base.validator.ts +64 -0
  134. package/ts/formats/cii/cii.decoder.ts +139 -0
  135. package/ts/formats/cii/cii.encoder.ts +64 -0
  136. package/ts/formats/cii/cii.types.ts +44 -0
  137. package/ts/formats/cii/cii.validator.ts +171 -0
  138. package/ts/formats/cii/facturx/facturx.decoder.ts +243 -0
  139. package/ts/formats/cii/facturx/facturx.encoder.ts +483 -0
  140. package/ts/formats/cii/facturx/facturx.types.ts +18 -0
  141. package/ts/formats/cii/facturx/facturx.validator.ts +180 -0
  142. package/ts/formats/cii/zugferd/zugferd.decoder.ts +256 -0
  143. package/ts/formats/cii/zugferd/zugferd.encoder.ts +661 -0
  144. package/ts/formats/cii/zugferd/zugferd.types.ts +18 -0
  145. package/ts/formats/cii/zugferd/zugferd.v1.decoder.ts +267 -0
  146. package/ts/formats/cii/zugferd/zugferd.validator.ts +31 -0
  147. package/ts/formats/factories/decoder.factory.ts +62 -0
  148. package/ts/formats/factories/encoder.factory.ts +47 -0
  149. package/ts/formats/factories/validator.factory.ts +134 -0
  150. package/ts/formats/pdf/extractors/associated.extractor.ts +78 -0
  151. package/ts/formats/pdf/extractors/base.extractor.ts +355 -0
  152. package/ts/formats/pdf/extractors/index.ts +4 -0
  153. package/ts/formats/pdf/extractors/standard.extractor.ts +86 -0
  154. package/ts/formats/pdf/extractors/text.extractor.ts +162 -0
  155. package/ts/formats/pdf/pdf.embedder.ts +242 -0
  156. package/ts/formats/pdf/pdf.extractor.ts +141 -0
  157. package/ts/formats/ubl/en16931.ubl.validator.ts +216 -0
  158. package/ts/formats/ubl/generic/ubl.encoder.ts +1041 -0
  159. package/ts/formats/ubl/ubl.decoder.ts +121 -0
  160. package/ts/formats/ubl/ubl.encoder.ts +63 -0
  161. package/ts/formats/ubl/ubl.types.ts +22 -0
  162. package/ts/formats/ubl/ubl.validator.ts +133 -0
  163. package/ts/formats/ubl/xrechnung/xrechnung.decoder.ts +471 -0
  164. package/ts/formats/ubl/xrechnung/xrechnung.encoder.ts +619 -0
  165. package/ts/formats/ubl/xrechnung.validator.ts +185 -0
  166. package/ts/formats/utils/format.detector.ts +306 -0
  167. package/ts/formats/validation/en16931.validator.ts +135 -0
  168. package/ts/index.ts +164 -0
  169. package/ts/interfaces/common.ts +90 -0
  170. package/ts/interfaces.ts +98 -0
  171. package/ts/plugins.ts +61 -0
@@ -0,0 +1,355 @@
1
+ import { PDFDocument, PDFDict, PDFName, PDFRawStream, PDFArray, PDFString, pako } from '../../../plugins.js';
2
+
3
+ /**
4
+ * Base class for PDF XML extractors with common functionality
5
+ */
6
+ export abstract class BaseXMLExtractor {
7
+ /**
8
+ * Known XML file names for different invoice formats
9
+ */
10
+ protected readonly knownFileNames = [
11
+ 'factur-x.xml',
12
+ 'zugferd-invoice.xml',
13
+ 'ZUGFeRD-invoice.xml',
14
+ 'xrechnung.xml',
15
+ 'ubl-invoice.xml',
16
+ 'invoice.xml',
17
+ 'metadata.xml'
18
+ ];
19
+
20
+ /**
21
+ * Known XML formats to validate extracted content
22
+ */
23
+ protected readonly knownFormats = [
24
+ 'CrossIndustryInvoice',
25
+ 'CrossIndustryDocument',
26
+ 'Invoice',
27
+ 'CreditNote',
28
+ 'ubl:Invoice',
29
+ 'ubl:CreditNote',
30
+ 'rsm:CrossIndustryInvoice',
31
+ 'rsm:CrossIndustryDocument',
32
+ 'ram:CrossIndustryDocument',
33
+ 'urn:un:unece:uncefact',
34
+ 'urn:ferd:CrossIndustryDocument',
35
+ 'urn:zugferd',
36
+ 'urn:factur-x',
37
+ 'factur-x.eu',
38
+ 'ZUGFeRD',
39
+ 'FatturaElettronica'
40
+ ];
41
+
42
+ /**
43
+ * Known XML end tags for extracting content from strings
44
+ */
45
+ protected readonly knownEndTags = [
46
+ '</CrossIndustryInvoice>',
47
+ '</CrossIndustryDocument>',
48
+ '</Invoice>',
49
+ '</CreditNote>',
50
+ '</rsm:CrossIndustryInvoice>',
51
+ '</rsm:CrossIndustryDocument>',
52
+ '</ram:CrossIndustryDocument>',
53
+ '</ubl:Invoice>',
54
+ '</ubl:CreditNote>',
55
+ '</FatturaElettronica>'
56
+ ];
57
+
58
+ /**
59
+ * Extract XML from a PDF buffer
60
+ * @param pdfBuffer PDF buffer
61
+ * @returns XML content or null if not found
62
+ */
63
+ public abstract extractXml(pdfBuffer: Uint8Array | Buffer): Promise<string | null>;
64
+
65
+ /**
66
+ * Check if an XML string is valid
67
+ * @param xmlString XML string to check
68
+ * @returns True if the XML is valid
69
+ */
70
+ protected isValidXml(xmlString: string): boolean {
71
+ try {
72
+ // Basic checks for XML validity
73
+ if (!xmlString || typeof xmlString !== 'string') {
74
+ return false;
75
+ }
76
+
77
+ // Check if it starts with XML declaration or a valid element
78
+ if (!xmlString.includes('<?xml') && !this.hasKnownXmlElement(xmlString)) {
79
+ return false;
80
+ }
81
+
82
+ // Check if the XML string contains known invoice formats
83
+ const hasKnownFormat = this.hasKnownFormat(xmlString);
84
+ if (!hasKnownFormat) {
85
+ return false;
86
+ }
87
+
88
+ // Check if the XML string contains binary data or invalid characters
89
+ if (this.hasBinaryData(xmlString)) {
90
+ return false;
91
+ }
92
+
93
+ // Check if the XML string is too short
94
+ if (xmlString.length < 100) {
95
+ return false;
96
+ }
97
+
98
+ // Check if XML has a proper structure (contains both opening and closing tags)
99
+ if (!this.hasProperXmlStructure(xmlString)) {
100
+ return false;
101
+ }
102
+
103
+ return true;
104
+ } catch (error) {
105
+ console.error('Error validating XML:', error);
106
+ return false;
107
+ }
108
+ }
109
+
110
+ /**
111
+ * Check if the XML string contains a known element
112
+ * @param xmlString XML string to check
113
+ * @returns True if the XML contains a known element
114
+ */
115
+ protected hasKnownXmlElement(xmlString: string): boolean {
116
+ for (const format of this.knownFormats) {
117
+ // Check for opening tag of format
118
+ if (xmlString.includes(`<${format}`)) {
119
+ return true;
120
+ }
121
+ }
122
+ return false;
123
+ }
124
+
125
+ /**
126
+ * Check if the XML string contains a known format
127
+ * @param xmlString XML string to check
128
+ * @returns True if the XML contains a known format
129
+ */
130
+ protected hasKnownFormat(xmlString: string): boolean {
131
+ for (const format of this.knownFormats) {
132
+ if (xmlString.includes(format)) {
133
+ return true;
134
+ }
135
+ }
136
+ return false;
137
+ }
138
+
139
+ /**
140
+ * Check if the XML string has a proper structure
141
+ * @param xmlString XML string to check
142
+ * @returns True if the XML has a proper structure
143
+ */
144
+ protected hasProperXmlStructure(xmlString: string): boolean {
145
+ // Check for at least one matching opening and closing tag
146
+ for (const endTag of this.knownEndTags) {
147
+ const startTag = endTag.replace('/', '');
148
+ if (xmlString.includes(startTag) && xmlString.includes(endTag)) {
149
+ return true;
150
+ }
151
+ }
152
+
153
+ // If no specific tag is found but it has a basic XML structure
154
+ return (
155
+ (xmlString.includes('<?xml') && xmlString.includes('?>')) ||
156
+ (xmlString.match(/<[^>]+>/) !== null && xmlString.match(/<\/[^>]+>/) !== null)
157
+ );
158
+ }
159
+
160
+ /**
161
+ * Check if the XML string contains binary data
162
+ * @param xmlString XML string to check
163
+ * @returns True if the XML contains binary data
164
+ */
165
+ protected hasBinaryData(xmlString: string): boolean {
166
+ // Check for common binary data indicators
167
+ const binaryChars = ['\u0000', '\u0001', '\u0002', '\u0003', '\u0004', '\u0005'];
168
+ const consecutiveNulls = '\u0000\u0000\u0000';
169
+
170
+ // Check for control characters that shouldn't be in XML
171
+ if (binaryChars.some(char => xmlString.includes(char))) {
172
+ return true;
173
+ }
174
+
175
+ // Check for consecutive null bytes which indicate binary data
176
+ if (xmlString.includes(consecutiveNulls)) {
177
+ return true;
178
+ }
179
+
180
+ // Check for high concentration of non-printable characters
181
+ const nonPrintableCount = (xmlString.match(/[\x00-\x08\x0B\x0C\x0E-\x1F]/g) || []).length;
182
+ if (nonPrintableCount > xmlString.length * 0.05) { // More than 5% non-printable
183
+ return true;
184
+ }
185
+
186
+ return false;
187
+ }
188
+
189
+ /**
190
+ * Extract XML from a string
191
+ * @param text Text to extract XML from
192
+ * @param startIndex Index to start extraction from
193
+ * @returns XML content or null if not found
194
+ */
195
+ protected extractXmlFromString(text: string, startIndex: number = 0): string | null {
196
+ try {
197
+ // Find the start of the XML document
198
+ let xmlStartIndex = text.indexOf('<?xml', startIndex);
199
+
200
+ // If no XML declaration, try to find known elements
201
+ if (xmlStartIndex === -1) {
202
+ for (const format of this.knownFormats) {
203
+ const formatStartIndex = text.indexOf(`<${format.split(':').pop()}`, startIndex);
204
+ if (formatStartIndex !== -1) {
205
+ xmlStartIndex = formatStartIndex;
206
+ break;
207
+ }
208
+ }
209
+
210
+ // Still didn't find any start marker
211
+ if (xmlStartIndex === -1) {
212
+ return null;
213
+ }
214
+ }
215
+
216
+ // Try to find the end of the XML document
217
+ let xmlEndIndex = -1;
218
+ for (const endTag of this.knownEndTags) {
219
+ const endIndex = text.indexOf(endTag, xmlStartIndex);
220
+ if (endIndex !== -1) {
221
+ xmlEndIndex = endIndex + endTag.length;
222
+ break;
223
+ }
224
+ }
225
+
226
+ // If no known end tag found, try to use a heuristic approach
227
+ if (xmlEndIndex === -1) {
228
+ // Try to find the last closing tag
229
+ const lastClosingTagMatch = text.slice(xmlStartIndex).match(/<\/[^>]+>(?!.*<\/[^>]+>)/);
230
+ if (lastClosingTagMatch && lastClosingTagMatch.index !== undefined) {
231
+ xmlEndIndex = xmlStartIndex + lastClosingTagMatch.index + lastClosingTagMatch[0].length;
232
+ } else {
233
+ return null;
234
+ }
235
+ }
236
+
237
+ // Extract the XML content
238
+ const xmlContent = text.substring(xmlStartIndex, xmlEndIndex);
239
+
240
+ // Validate the extracted content
241
+ if (this.isValidXml(xmlContent)) {
242
+ return xmlContent;
243
+ }
244
+
245
+ return null;
246
+ } catch (error) {
247
+ console.error('Error extracting XML from string:', error);
248
+ return null;
249
+ }
250
+ }
251
+
252
+ /**
253
+ * Decompress and decode XML content from a PDF stream
254
+ * @param stream PDF stream containing XML data
255
+ * @param fileName Name of the file (for logging)
256
+ * @returns XML content or null if not valid
257
+ */
258
+ protected async extractXmlFromStream(stream: PDFRawStream, fileName: string): Promise<string | null> {
259
+ try {
260
+ // Get the raw bytes from the stream
261
+ const rawBytes = stream.getContents();
262
+
263
+ // First try without decompression (in case the content is not compressed)
264
+ let xmlContent = this.tryDecodeBuffer(rawBytes);
265
+ if (xmlContent && this.isValidXml(xmlContent)) {
266
+ console.log(`Successfully extracted uncompressed XML from PDF file. File name: ${fileName}`);
267
+ return xmlContent;
268
+ }
269
+
270
+ // Try with decompression
271
+ try {
272
+ const decompressedBytes = this.tryDecompress(rawBytes);
273
+ if (decompressedBytes) {
274
+ xmlContent = this.tryDecodeBuffer(decompressedBytes);
275
+ if (xmlContent && this.isValidXml(xmlContent)) {
276
+ console.log(`Successfully extracted decompressed XML from PDF file. File name: ${fileName}`);
277
+ return xmlContent;
278
+ }
279
+ }
280
+ } catch (decompressError) {
281
+ console.log(`Decompression failed for ${fileName}: ${decompressError}`);
282
+ }
283
+
284
+ return null;
285
+ } catch (error) {
286
+ console.error('Error extracting XML from stream:', error);
287
+ return null;
288
+ }
289
+ }
290
+
291
+ /**
292
+ * Try to decompress a buffer using different methods
293
+ * @param buffer Buffer to decompress
294
+ * @returns Decompressed buffer or null if decompression failed
295
+ */
296
+ protected tryDecompress(buffer: Uint8Array): Uint8Array | null {
297
+ try {
298
+ // Try pako inflate (for deflate/zlib compression)
299
+ return pako.inflate(buffer);
300
+ } catch (error) {
301
+ // If pako fails, try other methods if needed
302
+ console.warn('Pako decompression failed, might be uncompressed or using a different algorithm');
303
+ return null;
304
+ }
305
+ }
306
+
307
+ /**
308
+ * Try to decode a buffer to a string using different encodings
309
+ * @param buffer Buffer to decode
310
+ * @returns Decoded string or null if decoding failed
311
+ */
312
+ protected tryDecodeBuffer(buffer: Uint8Array): string | null {
313
+ try {
314
+ // Try UTF-8 first
315
+ let content = new TextDecoder('utf-8').decode(buffer);
316
+ if (this.isPlausibleXml(content)) {
317
+ return content;
318
+ }
319
+
320
+ // Try ISO-8859-1 (Latin1)
321
+ content = this.decodeLatin1(buffer);
322
+ if (this.isPlausibleXml(content)) {
323
+ return content;
324
+ }
325
+
326
+ return null;
327
+ } catch (error) {
328
+ console.warn('Error decoding buffer:', error);
329
+ return null;
330
+ }
331
+ }
332
+
333
+ /**
334
+ * Decode a buffer using ISO-8859-1 (Latin1) encoding
335
+ * @param buffer Buffer to decode
336
+ * @returns Decoded string
337
+ */
338
+ protected decodeLatin1(buffer: Uint8Array): string {
339
+ return Array.from(buffer)
340
+ .map(byte => String.fromCharCode(byte))
341
+ .join('');
342
+ }
343
+
344
+ /**
345
+ * Check if a string is plausibly XML (quick check before validation)
346
+ * @param content String to check
347
+ * @returns True if the string is plausibly XML
348
+ */
349
+ protected isPlausibleXml(content: string): boolean {
350
+ return content.includes('<') &&
351
+ content.includes('>') &&
352
+ (content.includes('<?xml') ||
353
+ this.knownFormats.some(format => content.includes(format)));
354
+ }
355
+ }
@@ -0,0 +1,4 @@
1
+ export * from './base.extractor.js';
2
+ export * from './standard.extractor.js';
3
+ export * from './associated.extractor.js';
4
+ export * from './text.extractor.js';
@@ -0,0 +1,86 @@
1
+ import { PDFDocument, PDFDict, PDFName, PDFRawStream, PDFArray, PDFString } from '../../../plugins.js';
2
+ import { BaseXMLExtractor } from './base.extractor.js';
3
+
4
+ /**
5
+ * Standard PDF XML extractor that extracts XML from embedded files
6
+ * Works with PDF/A-3 documents that follow the standard for embedding files
7
+ */
8
+ export class StandardXMLExtractor extends BaseXMLExtractor {
9
+ /**
10
+ * Extract XML from a PDF buffer using standard PDF/A-3 embedded files
11
+ * @param pdfBuffer PDF buffer
12
+ * @returns XML content or null if not found
13
+ */
14
+ public async extractXml(pdfBuffer: Uint8Array | Buffer): Promise<string | null> {
15
+ try {
16
+ const pdfDoc = await PDFDocument.load(pdfBuffer);
17
+
18
+ // Get the document's metadata dictionary
19
+ const namesDictObj = pdfDoc.catalog.lookup(PDFName.of('Names'));
20
+ if (!(namesDictObj instanceof PDFDict)) {
21
+ console.warn('No Names dictionary found in PDF! This PDF does not contain embedded files.');
22
+ return null;
23
+ }
24
+
25
+ // Get the embedded files dictionary
26
+ const embeddedFilesDictObj = namesDictObj.lookup(PDFName.of('EmbeddedFiles'));
27
+ if (!(embeddedFilesDictObj instanceof PDFDict)) {
28
+ console.warn('No EmbeddedFiles dictionary found! This PDF does not contain embedded files.');
29
+ return null;
30
+ }
31
+
32
+ // Get the names array
33
+ const filesSpecObj = embeddedFilesDictObj.lookup(PDFName.of('Names'));
34
+ if (!(filesSpecObj instanceof PDFArray)) {
35
+ console.warn('No files specified in EmbeddedFiles dictionary!');
36
+ return null;
37
+ }
38
+
39
+ // Try to find an XML file in the embedded files
40
+ for (let i = 0; i < filesSpecObj.size(); i += 2) {
41
+ const fileNameObj = filesSpecObj.lookup(i);
42
+ const fileSpecObj = filesSpecObj.lookup(i + 1);
43
+
44
+ if (!(fileNameObj instanceof PDFString) || !(fileSpecObj instanceof PDFDict)) {
45
+ continue;
46
+ }
47
+
48
+ // Get the filename as string
49
+ const fileName = fileNameObj.decodeText();
50
+
51
+ // Check if it's a known invoice XML file name
52
+ const isKnownFileName = this.knownFileNames.some(
53
+ knownName => fileName.toLowerCase() === knownName.toLowerCase()
54
+ );
55
+
56
+ // Check if it's any XML file or has invoice-related keywords
57
+ const isXmlFile = fileName.toLowerCase().endsWith('.xml') ||
58
+ fileName.toLowerCase().includes('zugferd') ||
59
+ fileName.toLowerCase().includes('factur-x') ||
60
+ fileName.toLowerCase().includes('xrechnung') ||
61
+ fileName.toLowerCase().includes('invoice');
62
+
63
+ if (isKnownFileName || isXmlFile) {
64
+ const efDictObj = fileSpecObj.lookup(PDFName.of('EF'));
65
+ if (!(efDictObj instanceof PDFDict)) {
66
+ continue;
67
+ }
68
+
69
+ const fileStream = efDictObj.lookup(PDFName.of('F'));
70
+ if (fileStream instanceof PDFRawStream) {
71
+ const xmlContent = await this.extractXmlFromStream(fileStream, fileName);
72
+ if (xmlContent) {
73
+ return xmlContent;
74
+ }
75
+ }
76
+ }
77
+ }
78
+
79
+ console.warn('No valid XML found in embedded files');
80
+ return null;
81
+ } catch (error) {
82
+ console.error('Error in standard extraction:', error);
83
+ return null;
84
+ }
85
+ }
86
+ }
@@ -0,0 +1,162 @@
1
+ import { BaseXMLExtractor } from './base.extractor.js';
2
+
3
+ /**
4
+ * Text-based XML extractor for PDF documents
5
+ * Extracts XML by searching for XML patterns in the PDF text
6
+ * Used as a fallback when other extraction methods fail
7
+ */
8
+ export class TextXMLExtractor extends BaseXMLExtractor {
9
+ // Maximum chunk size to process at once (4MB)
10
+ private readonly CHUNK_SIZE = 4 * 1024 * 1024;
11
+
12
+ // Maximum number of chunks to check (effective 20MB search limit)
13
+ private readonly MAX_CHUNKS = 5;
14
+
15
+ // Common XML patterns to look for
16
+ private readonly XML_PATTERNS = [
17
+ '<?xml',
18
+ '<CrossIndustryInvoice',
19
+ '<CrossIndustryDocument',
20
+ '<Invoice',
21
+ '<CreditNote',
22
+ '<rsm:CrossIndustryInvoice',
23
+ '<rsm:CrossIndustryDocument',
24
+ '<ram:CrossIndustryDocument',
25
+ '<ubl:Invoice',
26
+ '<ubl:CreditNote',
27
+ '<FatturaElettronica'
28
+ ];
29
+
30
+ /**
31
+ * Extract XML from a PDF buffer by searching for XML patterns in the text
32
+ * Uses a chunked approach to handle large files efficiently
33
+ * @param pdfBuffer PDF buffer
34
+ * @returns XML content or null if not found
35
+ */
36
+ public async extractXml(pdfBuffer: Uint8Array | Buffer): Promise<string | null> {
37
+ try {
38
+ console.log('Attempting text-based XML extraction from PDF...');
39
+
40
+ // Convert Buffer to Uint8Array if needed
41
+ const buffer = Buffer.isBuffer(pdfBuffer) ? new Uint8Array(pdfBuffer) : pdfBuffer;
42
+
43
+ // Try extracting XML using the chunked approach
44
+ return this.extractXmlFromBufferChunked(buffer);
45
+ } catch (error) {
46
+ console.error('Error in text-based extraction:', error);
47
+ return null;
48
+ }
49
+ }
50
+
51
+ /**
52
+ * Extract XML from buffer using a chunked approach
53
+ * This helps avoid memory issues with large PDFs
54
+ * @param buffer Buffer to search in
55
+ * @returns XML content or null if not found
56
+ */
57
+ private extractXmlFromBufferChunked(buffer: Uint8Array): string | null {
58
+ // Process the PDF in chunks
59
+ for (let chunkIndex = 0; chunkIndex < this.MAX_CHUNKS; chunkIndex++) {
60
+ const startPos = chunkIndex * this.CHUNK_SIZE;
61
+ if (startPos >= buffer.length) break;
62
+
63
+ const endPos = Math.min(startPos + this.CHUNK_SIZE, buffer.length);
64
+ const chunk = buffer.slice(startPos, endPos);
65
+
66
+ // Try to extract XML from this chunk
67
+ const chunkResult = this.processChunk(chunk, startPos);
68
+ if (chunkResult) {
69
+ return chunkResult;
70
+ }
71
+ }
72
+
73
+ console.warn('No valid XML found in any chunk of the PDF');
74
+ return null;
75
+ }
76
+
77
+ /**
78
+ * Process a single chunk of the PDF buffer
79
+ * @param chunk Chunk buffer to process
80
+ * @param chunkOffset Offset position of the chunk in the original buffer
81
+ * @returns XML content or null if not found
82
+ */
83
+ private processChunk(chunk: Uint8Array, chunkOffset: number): string | null {
84
+ try {
85
+ // First try UTF-8 encoding for this chunk
86
+ const utf8String = this.decodeBufferToString(chunk, 'utf-8');
87
+ let xmlContent = this.searchForXmlInString(utf8String);
88
+
89
+ if (xmlContent) {
90
+ console.log(`Found XML content in chunk at offset ${chunkOffset} using UTF-8 encoding`);
91
+ return xmlContent;
92
+ }
93
+
94
+ // If UTF-8 fails, try Latin-1 (ISO-8859-1) which can handle binary better
95
+ const latin1String = this.decodeBufferToString(chunk, 'latin1');
96
+ xmlContent = this.searchForXmlInString(latin1String);
97
+
98
+ if (xmlContent) {
99
+ console.log(`Found XML content in chunk at offset ${chunkOffset} using Latin-1 encoding`);
100
+ return xmlContent;
101
+ }
102
+
103
+ // No XML found in this chunk
104
+ return null;
105
+ } catch (error) {
106
+ console.warn(`Error processing chunk at offset ${chunkOffset}:`, error);
107
+ return null;
108
+ }
109
+ }
110
+
111
+ /**
112
+ * Safely decode a buffer to string using the specified encoding
113
+ * @param buffer Buffer to decode
114
+ * @param encoding Encoding to use ('utf-8' or 'latin1')
115
+ * @returns Decoded string
116
+ */
117
+ private decodeBufferToString(buffer: Uint8Array, encoding: 'utf-8' | 'latin1'): string {
118
+ try {
119
+ if (encoding === 'utf-8') {
120
+ return new TextDecoder('utf-8', { fatal: false }).decode(buffer);
121
+ } else {
122
+ // For Latin-1 we can use a direct mapping (bytes 0-255 map directly to code points 0-255)
123
+ // This is more reliable for binary data than TextDecoder for legacy encodings
124
+ return Array.from(buffer)
125
+ .map(byte => String.fromCharCode(byte))
126
+ .join('');
127
+ }
128
+ } catch (error) {
129
+ console.warn(`Error decoding buffer using ${encoding}:`, error);
130
+ // Return empty string on error to allow processing to continue
131
+ return '';
132
+ }
133
+ }
134
+
135
+ /**
136
+ * Search for XML patterns in a string
137
+ * @param content String to search in
138
+ * @returns XML content or null if not found
139
+ */
140
+ private searchForXmlInString(content: string): string | null {
141
+ if (!content) return null;
142
+
143
+ // Search for each XML pattern
144
+ for (const pattern of this.XML_PATTERNS) {
145
+ const patternIndex = content.indexOf(pattern);
146
+ if (patternIndex !== -1) {
147
+ console.log(`Found XML pattern "${pattern}" at position ${patternIndex}`);
148
+
149
+ // Try to extract the XML content starting from the pattern position
150
+ const xmlContent = this.extractXmlFromString(content, patternIndex);
151
+
152
+ // Validate the extracted content
153
+ if (xmlContent && this.isValidXml(xmlContent)) {
154
+ console.log('Successfully extracted and validated XML from text');
155
+ return xmlContent;
156
+ }
157
+ }
158
+ }
159
+
160
+ return null;
161
+ }
162
+ }