@elisra-devops/docgen-skins 0.10.8 → 0.10.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,8 +1,16 @@
1
- import { Run, WIProperty, StyleOptions } from '../wordJsonModels';
2
- import JSONHtml from '../html/JSONHtml';
3
- import JSONParagraph from './JSONParagraph';
4
- import JSONTable from '../table/JSONTable';
5
1
  import logger from '../../../services/logger';
2
+ import {
3
+ Run,
4
+ WIProperty,
5
+ StyleOptions,
6
+ RichNode,
7
+ Paragraph,
8
+ Table,
9
+ TableRow,
10
+ TableCell,
11
+ List,
12
+ ListItem,
13
+ } from '../wordJsonModels';
6
14
 
7
15
  export default class JSONRichTextParagraph {
8
16
  paragraphStyles: StyleOptions;
@@ -15,413 +23,432 @@ export default class JSONRichTextParagraph {
15
23
  headingLevel: number,
16
24
  skipImageFormat = false
17
25
  ) {
18
- this.paragraphStyles = paragraphStyles;
19
- this.wiId = wiId;
20
- this.paragraphTemplate = this.generateJsonRichTextParagraphs(
21
- field,
22
- this.paragraphStyles,
23
- skipImageFormat
24
- );
25
- if (headingLevel && field.name === 'Title: ') {
26
- this.paragraphTemplate.headingLevel = headingLevel;
26
+ if (!field || !field.richTextNodes) {
27
+ throw new Error('Invalid field or missing richTextNodes');
27
28
  }
29
+
30
+ this.paragraphStyles = paragraphStyles || {
31
+ isBold: false,
32
+ IsItalic: false,
33
+ IsUnderline: false,
34
+ Size: 12,
35
+ Uri: null,
36
+ Font: 'Arial',
37
+ InsertLineBreak: false,
38
+ InsertSpace: false,
39
+ };
40
+ this.wiId = wiId;
41
+ this.paragraphTemplate = this.buildDocFromNodes(field.richTextNodes);
28
42
  } //constructor
29
43
 
30
- generateJsonRichTextParagraphs(field: WIProperty, paragraphStyles, skipImageFormat = false): Run[] {
31
- let titleStyle = JSON.parse(JSON.stringify(paragraphStyles));
32
-
33
- titleStyle.isBold = true;
34
- titleStyle.IsUnderline = true;
35
- titleStyle.InsertSpace = true;
36
- titleStyle.InsertLineBreak = false;
37
-
38
- let richTextSkin: any[] = [];
39
- //handle rich text
40
- try {
41
- field.richText.forEach((content) => {
42
- switch (content.type) {
43
- case 'paragraph':
44
- content.data.fields.forEach((richTextField: WIProperty) => {
45
- if (/<\/?[a-z][\s\S]*>/.test(richTextField.value)) {
46
- let html = new JSONHtml(richTextField.value);
47
- richTextSkin.push(html.getJSONHtml());
48
- } else {
49
- //making sure ID is not printed
50
- let paragraphSkin = new JSONParagraph(richTextField, paragraphStyles, 0, 0);
51
- richTextSkin.push(paragraphSkin.getJSONParagraph());
52
- }
53
- });
54
- break;
55
- case 'picture':
56
- if (!skipImageFormat) {
57
- richTextSkin.push({
58
- type: 'picture',
59
- Path: content.data,
60
- });
61
- }
62
-
63
- break;
64
- case 'table':
65
- if (content.data) {
66
- let tableSkin = new JSONTable(content.data, undefined, paragraphStyles, 0);
67
- richTextSkin.push(tableSkin.getJSONTable());
68
- }
69
- break;
44
+ getJSONRichTextParagraph(): any[] {
45
+ return this.paragraphTemplate;
46
+ } //getJSONTable
47
+
48
+ /**
49
+ * Parses a given RichNode and returns a corresponding Paragraph, Table, List,
50
+ * an array of these types, or null if the node type is not recognized.
51
+ *
52
+ * @param node - The RichNode to be parsed.
53
+ * @returns A Paragraph, Table, List, an array of these types, or null.
54
+ *
55
+ * The function handles different node types as follows:
56
+ * - 'paragraph': Parses the node as a paragraph.
57
+ * - 'table': Parses the node as a table.
58
+ * - 'list': Parses the node as a list.
59
+ * - 'text', 'image', 'break': Converts the node to a paragraph with a single run.
60
+ * - 'other': Recursively parses child nodes and returns an array of results.
61
+ * - Default: Returns null for unrecognized node types.
62
+ */
63
+ private parseNode(node: RichNode): Paragraph | Table | List | Array<Paragraph | Table | List> | null {
64
+ if (!node) return null;
65
+
66
+ // Ensure node has valid type
67
+ if (typeof node.type !== 'string') {
68
+ logger.warn(`Invalid node type encountered: ${JSON.stringify(node)}`);
69
+ return null;
70
+ }
71
+
72
+ switch (node.type) {
73
+ case 'paragraph':
74
+ return this.parseParagraphNode(node);
75
+ case 'table':
76
+ return this.parseTableNode(node);
77
+ case 'list':
78
+ return this.parseList(node);
79
+ case 'text':
80
+ case 'image':
81
+ case 'break':
82
+ return { type: 'paragraph', runs: [this.nodeToRun(node)] };
83
+ case 'other':
84
+ let childrenResults: Array<Paragraph | Table | List> = [];
85
+ for (const child of node.children) {
86
+ const childResult = this.parseNode(child);
87
+ if (Array.isArray(childResult)) {
88
+ childrenResults.push(...childResult);
89
+ } else if (childResult) {
90
+ childrenResults.push(childResult);
91
+ }
70
92
  }
93
+ return childrenResults.length > 0 ? childrenResults : null;
94
+ default:
95
+ return null;
96
+ }
97
+ }
98
+
99
+ /**
100
+ * Parses a paragraph node and converts it into a Paragraph object.
101
+ *
102
+ * @param node - The node to parse, expected to have a `children` property.
103
+ * @returns A Paragraph object containing the parsed runs.
104
+ */
105
+ private parseParagraphNode(node: any): Paragraph {
106
+ if (!node || !Array.isArray(node.children)) {
107
+ return { type: 'paragraph', runs: [] };
108
+ }
109
+
110
+ const runs: Run[] = this.gatherRuns(node.children);
111
+ // Ensure at least one run exists
112
+ if (runs.length === 0) {
113
+ runs.push({
114
+ type: 'text',
115
+ value: '',
116
+ textStyling: {
117
+ Bold: this.paragraphStyles.isBold,
118
+ Italic: this.paragraphStyles.IsItalic,
119
+ Size: this.paragraphStyles.Size,
120
+ Font: this.paragraphStyles.Font,
121
+ Uri: this.paragraphStyles.Uri,
122
+ InsertLineBreak: this.paragraphStyles.InsertLineBreak,
123
+ InsertSpace: this.paragraphStyles.InsertSpace,
124
+ Underline: this.paragraphStyles.IsUnderline,
125
+ },
71
126
  });
72
- return richTextSkin;
73
- } catch (e) {
74
- logger.warn(e);
127
+ }
128
+
129
+ return { type: 'paragraph', runs };
130
+ }
131
+
132
+ /**
133
+ * Gathers runs from the provided children nodes.
134
+ *
135
+ * This method processes an array of child nodes and converts them into an array of `Run` objects.
136
+ * It handles different types of nodes such as text, image, break, paragraph, table, list, and other.
137
+ *
138
+ * - For `text`, `image`, and `break` nodes, it directly converts them to `Run` objects.
139
+ * - For `paragraph` nodes, it recursively gathers runs from the child paragraphs.
140
+ * - For `table` and `list` nodes, it adds a placeholder `Run` object indicating the embedded type.
141
+ * - For `other` nodes, it recursively gathers runs from the child nodes.
142
+ *
143
+ * @param children - An array of child nodes to process.
144
+ * @returns An array of `Run` objects representing the gathered runs.
145
+ */
146
+ private gatherRuns(children: any[]): Run[] {
147
+ if (!Array.isArray(children)) return [];
148
+
149
+ const runs: Run[] = [];
150
+
151
+ for (const child of children) {
152
+ if (!child || typeof child.type !== 'string') {
153
+ logger.warn(`Invalid child node: ${JSON.stringify(child)}`);
154
+ continue;
155
+ }
156
+
157
+ try {
158
+ if (child.type === 'text' || child.type === 'image' || child.type === 'break') {
159
+ runs.push(this.nodeToRun(child));
160
+ } else if (child.type === 'paragraph') {
161
+ // Recursively gather runs from child paragraphs
162
+ runs.push(...this.gatherRuns(child.children));
163
+ } else if (child.type === 'table' || child.type === 'list') {
164
+ // Not supported in this context, so we'll just add a placeholder
165
+ runs.push({ type: 'other', value: `[Embedded ${child.type}]` });
166
+ } else if (child.type === 'other') {
167
+ // Recursively gather runs from inside <span> or <b>, etc.
168
+ runs.push(...this.gatherRuns(child.children));
169
+ }
170
+ } catch (error) {
171
+ logger.error(`Error gathering runs: ${error.message}`);
172
+ logger.error(`Child node: ${JSON.stringify(child)}`);
173
+ logger.error(`Error stack: ${error.stack}`);
174
+ continue;
175
+ }
176
+ }
177
+
178
+ return runs;
179
+ }
180
+
181
+ /**
182
+ * Converts a node object to a Run object based on its type and text styling.
183
+ *
184
+ * @param node - The node object to be converted. It can be of type 'text', 'image', or 'break'.
185
+ * @returns A Run object with the appropriate type and properties based on the input node.
186
+ *
187
+ * The Run object can have the following structure:
188
+ * - For 'text' nodes: { type: 'text', value: string, textStyling: object }
189
+ * - For 'image' nodes: { type: 'image', value: '', src: string }
190
+ * - For 'break' nodes: { type: 'break' }
191
+ * - For other types: { type: 'other', value: '' }
192
+ *
193
+ * The textStyling object includes:
194
+ * - Bold: boolean
195
+ * - Italic: boolean
196
+ * - Underline: boolean
197
+ * - Size: number
198
+ * - Uri: string | null
199
+ * - Font: string
200
+ * - InsertLineBreak: boolean
201
+ * - InsertSpace: boolean
202
+ */
203
+ private nodeToRun(node: any): Run {
204
+ const textStyling = {
205
+ Bold: node.textStyling?.Bold || this.paragraphStyles.isBold,
206
+ Italic: node.textStyling?.Italic || this.paragraphStyles.IsItalic,
207
+ Underline: node.textStyling?.Underline || this.paragraphStyles.IsUnderline,
208
+ Size: this.paragraphStyles.Size || 12,
209
+ Uri: this.paragraphStyles.Uri || null,
210
+ Font: this.paragraphStyles.Font || 'Arial',
211
+ InsertLineBreak: this.paragraphStyles.InsertLineBreak || false,
212
+ InsertSpace: node.textStyling?.InsertSpace || this.paragraphStyles.InsertSpace,
213
+ };
214
+
215
+ if (node.type === 'text') {
216
+ return { type: 'text', value: node.value, textStyling: textStyling };
217
+ } else if (node.type === 'image') {
218
+ return { type: 'image', value: '', src: node.src || '' };
219
+ } else if (node.type === 'break') {
220
+ return { type: 'break' };
221
+ } else {
222
+ return { type: 'other', value: '' };
223
+ }
224
+ }
225
+
226
+ /**
227
+ * Converts a given node object into a Table structure.
228
+ *
229
+ * @remarks
230
+ * This function processes the children of the specified node and
231
+ * transforms them into a table representation, adding default properties
232
+ * such as heading level, page break insertion, and an empty rows array.
233
+ *
234
+ * @param node - The node object containing the structural details needed
235
+ * to create a table.
236
+ * @returns A fully constructed Table object with the parsed node data.
237
+ */
238
+ private parseTableNode(node: any): Table {
239
+ const table: Table = {
240
+ type: 'table',
241
+ headingLevel: 0,
242
+ insertPageBreak: false,
243
+ Rows: [],
244
+ };
245
+
246
+ if (!node || !Array.isArray(node.children)) {
247
+ logger.warn('Invalid table node structure');
248
+ return table;
249
+ }
250
+
251
+ this.parseTableChildren(node.children, table);
252
+
253
+ if (!this.validateTableStructure(table)) {
254
+ logger.warn('Inconsistent table structure detected');
255
+ }
256
+
257
+ return table;
258
+ }
259
+
260
+ /**
261
+ * Parses child elements of a table structure and updates the provided Table instance accordingly.
262
+ *
263
+ * @param children - An array of child nodes to be evaluated for table content.
264
+ * @param table - The Table object to which parsed rows and nested structures will be appended.
265
+ *
266
+ * @remarks
267
+ * This method recursively traverses and identifies relevant table elements (e.g., <tr>, <thead>, <tbody>, etc.)
268
+ * to generate the final row data for the given Table instance.
269
+ */
270
+ private parseTableChildren(children: any[], table: Table): void {
271
+ for (const child of children || []) {
272
+ if (child.type === 'other') {
273
+ const tn = (child.tagName || '').toLowerCase();
274
+ if (tn === 'tr') {
275
+ // Parse a table row
276
+ const row = this.parseTableRow(child);
277
+ table.Rows.push(row);
278
+ } else if (tn === 'thead' || tn === 'tbody' || tn === 'tfoot') {
279
+ // Recurse into thead/tbody/tfoot
280
+ this.parseTableChildren(child.children, table);
281
+ } else {
282
+ // Recurse into other nodes
283
+ this.parseTableChildren(child.children, table);
284
+ }
285
+ } else if (child.type === 'table') {
286
+ this.parseTableChildren(child.children, table);
287
+ } else {
288
+ //skip
289
+ continue;
290
+ }
291
+ }
292
+ }
293
+
294
+ /**
295
+ * Parses the given table row node and extracts its cells by iterating through its child elements.
296
+ * Each child element that is recognized as a table cell (td or th) is converted into a corresponding cell object.
297
+ *
298
+ * @param trNode - An AST node representing the table row, containing possible table cell or header elements.
299
+ * @returns A TableRow object containing an array of processed cells.
300
+ */
301
+ private parseTableRow(trNode: any): TableRow {
302
+ const row: TableRow = { Cells: [] };
303
+
304
+ // parse <td> or <th> inside this <tr>
305
+ for (const child of trNode.children || []) {
306
+ if (child.type === 'other' && (child.tagName === 'td' || child.tagName === 'th')) {
307
+ row.Cells.push(this.parseTableCell(child));
308
+ } else {
309
+ // skip or recurse
310
+ }
311
+ }
312
+
313
+ return row;
314
+ }
315
+
316
+ /**
317
+ * Build one TableCell from <td> node
318
+ */
319
+ private parseTableCell(tdNode: any): TableCell {
320
+ // We'll collect everything in one paragraph
321
+ const cellParagraphRuns = this.gatherRunsFromAllParagraphs(tdNode.children);
322
+
323
+ return {
324
+ Paragraphs: [
325
+ {
326
+ Runs: cellParagraphRuns,
327
+ },
328
+ ],
329
+ };
330
+ }
331
+
332
+ /**
333
+ * Gather runs from any text/image/paragraph in <td>
334
+ * (including nested <other> tags).
335
+ */
336
+ private gatherRunsFromAllParagraphs(children: any[]): Run[] {
337
+ const runs: Run[] = [];
338
+ for (const child of children || []) {
339
+ if (child.type === 'text' || child.type === 'image') {
340
+ runs.push(this.nodeToRun(child));
341
+ } else if (child.type === 'paragraph') {
342
+ runs.push(...this.gatherRuns(child.children));
343
+ } else if (child.type === 'other' || child.type === 'table' || child.type === 'list') {
344
+ // recursively gather
345
+ runs.push(...this.gatherRunsFromAllParagraphs(child.children));
346
+ }
347
+ }
348
+ return runs;
349
+ }
350
+
351
+ /**
352
+ * Parses a given node to create a List object.
353
+ *
354
+ * @param node - The node to parse, expected to have properties `isOrdered` and `children`.
355
+ * @returns A List object with the parsed data.
356
+ */
357
+ private parseList(node: any): List {
358
+ const list: List = {
359
+ type: 'list',
360
+ isOrdered: node.isOrdered,
361
+ listItems: [],
362
+ };
363
+
364
+ this.parseListChildren(node.children, list);
365
+ return list;
366
+ }
367
+
368
+ /**
369
+ * Parses the children of a list and adds them to the provided list object.
370
+ *
371
+ * @param children - An array of child nodes to be parsed.
372
+ * @param list - The list object to which parsed items will be added.
373
+ * @param level - The current depth level of the list (default is 0).
374
+ */
375
+ private parseListChildren(children: any[], list: List, level: number = 0): void {
376
+ for (const child of children || []) {
377
+ if (child.type === 'other') {
378
+ const tn = (child.tagName || '').toLowerCase();
379
+ if (tn === 'li') {
380
+ // Parse a list item
381
+ const item = this.parseListItem(child, level);
382
+ list.listItems.push(item);
383
+ } else {
384
+ // Recurse into other nodes
385
+ this.parseListChildren(child.children, list, level + 1);
386
+ }
387
+ } else {
388
+ //skip
389
+ continue;
390
+ }
391
+ }
392
+ }
393
+
394
+ /**
395
+ * Parses a list item node and converts it into a `ListItem` object.
396
+ *
397
+ * @param liNode - The list item node to parse.
398
+ * @param level - The nesting level of the list item.
399
+ * @returns A `ListItem` object containing the parsed list item data.
400
+ */
401
+ private parseListItem(liNode: any, level: number): ListItem {
402
+ const items = this.gatherRunsFromAllParagraphs(liNode.children);
403
+
404
+ return {
405
+ Runs: items,
406
+ level: level,
407
+ };
408
+ }
409
+
410
+ /**
411
+ * Builds an array of document elements (Paragraph, Table, or List) from an array of RichNode objects.
412
+ *
413
+ * @param nodes - An array of RichNode objects to be parsed into document elements.
414
+ * @returns An array of Paragraph, Table, or List elements.
415
+ */
416
+ private buildDocFromNodes(nodes: RichNode[]): Array<Paragraph | Table | List> {
417
+ if (!Array.isArray(nodes)) {
418
+ logger.error('Invalid nodes array');
75
419
  return [];
76
420
  }
77
- } //generateJsonRichTextParagraphs
78
421
 
79
- getJSONRichTextParagraph(): any[] {
80
- return this.paragraphTemplate;
81
- } //getJSONRichTextParagraph
82
-
83
- // /**
84
- // * Parses a given RichNode and returns a corresponding Paragraph, Table, List,
85
- // * an array of these types, or null if the node type is not recognized.
86
- // *
87
- // * @param node - The RichNode to be parsed.
88
- // * @returns A Paragraph, Table, List, an array of these types, or null.
89
- // *
90
- // * The function handles different node types as follows:
91
- // * - 'paragraph': Parses the node as a paragraph.
92
- // * - 'table': Parses the node as a table.
93
- // * - 'list': Parses the node as a list.
94
- // * - 'text', 'image', 'break': Converts the node to a paragraph with a single run.
95
- // * - 'other': Recursively parses child nodes and returns an array of results.
96
- // * - Default: Returns null for unrecognized node types.
97
- // */
98
- // private parseNode(node: RichNode): Paragraph | Table | List | Array<Paragraph | Table | List> | null {
99
- // if (!node || typeof node.type !== 'string') {
100
- // logger.warn(`Invalid node structure: ${JSON.stringify(node)}`);
101
- // return null;
102
- // }
103
-
104
- // switch (node.type) {
105
- // case 'paragraph':
106
- // return this.parseParagraphNode(node);
107
- // case 'table':
108
- // return this.parseTableNode(node);
109
- // case 'list':
110
- // return this.parseList(node);
111
- // case 'text':
112
- // case 'image':
113
- // case 'break':
114
- // return { type: 'paragraph', runs: [this.nodeToRun(node)] };
115
- // case 'other':
116
- // let childrenResults: Array<Paragraph | Table | List> = [];
117
- // for (const child of node.children) {
118
- // const childResult = this.parseNode(child);
119
- // if (Array.isArray(childResult)) {
120
- // childrenResults.push(...childResult);
121
- // } else if (childResult) {
122
- // childrenResults.push(childResult);
123
- // }
124
- // }
125
- // return childrenResults.length > 0 ? childrenResults : null;
126
- // default:
127
- // return null;
128
- // }
129
- // }
130
-
131
- // /**
132
- // * Parses a paragraph node and converts it into a Paragraph object.
133
- // *
134
- // * @param node - The node to parse, expected to have a `children` property.
135
- // * @returns A Paragraph object containing the parsed runs.
136
- // */
137
- // private parseParagraphNode(node: any): Paragraph {
138
- // const runs: Run[] = this.gatherRuns(node.children);
139
- // return { type: 'paragraph', runs };
140
- // }
141
-
142
- // /**
143
- // * Gathers runs from the provided children nodes.
144
- // *
145
- // * This method processes an array of child nodes and converts them into an array of `Run` objects.
146
- // * It handles different types of nodes such as text, image, break, paragraph, table, list, and other.
147
- // *
148
- // * - For `text`, `image`, and `break` nodes, it directly converts them to `Run` objects.
149
- // * - For `paragraph` nodes, it recursively gathers runs from the child paragraphs.
150
- // * - For `table` and `list` nodes, it adds a placeholder `Run` object indicating the embedded type.
151
- // * - For `other` nodes, it recursively gathers runs from the child nodes.
152
- // *
153
- // * @param children - An array of child nodes to process.
154
- // * @returns An array of `Run` objects representing the gathered runs.
155
- // */
156
- // private gatherRuns(children: any[]): Run[] {
157
- // const runs: Run[] = [];
158
- // if (!Array.isArray(children)) {
159
- // logger.warn(`Invalid children structure: ${JSON.stringify(children)}`);
160
- // return runs;
161
- // }
162
- // for (const child of children || []) {
163
- // if (child.type === 'text' || child.type === 'image' || child.type === 'break') {
164
- // runs.push(this.nodeToRun(child));
165
- // } else if (child.type === 'paragraph') {
166
- // // Recursively gather runs from child paragraphs
167
- // runs.push(...this.gatherRuns(child.children));
168
- // } else if (child.type === 'table' || child.type === 'list') {
169
- // // Not supported in this context, so we'll just add a placeholder
170
- // runs.push({ type: 'other', value: `[Embedded ${child.type}]` });
171
- // } else if (child.type === 'other') {
172
- // // Recursively gather runs from inside <span> or <b>, etc.
173
- // // TODO: Handle b and i, u tags and more relevant tags for text styling
174
- // runs.push(...this.gatherRuns(child.children));
175
- // }
176
- // }
177
- // return runs;
178
- // }
179
-
180
- // /**
181
- // * Converts a node object to a Run object based on its type and text styling.
182
- // *
183
- // * @param node - The node object to be converted. It can be of type 'text', 'image', or 'break'.
184
- // * @returns A Run object with the appropriate type and properties based on the input node.
185
- // *
186
- // * The Run object can have the following structure:
187
- // * - For 'text' nodes: { type: 'text', value: string, textStyling: object }
188
- // * - For 'image' nodes: { type: 'image', value: '', src: string }
189
- // * - For 'break' nodes: { type: 'break' }
190
- // * - For other types: { type: 'other', value: '' }
191
- // *
192
- // * The textStyling object includes:
193
- // * - Bold: boolean
194
- // * - Italic: boolean
195
- // * - Underline: boolean
196
- // * - Size: number
197
- // * - Uri: string | null
198
- // * - Font: string
199
- // * - InsertLineBreak: boolean
200
- // * - InsertSpace: boolean
201
- // */
202
- // private nodeToRun(node: any): Run {
203
- // const textStyling = {
204
- // ...this.paragraphStyles, // Base styles
205
- // ...node.textStyling, // Override with node specific styling
206
- // Size: this.paragraphStyles.Size || 12,
207
- // Uri: this.paragraphStyles.Uri || null,
208
- // Font: this.paragraphStyles.Font || 'Arial',
209
- // InsertLineBreak: this.paragraphStyles.InsertLineBreak || false,
210
- // };
211
-
212
- // if (node.type === 'text') {
213
- // return { type: 'text', value: node.value, textStyling: textStyling };
214
- // } else if (node.type === 'image') {
215
- // return { type: 'image', value: '', src: node.src || '' };
216
- // } else if (node.type === 'break') {
217
- // return { type: 'break' };
218
- // } else {
219
- // return { type: 'other', value: '' };
220
- // }
221
- // }
222
-
223
- // /**
224
- // * Converts a given node object into a Table structure.
225
- // *
226
- // * @remarks
227
- // * This function processes the children of the specified node and
228
- // * transforms them into a table representation, adding default properties
229
- // * such as heading level, page break insertion, and an empty rows array.
230
- // *
231
- // * @param node - The node object containing the structural details needed
232
- // * to create a table.
233
- // * @returns A fully constructed Table object with the parsed node data.
234
- // */
235
- // private parseTableNode(node: any): Table {
236
- // const table: Table = {
237
- // type: 'table',
238
- // headingLevel: 0,
239
- // insertPageBreak: false,
240
- // Rows: [],
241
- // };
242
-
243
- // this.parseTableChildren(node.children, table);
244
- // return table;
245
- // }
246
-
247
- // /**
248
- // * Parses child elements of a table structure and updates the provided Table instance accordingly.
249
- // *
250
- // * @param children - An array of child nodes to be evaluated for table content.
251
- // * @param table - The Table object to which parsed rows and nested structures will be appended.
252
- // *
253
- // * @remarks
254
- // * This method recursively traverses and identifies relevant table elements (e.g., <tr>, <thead>, <tbody>, etc.)
255
- // * to generate the final row data for the given Table instance.
256
- // */
257
- // private parseTableChildren(children: any[], table: Table): void {
258
- // for (const child of children || []) {
259
- // if (child.type === 'other') {
260
- // const tn = (child.tagName || '').toLowerCase();
261
- // if (tn === 'tr') {
262
- // // Parse a table row
263
- // const row = this.parseTableRow(child);
264
- // table.Rows.push(row);
265
- // } else if (tn === 'thead' || tn === 'tbody' || tn === 'tfoot') {
266
- // // Recurse into thead/tbody/tfoot
267
- // this.parseTableChildren(child.children, table);
268
- // } else {
269
- // // Recurse into other nodes
270
- // this.parseTableChildren(child.children, table);
271
- // }
272
- // } else if (child.type === 'table') {
273
- // this.parseTableChildren(child.children, table);
274
- // } else {
275
- // //skip
276
- // continue;
277
- // }
278
- // }
279
- // }
280
-
281
- // /**
282
- // * Parses the given table row node and extracts its cells by iterating through its child elements.
283
- // * Each child element that is recognized as a table cell (td or th) is converted into a corresponding cell object.
284
- // *
285
- // * @param trNode - An AST node representing the table row, containing possible table cell or header elements.
286
- // * @returns A TableRow object containing an array of processed cells.
287
- // */
288
- // private parseTableRow(trNode: any): TableRow {
289
- // const row: TableRow = { Cells: [] };
290
-
291
- // // parse <td> or <th> inside this <tr>
292
- // for (const child of trNode.children || []) {
293
- // if (child.type === 'other' && (child.tagName === 'td' || child.tagName === 'th')) {
294
- // row.Cells.push(this.parseTableCell(child));
295
- // } else {
296
- // // skip or recurse
297
- // }
298
- // }
299
-
300
- // return row;
301
- // }
302
-
303
- // /**
304
- // * Build one TableCell from <td> node
305
- // */
306
- // private parseTableCell(tdNode: any): TableCell {
307
- // // We'll collect everything in one paragraph
308
- // const cellParagraphRuns = this.gatherRunsFromAllParagraphs(tdNode.children);
309
-
310
- // return {
311
- // Paragraphs: [
312
- // {
313
- // Runs: cellParagraphRuns,
314
- // },
315
- // ],
316
- // };
317
- // }
318
-
319
- // /**
320
- // * Gather runs from any text/image/paragraph in <td>
321
- // * (including nested <other> tags).
322
- // */
323
- // private gatherRunsFromAllParagraphs(children: any[]): Run[] {
324
- // const runs: Run[] = [];
325
- // for (const child of children || []) {
326
- // if (child.type === 'text' || child.type === 'image') {
327
- // runs.push(this.nodeToRun(child));
328
- // } else if (child.type === 'paragraph') {
329
- // runs.push(...this.gatherRuns(child.children));
330
- // } else if (child.type === 'other' || child.type === 'table' || child.type === 'list') {
331
- // // recursively gather
332
- // runs.push(...this.gatherRunsFromAllParagraphs(child.children));
333
- // }
334
- // }
335
- // return runs;
336
- // }
337
-
338
- // /**
339
- // * Parses a given node to create a List object.
340
- // *
341
- // * @param node - The node to parse, expected to have properties `isOrdered` and `children`.
342
- // * @returns A List object with the parsed data.
343
- // */
344
- // private parseList(node: any): List {
345
- // const list: List = {
346
- // type: 'list',
347
- // isOrdered: node.isOrdered,
348
- // listItems: [],
349
- // };
350
-
351
- // this.parseListChildren(node.children, list);
352
- // return list;
353
- // }
354
-
355
- // /**
356
- // * Parses the children of a list and adds them to the provided list object.
357
- // *
358
- // * @param children - An array of child nodes to be parsed.
359
- // * @param list - The list object to which parsed items will be added.
360
- // * @param level - The current depth level of the list (default is 0).
361
- // */
362
- // private parseListChildren(children: any[], list: List, level: number = 0): void {
363
- // if (level > 10) {
364
- // // Prevent infinite recursion
365
- // logger.warn('Maximum list nesting level reached');
366
- // return;
367
- // }
368
-
369
- // for (const child of children || []) {
370
- // if (child.type === 'other') {
371
- // const tn = (child.tagName || '').toLowerCase();
372
- // if (tn === 'li') {
373
- // // Parse a list item
374
- // const item = this.parseListItem(child, level);
375
- // list.listItems.push(item);
376
- // } else {
377
- // // Recurse into other nodes
378
- // this.parseListChildren(child.children, list, level + 1);
379
- // }
380
- // } else {
381
- // //skip
382
- // continue;
383
- // }
384
- // }
385
- // }
386
-
387
- // /**
388
- // * Parses a list item node and converts it into a `ListItem` object.
389
- // *
390
- // * @param liNode - The list item node to parse.
391
- // * @param level - The nesting level of the list item.
392
- // * @returns A `ListItem` object containing the parsed list item data.
393
- // */
394
- // private parseListItem(liNode: any, level: number): ListItem {
395
- // const items = this.gatherRunsFromAllParagraphs(liNode.children);
396
-
397
- // return {
398
- // Runs: items,
399
- // level: level,
400
- // };
401
- // }
402
-
403
- // /**
404
- // * Builds an array of document elements (Paragraph, Table, or List) from an array of RichNode objects.
405
- // *
406
- // * @param nodes - An array of RichNode objects to be parsed into document elements.
407
- // * @returns An array of Paragraph, Table, or List elements.
408
- // */
409
- // private buildDocFromNodes(nodes: RichNode[]): Array<Paragraph | Table | List> {
410
- // const results: Array<Paragraph | Table | List> = [];
411
- // for (const node of nodes) {
412
- // try {
413
- // const items = this.parseNode(node);
414
- // if (Array.isArray(items)) {
415
- // results.push(...items);
416
- // } else if (items) {
417
- // results.push(items);
418
- // }
419
- // } catch (err: any) {
420
- // logger.error(`Error parsing node: ${err.message}`);
421
- // // Continue processing other nodes
422
- // continue;
423
- // }
424
- // }
425
- // return results;
426
- // }
422
+ const results: Array<Paragraph | Table | List> = [];
423
+
424
+ for (const node of nodes) {
425
+ try {
426
+ const items = this.parseNode(node);
427
+ if (Array.isArray(items)) {
428
+ results.push(...items.filter(Boolean));
429
+ } else if (items) {
430
+ results.push(items);
431
+ }
432
+ } catch (error) {
433
+ logger.error(`Error processing node: ${error.message}`);
434
+ logger.error(`Node: ${JSON.stringify(node)}`);
435
+ logger.error(`Error stack: ${error.stack}`);
436
+ continue;
437
+ }
438
+ }
439
+
440
+ // Ensure there's at least one paragraph
441
+ if (results.length === 0) {
442
+ results.push({ type: 'paragraph', runs: [] });
443
+ }
444
+
445
+ return results;
446
+ }
447
+
448
+ private validateTableStructure(table: Table): boolean {
449
+ if (!table.Rows) return false;
450
+
451
+ const firstRowCellCount = table.Rows[0]?.Cells?.length || 0;
452
+ return table.Rows.every((row) => row.Cells.length === firstRowCellCount);
453
+ }
427
454
  } //class