@kimdayoun/hwpx-mcp 0.3.0 → 0.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +201 -0
- package/README.md +30 -49
- package/dist/HangingIndentCalculator.js +136 -166
- package/dist/HwpxDocument.d.ts +275 -12
- package/dist/HwpxDocument.js +1589 -627
- package/dist/HwpxParser.js +18 -2
- package/dist/XmlWellFormed.d.ts +5 -0
- package/dist/XmlWellFormed.js +52 -0
- package/dist/index.js +117 -62
- package/dist/types.d.ts +5 -0
- package/package.json +10 -2
package/dist/HwpxDocument.js
CHANGED
|
@@ -9,19 +9,47 @@ const pako_1 = __importDefault(require("pako"));
|
|
|
9
9
|
const HwpxParser_1 = require("./HwpxParser");
|
|
10
10
|
const HangingIndentCalculator_1 = require("./HangingIndentCalculator");
|
|
11
11
|
const MAX_UNDO_STACK_SIZE = 50;
|
|
12
|
+
/**
|
|
13
|
+
* HWPX 텍스트 노드 `<hp:t>` / `<hs:t>` 전용 매처.
|
|
14
|
+
*
|
|
15
|
+
* `<(hp|hs):t([^>]*)>` 처럼 태그명 뒤 경계를 두지 않으면 `<hp:tc>`·`<hp:tr>`·
|
|
16
|
+
* `<hp:tbl>` 같은 형제 태그의 접두사까지 삼킨다. 그 상태로 본문을 지우면
|
|
17
|
+
* 셀 구조가 통째로 사라지고 닫는 태그만 남아 한/글이 파일을 열지 못한다.
|
|
18
|
+
* 뒤에 오는 문자가 공백·`/`·`>` 중 하나임을 강제해 태그명을 정확히 끊는다.
|
|
19
|
+
*/
|
|
20
|
+
const T_TAG_WITH_CONTENT = /<(hp|hs):t((?:\s[^>]*)?)>[\s\S]*?<\/\1:t>/g;
|
|
21
|
+
const T_TAG_EMPTY = /<(hp|hs):t((?:\s[^>]*)?)><\/\1:t>/g;
|
|
12
22
|
class HwpxDocument {
|
|
13
23
|
constructor(id, path, zip, content, format) {
|
|
14
24
|
this._isDirty = false;
|
|
25
|
+
/**
|
|
26
|
+
* True once the section element list has changed shape since the XML was
|
|
27
|
+
* parsed (insert/delete/copy/move of paragraphs, tables or images).
|
|
28
|
+
* Parsed XML offsets are unusable from that point until the next save.
|
|
29
|
+
*/
|
|
30
|
+
this._structureChanged = false;
|
|
15
31
|
this._undoStack = [];
|
|
16
32
|
this._redoStack = [];
|
|
17
33
|
this._pendingTextReplacements = [];
|
|
18
34
|
this._pendingDirectTextUpdates = [];
|
|
35
|
+
/**
|
|
36
|
+
* `col` is the cell's position in the memory row; `colAddr` is its grid column.
|
|
37
|
+
* They differ after a merge: memory keeps covered cells, the XML drops them.
|
|
38
|
+
* The XML writer finds the target by colAddr so a write made after a merge
|
|
39
|
+
* lands in the right cell (writes now replay in call order).
|
|
40
|
+
*/
|
|
19
41
|
this._pendingTableCellUpdates = [];
|
|
20
42
|
this._pendingNestedTableInserts = [];
|
|
21
43
|
this._pendingImageInserts = [];
|
|
22
44
|
this._pendingCellImageInserts = [];
|
|
23
45
|
this._pendingTableInserts = [];
|
|
24
|
-
|
|
46
|
+
/**
|
|
47
|
+
* Monotonic counter shared by paragraph and table inserts. Both kinds are
|
|
48
|
+
* replayed into XML in this order so each insert sees exactly the elements
|
|
49
|
+
* that existed when it was made. Replaying all tables before all paragraphs
|
|
50
|
+
* wrote "A, table, A-2, table" to disk as "A, A-2, table, table".
|
|
51
|
+
*/
|
|
52
|
+
this._tableInsertCounter = 0;
|
|
25
53
|
this._pendingImageDeletes = [];
|
|
26
54
|
this._pendingTableDeletes = [];
|
|
27
55
|
this._pendingParagraphDeletes = [];
|
|
@@ -36,10 +64,25 @@ class HwpxDocument {
|
|
|
36
64
|
this._pendingTableRowDeletes = [];
|
|
37
65
|
this._pendingTableColumnInserts = [];
|
|
38
66
|
this._pendingTableColumnDeletes = [];
|
|
67
|
+
/**
|
|
68
|
+
* Call order of every pending edit that names a table cell or row/column by
|
|
69
|
+
* index. Each such index is relative to the table as it was at call time,
|
|
70
|
+
* so save must replay these edits in call order (applyTableOpsInCallOrder).
|
|
71
|
+
* A WeakMap keeps the queue element types unchanged and drops entries with
|
|
72
|
+
* their ops (undo, section delete).
|
|
73
|
+
*/
|
|
74
|
+
this._tableOpSeq = new WeakMap();
|
|
75
|
+
this._tableOpCounter = 0;
|
|
39
76
|
this._pendingParagraphCopies = [];
|
|
40
77
|
this._pendingParagraphMoves = [];
|
|
41
78
|
this._pendingHeaderUpdates = [];
|
|
42
79
|
this._pendingFooterUpdates = [];
|
|
80
|
+
/**
|
|
81
|
+
* New sections to materialise as Contents/sectionN.xml on save, in call
|
|
82
|
+
* order. `templateFrom` is the section whose <hp:secPr> (page size, margins)
|
|
83
|
+
* the new section copies — Hancom's own "insert section" does the same.
|
|
84
|
+
*/
|
|
85
|
+
this._pendingSectionOps = [];
|
|
43
86
|
// Cache for character properties (id → font size in pt)
|
|
44
87
|
this._charPrCache = null;
|
|
45
88
|
// Private: Pending table move/copy operations
|
|
@@ -127,7 +170,10 @@ class HwpxDocument {
|
|
|
127
170
|
elements: [{
|
|
128
171
|
type: 'paragraph',
|
|
129
172
|
data: {
|
|
130
|
-
id
|
|
173
|
+
// Must match the id written into Contents/section0.xml below. Later
|
|
174
|
+
// inserts anchor on this id; a random value here pointed at a node
|
|
175
|
+
// that does not exist in the XML.
|
|
176
|
+
id: '0',
|
|
131
177
|
runs: [{ text: '' }],
|
|
132
178
|
},
|
|
133
179
|
}],
|
|
@@ -227,14 +273,29 @@ class HwpxDocument {
|
|
|
227
273
|
zip.file('Contents/section0.xml', `<?xml version="1.0" encoding="UTF-8" standalone="yes" ?><hs:sec xmlns:ha="http://www.hancom.co.kr/hwpml/2011/app" xmlns:hp="http://www.hancom.co.kr/hwpml/2011/paragraph" xmlns:hp10="http://www.hancom.co.kr/hwpml/2016/paragraph" xmlns:hs="http://www.hancom.co.kr/hwpml/2011/section" xmlns:hc="http://www.hancom.co.kr/hwpml/2011/core" xmlns:hh="http://www.hancom.co.kr/hwpml/2011/head" xmlns:hhs="http://www.hancom.co.kr/hwpml/2011/history" xmlns:hm="http://www.hancom.co.kr/hwpml/2011/master-page" xmlns:hpf="http://www.hancom.co.kr/schema/2011/hpf" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:opf="http://www.idpf.org/2007/opf/" xmlns:ooxmlchart="http://www.hancom.co.kr/hwpml/2016/ooxmlchart" xmlns:hwpunitchar="http://www.hancom.co.kr/hwpml/2016/HwpUnitChar" xmlns:epub="http://www.idpf.org/2007/ops" xmlns:config="urn:oasis:names:tc:opendocument:xmlns:config:1.0"><hp:p id="0" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0"><hp:secPr id="" textDirection="HORIZONTAL" spaceColumns="1134" tabStop="8000" tabStopVal="4000" tabStopUnit="HWPUNIT" outlineShapeIDRef="1" memoShapeIDRef="0" textVerticalWidthHead="0" masterPageCnt="0"><hp:grid lineGrid="0" charGrid="0" wongoji="0"/><hp:startNum pageStartsOn="BOTH" page="0" pic="0" tbl="0" equation="0"/><hp:visibility hideFirstHeader="0" hideFirstFooter="0" hideFirstMasterPage="0" border="SHOW_ALL" fill="SHOW_ALL" hideFirstPageNum="0" hideFirstEmptyLine="0" showLineNumber="0"/><hp:pagePr landscape="0" width="59528" height="84188" gutterType="LEFT_ONLY"><hp:pageMar header="4252" footer="4252" left="8504" right="8504" top="5668" bottom="4252" gutter="0"/></hp:pagePr><hp:footNotePr><hp:autoNumFormat type="DIGIT"/><hp:noteLine length="-1" type="SOLID" width="0.12mm" color="#000000"/><hp:noteSpacing aboveLine="850" belowLine="567" betweenNotes="283"/><hp:numbering type="CONTINUOUS" newNum="1"/><hp:placement place="EACH_COLUMN" beneathText="0"/></hp:footNotePr><hp:endNotePr><hp:autoNumFormat type="DIGIT"/><hp:noteLine length="14692" type="SOLID" width="0.12mm" color="#000000"/><hp:noteSpacing aboveLine="850" belowLine="567" betweenNotes="0"/><hp:numbering type="CONTINUOUS" newNum="1"/><hp:placement place="END_OF_DOCUMENT" beneathText="0"/></hp:endNotePr></hp:secPr><hp:t></hp:t></hp:run></hp:p></hs:sec>`);
|
|
228
274
|
// Create empty BinData folder
|
|
229
275
|
zip.folder('BinData');
|
|
230
|
-
|
|
276
|
+
// A new document has no location on disk yet. Seeding a bare filename here
|
|
277
|
+
// made save_document resolve it against the server process cwd, so callers
|
|
278
|
+
// could not find the file they had just written.
|
|
279
|
+
return new HwpxDocument(id, '', zip, content, 'hwpx');
|
|
231
280
|
}
|
|
232
281
|
get id() { return this._id; }
|
|
233
282
|
get path() { return this._path; }
|
|
283
|
+
/** True once the document has a real location on disk. */
|
|
284
|
+
get hasPath() { return this._path.length > 0; }
|
|
285
|
+
/**
|
|
286
|
+
* Record where the document now lives after a successful write, so the next
|
|
287
|
+
* save without an explicit path targets the same file.
|
|
288
|
+
*/
|
|
289
|
+
setPath(newPath) { this._path = newPath; }
|
|
234
290
|
get format() { return this._format; }
|
|
235
291
|
get isDirty() { return this._isDirty; }
|
|
236
292
|
get zip() { return this._zip; }
|
|
237
293
|
get content() { return this._content; }
|
|
294
|
+
/** Push a table-structure or table-cell edit and remember its call order. */
|
|
295
|
+
queueTableOp(queue, op) {
|
|
296
|
+
this._tableOpSeq.set(op, ++this._tableOpCounter);
|
|
297
|
+
queue.push(op);
|
|
298
|
+
}
|
|
238
299
|
// ============================================================
|
|
239
300
|
// Undo/Redo
|
|
240
301
|
// ============================================================
|
|
@@ -256,6 +317,8 @@ class HwpxDocument {
|
|
|
256
317
|
const parsed = JSON.parse(state);
|
|
257
318
|
this._content.sections = parsed.sections;
|
|
258
319
|
this._content.metadata = parsed.metadata;
|
|
320
|
+
// Undo/redo swaps in a whole element list; parse-time offsets no longer apply.
|
|
321
|
+
this.markStructureChanged();
|
|
259
322
|
}
|
|
260
323
|
canUndo() { return this._undoStack.length > 0; }
|
|
261
324
|
canRedo() { return this._redoStack.length > 0; }
|
|
@@ -311,6 +374,7 @@ class HwpxDocument {
|
|
|
311
374
|
this._pendingParagraphMoves = [];
|
|
312
375
|
this._pendingHeaderUpdates = [];
|
|
313
376
|
this._pendingFooterUpdates = [];
|
|
377
|
+
this._pendingSectionOps = [];
|
|
314
378
|
if (this._pendingTableMoves)
|
|
315
379
|
this._pendingTableMoves = [];
|
|
316
380
|
}
|
|
@@ -322,6 +386,60 @@ class HwpxDocument {
|
|
|
322
386
|
this._isDirty = true;
|
|
323
387
|
this.invalidateReadingCache();
|
|
324
388
|
}
|
|
389
|
+
/**
|
|
390
|
+
* Record that the section element list changed shape. Call from every method
|
|
391
|
+
* that inserts, removes, copies or moves a section-level element. Once set,
|
|
392
|
+
* paragraph edits stop trusting offsets cached at parse time and locate their
|
|
393
|
+
* target in the current XML instead.
|
|
394
|
+
*/
|
|
395
|
+
markStructureChanged() {
|
|
396
|
+
this._structureChanged = true;
|
|
397
|
+
}
|
|
398
|
+
/**
|
|
399
|
+
* Resolve "after element N" into an id-based anchor using the memory model
|
|
400
|
+
* as it is right now (before the new element is spliced in).
|
|
401
|
+
*
|
|
402
|
+
* Returns null for "before everything" (N < 0). Elements without an XML
|
|
403
|
+
* paragraph/table of their own (images, shapes) are skipped backwards to the
|
|
404
|
+
* nearest paragraph or table, which is what the XML placement needs.
|
|
405
|
+
*
|
|
406
|
+
* The parser turns a paragraph that is only a line of ─/━/═ into an 'hr'
|
|
407
|
+
* element with a fresh id. That paragraph is still in the XML, so it still
|
|
408
|
+
* takes an occurrence slot there: skipping it here put later anchors one
|
|
409
|
+
* paragraph early (measured on Hancom files with divider lines).
|
|
410
|
+
*/
|
|
411
|
+
resolveElementAnchor(sectionIndex, afterElementIndex) {
|
|
412
|
+
const elements = this._content.sections[sectionIndex]?.elements ?? [];
|
|
413
|
+
for (let i = Math.min(afterElementIndex, elements.length - 1); i >= 0; i--) {
|
|
414
|
+
const key = this.anchorKeyOf(elements[i]);
|
|
415
|
+
if (!key)
|
|
416
|
+
continue;
|
|
417
|
+
let occurrence = 0;
|
|
418
|
+
for (let j = 0; j < i; j++) {
|
|
419
|
+
const other = this.anchorKeyOf(elements[j]);
|
|
420
|
+
if (other && other.kind === key.kind && other.id === key.id)
|
|
421
|
+
occurrence++;
|
|
422
|
+
}
|
|
423
|
+
return { ...key, occurrence };
|
|
424
|
+
}
|
|
425
|
+
return null;
|
|
426
|
+
}
|
|
427
|
+
/**
|
|
428
|
+
* The XML node a memory element stands for, or null if it has none of its
|
|
429
|
+
* own. An 'hr' parsed from a divider paragraph stands for that paragraph.
|
|
430
|
+
*/
|
|
431
|
+
anchorKeyOf(el) {
|
|
432
|
+
if (!el)
|
|
433
|
+
return null;
|
|
434
|
+
if (el.type === 'hr') {
|
|
435
|
+
const src = el.data.sourceParagraphId;
|
|
436
|
+
return src ? { kind: 'paragraph', id: String(src) } : null;
|
|
437
|
+
}
|
|
438
|
+
if (el.type !== 'paragraph' && el.type !== 'table')
|
|
439
|
+
return null;
|
|
440
|
+
const id = String(el.data.id ?? '');
|
|
441
|
+
return id ? { kind: el.type, id } : null;
|
|
442
|
+
}
|
|
325
443
|
// ============================================================
|
|
326
444
|
// Content Access
|
|
327
445
|
// ============================================================
|
|
@@ -389,6 +507,8 @@ class HwpxDocument {
|
|
|
389
507
|
index: ei,
|
|
390
508
|
text: el.data.runs.map(r => r.text).join(''),
|
|
391
509
|
style: el.data.paraStyle,
|
|
510
|
+
paraPrIDRef: el.data.paraPrId,
|
|
511
|
+
charPrIDRef: el.data.runs.find(r => r.charPrIDRef !== undefined)?.charPrIDRef,
|
|
392
512
|
});
|
|
393
513
|
}
|
|
394
514
|
});
|
|
@@ -403,17 +523,32 @@ class HwpxDocument {
|
|
|
403
523
|
text: para.runs.map(r => r.text).join(''),
|
|
404
524
|
runs: para.runs,
|
|
405
525
|
style: para.paraStyle,
|
|
526
|
+
// Raw header.xml references, so callers can build XML without scraping it.
|
|
527
|
+
paraPrIDRef: para.paraPrId,
|
|
528
|
+
charPrIDRef: para.runs.find(r => r.charPrIDRef !== undefined)?.charPrIDRef,
|
|
406
529
|
};
|
|
407
530
|
}
|
|
408
531
|
updateParagraphText(sectionIndex, elementIndex, runIndex, text) {
|
|
409
|
-
const
|
|
410
|
-
if (!
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
if (
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
532
|
+
const section = this._content.sections[sectionIndex];
|
|
533
|
+
if (!section)
|
|
534
|
+
throw new Error(`Section ${sectionIndex} does not exist.`);
|
|
535
|
+
const element = section.elements[elementIndex];
|
|
536
|
+
if (!element) {
|
|
537
|
+
throw new Error(`Element ${elementIndex} does not exist in section ${sectionIndex} (${section.elements.length} elements).`);
|
|
538
|
+
}
|
|
539
|
+
if (element.type !== 'paragraph') {
|
|
540
|
+
// Reported 2026-09-24: aimed at a table, this answered "Paragraph updated"
|
|
541
|
+
// and changed nothing. Say what is there instead.
|
|
542
|
+
throw new Error(`Element ${elementIndex} in section ${sectionIndex} is a ${element.type}, not a paragraph. ` +
|
|
543
|
+
(element.type === 'table' ? 'Use update_table_cell to change table text.' : 'It has no paragraph text to replace.'));
|
|
544
|
+
}
|
|
545
|
+
const paragraph = element.data;
|
|
546
|
+
// Replacing run 0 means "replace the whole paragraph": the new text goes
|
|
547
|
+
// into the first run and every other run is emptied, so the result takes
|
|
548
|
+
// the first run's character shape. Spreading the text across the old runs
|
|
549
|
+
// (preserve-styles) instead gave the tail of the sentence whatever shape
|
|
550
|
+
// those runs had — reported 2026-09-24: plain + bold paragraph, replaced
|
|
551
|
+
// wholesale, came out bold from the third line on.
|
|
417
552
|
// Handle case where paragraph has no runs (e.g., run without hp:t tag)
|
|
418
553
|
// We need to create a run in memory and track the update for XML modification
|
|
419
554
|
if (!paragraph.runs[runIndex]) {
|
|
@@ -436,6 +571,7 @@ class HwpxDocument {
|
|
|
436
571
|
elementIndex,
|
|
437
572
|
paragraphId: paragraph.id || '', // Use stable paragraph ID for reliable identification
|
|
438
573
|
paragraphOccurrence,
|
|
574
|
+
paragraph,
|
|
439
575
|
runIndex,
|
|
440
576
|
oldText,
|
|
441
577
|
newText: text
|
|
@@ -450,6 +586,7 @@ class HwpxDocument {
|
|
|
450
586
|
elementIndex,
|
|
451
587
|
paragraphId: paragraph.id || '',
|
|
452
588
|
paragraphOccurrence,
|
|
589
|
+
paragraph,
|
|
453
590
|
runIndex: i,
|
|
454
591
|
oldText: otherOldText,
|
|
455
592
|
newText: '' // Clear other runs
|
|
@@ -530,6 +667,7 @@ class HwpxDocument {
|
|
|
530
667
|
elementIndex,
|
|
531
668
|
paragraphId: paragraph.id || '',
|
|
532
669
|
paragraphOccurrence,
|
|
670
|
+
paragraph,
|
|
533
671
|
runIndex: i,
|
|
534
672
|
oldText: oldText || '',
|
|
535
673
|
newText: run.text
|
|
@@ -549,12 +687,17 @@ class HwpxDocument {
|
|
|
549
687
|
id: paragraphId,
|
|
550
688
|
runs: [{ text }],
|
|
551
689
|
};
|
|
690
|
+
// Resolve the XML position before the new paragraph joins the element list.
|
|
691
|
+
const anchor = this.resolveElementAnchor(sectionIndex, afterElementIndex);
|
|
552
692
|
const newElement = { type: 'paragraph', data: newParagraph };
|
|
553
693
|
section.elements.splice(afterElementIndex + 1, 0, newElement);
|
|
694
|
+
this.markStructureChanged();
|
|
554
695
|
// Add to pending list for XML sync
|
|
555
696
|
this._pendingParagraphInserts.push({
|
|
556
697
|
sectionIndex,
|
|
557
698
|
afterElementIndex,
|
|
699
|
+
anchor,
|
|
700
|
+
insertOrder: this._tableInsertCounter++,
|
|
558
701
|
paragraphId,
|
|
559
702
|
text,
|
|
560
703
|
});
|
|
@@ -578,6 +721,7 @@ class HwpxDocument {
|
|
|
578
721
|
});
|
|
579
722
|
// Remove from memory
|
|
580
723
|
section.elements.splice(elementIndex, 1);
|
|
724
|
+
this.markStructureChanged();
|
|
581
725
|
this.markModified();
|
|
582
726
|
this.invalidateReadingCache();
|
|
583
727
|
return true;
|
|
@@ -600,6 +744,7 @@ class HwpxDocument {
|
|
|
600
744
|
elementIndex,
|
|
601
745
|
paragraphId: paragraph.id || '', // Use stable paragraph ID
|
|
602
746
|
paragraphOccurrence,
|
|
747
|
+
paragraph,
|
|
603
748
|
runIndex: lastRunIndex,
|
|
604
749
|
oldText,
|
|
605
750
|
newText
|
|
@@ -618,6 +763,7 @@ class HwpxDocument {
|
|
|
618
763
|
elementIndex,
|
|
619
764
|
paragraphId: paragraph.id || '', // Use stable paragraph ID
|
|
620
765
|
paragraphOccurrence,
|
|
766
|
+
paragraph,
|
|
621
767
|
runIndex: 0,
|
|
622
768
|
oldText: '',
|
|
623
769
|
newText: text
|
|
@@ -986,7 +1132,7 @@ class HwpxDocument {
|
|
|
986
1132
|
this._pendingTableCellHangingIndents[existingIdx].indentPt = indentPt;
|
|
987
1133
|
}
|
|
988
1134
|
else {
|
|
989
|
-
this._pendingTableCellHangingIndents
|
|
1135
|
+
this.queueTableOp(this._pendingTableCellHangingIndents, {
|
|
990
1136
|
sectionIndex,
|
|
991
1137
|
tableIndex,
|
|
992
1138
|
row,
|
|
@@ -1065,7 +1211,7 @@ class HwpxDocument {
|
|
|
1065
1211
|
this._pendingTableCellHangingIndents[existingIdx].indentPt = 0; // 0 means remove
|
|
1066
1212
|
}
|
|
1067
1213
|
else {
|
|
1068
|
-
this._pendingTableCellHangingIndents
|
|
1214
|
+
this.queueTableOp(this._pendingTableCellHangingIndents, {
|
|
1069
1215
|
sectionIndex,
|
|
1070
1216
|
tableIndex,
|
|
1071
1217
|
row,
|
|
@@ -1181,12 +1327,22 @@ class HwpxDocument {
|
|
|
1181
1327
|
/**
|
|
1182
1328
|
* Get table map with headers - maps table indices to their header paragraphs
|
|
1183
1329
|
* Returns array of table info including the header text from the preceding paragraph
|
|
1330
|
+
*
|
|
1331
|
+
* Two indices are returned because they differ once a document has more than
|
|
1332
|
+
* one section:
|
|
1333
|
+
* - `table_index_in_section` — what every table tool (update_table_cell,
|
|
1334
|
+
* get_table_cell, insert_table_row, …) expects together with
|
|
1335
|
+
* `section_index`. Use this one.
|
|
1336
|
+
* - `table_index` — position across the whole document, kept for callers
|
|
1337
|
+
* that list tables. Passing it to a table tool in section 1+ addresses a
|
|
1338
|
+
* DIFFERENT table (reported 2026-09-24: map said 5, the tool needed 4).
|
|
1184
1339
|
*/
|
|
1185
1340
|
getTableMap() {
|
|
1186
1341
|
const result = [];
|
|
1187
1342
|
let globalTableIndex = 0;
|
|
1188
1343
|
this._content.sections.forEach((section, sectionIndex) => {
|
|
1189
1344
|
let lastParagraphText = '';
|
|
1345
|
+
let sectionTableIndex = 0;
|
|
1190
1346
|
section.elements.forEach((element, _elementIndex) => {
|
|
1191
1347
|
if (element.type === 'paragraph') {
|
|
1192
1348
|
// Store the paragraph text as potential header
|
|
@@ -1209,6 +1365,7 @@ class HwpxDocument {
|
|
|
1209
1365
|
}) || [];
|
|
1210
1366
|
result.push({
|
|
1211
1367
|
table_index: globalTableIndex,
|
|
1368
|
+
table_index_in_section: sectionTableIndex,
|
|
1212
1369
|
section_index: sectionIndex,
|
|
1213
1370
|
header: lastParagraphText,
|
|
1214
1371
|
rows,
|
|
@@ -1217,6 +1374,7 @@ class HwpxDocument {
|
|
|
1217
1374
|
first_row_preview: firstRowPreview,
|
|
1218
1375
|
});
|
|
1219
1376
|
globalTableIndex++;
|
|
1377
|
+
sectionTableIndex++;
|
|
1220
1378
|
// Don't reset lastParagraphText here - next table might reuse same header if consecutive
|
|
1221
1379
|
}
|
|
1222
1380
|
});
|
|
@@ -1391,8 +1549,15 @@ class HwpxDocument {
|
|
|
1391
1549
|
// Get previous value
|
|
1392
1550
|
const cellData = this.getTableCell(tableInfo.section_index, tableInfo.local_index, position.row, position.col);
|
|
1393
1551
|
const previousValue = cellData?.text || '';
|
|
1394
|
-
// Update the cell
|
|
1395
|
-
|
|
1552
|
+
// Update the cell. A cell covered by a merge throws; treat it as a failed
|
|
1553
|
+
// path rather than aborting the remaining entries.
|
|
1554
|
+
let updated = false;
|
|
1555
|
+
try {
|
|
1556
|
+
updated = this.updateTableCell(tableInfo.section_index, tableInfo.local_index, position.row, position.col, value);
|
|
1557
|
+
}
|
|
1558
|
+
catch {
|
|
1559
|
+
updated = false;
|
|
1560
|
+
}
|
|
1396
1561
|
if (updated) {
|
|
1397
1562
|
result.success++;
|
|
1398
1563
|
result.details.push({
|
|
@@ -1531,6 +1696,7 @@ class HwpxDocument {
|
|
|
1531
1696
|
const result = {
|
|
1532
1697
|
success: 0,
|
|
1533
1698
|
outOfBounds: [],
|
|
1699
|
+
failed: [],
|
|
1534
1700
|
updated: [],
|
|
1535
1701
|
};
|
|
1536
1702
|
const tableInfo = this.convertGlobalToLocalTableIndex(tableIndex);
|
|
@@ -1555,8 +1721,16 @@ class HwpxDocument {
|
|
|
1555
1721
|
// Get previous value
|
|
1556
1722
|
const cellData = this.getTableCell(tableInfo.section_index, tableInfo.local_index, targetRow, targetCol);
|
|
1557
1723
|
const previousValue = cellData?.text || '';
|
|
1558
|
-
// Update cell
|
|
1559
|
-
|
|
1724
|
+
// Update cell. A cell covered by a merge throws; record it and keep
|
|
1725
|
+
// going so one merged position does not discard the whole batch.
|
|
1726
|
+
let updated = false;
|
|
1727
|
+
let failure = '';
|
|
1728
|
+
try {
|
|
1729
|
+
updated = this.updateTableCell(tableInfo.section_index, tableInfo.local_index, targetRow, targetCol, value);
|
|
1730
|
+
}
|
|
1731
|
+
catch (err) {
|
|
1732
|
+
failure = err instanceof Error ? err.message : String(err);
|
|
1733
|
+
}
|
|
1560
1734
|
if (updated) {
|
|
1561
1735
|
result.success++;
|
|
1562
1736
|
result.updated.push({
|
|
@@ -1566,6 +1740,14 @@ class HwpxDocument {
|
|
|
1566
1740
|
newValue: value,
|
|
1567
1741
|
});
|
|
1568
1742
|
}
|
|
1743
|
+
else {
|
|
1744
|
+
result.failed.push({
|
|
1745
|
+
row: targetRow,
|
|
1746
|
+
col: targetCol,
|
|
1747
|
+
value,
|
|
1748
|
+
error: failure || 'Cell update failed',
|
|
1749
|
+
});
|
|
1750
|
+
}
|
|
1569
1751
|
}
|
|
1570
1752
|
}
|
|
1571
1753
|
return result;
|
|
@@ -1990,6 +2172,43 @@ class HwpxDocument {
|
|
|
1990
2172
|
cell,
|
|
1991
2173
|
};
|
|
1992
2174
|
}
|
|
2175
|
+
/**
|
|
2176
|
+
* Find the merged cell that covers (row, col), if that position is not itself
|
|
2177
|
+
* a master cell. A covered cell has no <hp:tc> of its own in the saved XML,
|
|
2178
|
+
* so writing to it succeeds in memory and then silently vanishes on save.
|
|
2179
|
+
*/
|
|
2180
|
+
findCoveringMergedCell(table, row, col) {
|
|
2181
|
+
const rows = table.rows;
|
|
2182
|
+
if (!rows)
|
|
2183
|
+
return null;
|
|
2184
|
+
const target = rows[row]?.cells?.[col];
|
|
2185
|
+
if (target && ((target.colSpan ?? 1) > 1 || (target.rowSpan ?? 1) > 1)) {
|
|
2186
|
+
return null; // the position is a master cell
|
|
2187
|
+
}
|
|
2188
|
+
// Merged cells can only originate at or before (row, col), and a table with
|
|
2189
|
+
// no spans at all — the common case — exits on the first row scan.
|
|
2190
|
+
for (let r = 0; r <= row && r < rows.length; r++) {
|
|
2191
|
+
const cells = rows[r]?.cells;
|
|
2192
|
+
if (!cells)
|
|
2193
|
+
continue;
|
|
2194
|
+
const lastCol = Math.min(col, cells.length - 1);
|
|
2195
|
+
for (let c = 0; c <= lastCol; c++) {
|
|
2196
|
+
const cell = cells[c];
|
|
2197
|
+
if (!cell)
|
|
2198
|
+
continue;
|
|
2199
|
+
const rowSpan = cell.rowSpan ?? 1;
|
|
2200
|
+
const colSpan = cell.colSpan ?? 1;
|
|
2201
|
+
if (rowSpan <= 1 && colSpan <= 1)
|
|
2202
|
+
continue;
|
|
2203
|
+
if (r === row && c === col)
|
|
2204
|
+
continue;
|
|
2205
|
+
if (row < r + rowSpan && col < c + colSpan) {
|
|
2206
|
+
return { row: r, col: c };
|
|
2207
|
+
}
|
|
2208
|
+
}
|
|
2209
|
+
}
|
|
2210
|
+
return null;
|
|
2211
|
+
}
|
|
1993
2212
|
updateTableCell(sectionIndex, tableIndex, row, col, text, charShapeId) {
|
|
1994
2213
|
const table = this.findTable(sectionIndex, tableIndex);
|
|
1995
2214
|
if (!table)
|
|
@@ -1997,10 +2216,16 @@ class HwpxDocument {
|
|
|
1997
2216
|
const cell = table.rows[row]?.cells[col];
|
|
1998
2217
|
if (!cell)
|
|
1999
2218
|
return false;
|
|
2219
|
+
// Refuse instead of reporting success and losing the text at save time.
|
|
2220
|
+
const covering = this.findCoveringMergedCell(table, row, col);
|
|
2221
|
+
if (covering) {
|
|
2222
|
+
throw new Error(`Cell (${row}, ${col}) is covered by the merged cell at ` +
|
|
2223
|
+
`(${covering.row}, ${covering.col}); write to the master cell instead`);
|
|
2224
|
+
}
|
|
2000
2225
|
// Track cell update for XML sync (works for both empty and non-empty cells)
|
|
2001
2226
|
// Store table ID for reliable XML matching
|
|
2002
2227
|
// charShapeId is optional - if provided, it will override the existing charPrIDRef
|
|
2003
|
-
this._pendingTableCellUpdates
|
|
2228
|
+
this.queueTableOp(this._pendingTableCellUpdates, { sectionIndex, tableIndex, tableId: table.id, row, col, colAddr: cell.colAddr, text, charShapeId });
|
|
2004
2229
|
this.saveState();
|
|
2005
2230
|
if (cell.paragraphs.length > 0 && cell.paragraphs[0].runs.length > 0) {
|
|
2006
2231
|
cell.paragraphs[0].runs[0].text = text;
|
|
@@ -2027,19 +2252,80 @@ class HwpxDocument {
|
|
|
2027
2252
|
const table = this.findTable(sectionIndex, tableIndex);
|
|
2028
2253
|
if (!table || !table.rows[afterRowIndex])
|
|
2029
2254
|
return false;
|
|
2255
|
+
// A new row between afterRowIndex and afterRowIndex+1 must not cut through
|
|
2256
|
+
// a vertical merge. Cloning a row that holds a rowSpan>1 master, or one that
|
|
2257
|
+
// sits inside such a span, copied the span into the gap and made the merged
|
|
2258
|
+
// area overlap the new row (reported 2026-09-24: rowSpan=2 header, after_row 0).
|
|
2259
|
+
for (const row of table.rows) {
|
|
2260
|
+
for (const cell of row.cells) {
|
|
2261
|
+
const top = cell.rowAddr ?? table.rows.indexOf(row);
|
|
2262
|
+
const span = cell.rowSpan ?? 1;
|
|
2263
|
+
if (span > 1 && top <= afterRowIndex && afterRowIndex < top + span - 1) {
|
|
2264
|
+
throw new Error(`Cannot insert a row after row ${afterRowIndex}: it would split the merged cell at ` +
|
|
2265
|
+
`(${top}, ${cell.colAddr ?? 0}) that spans rows ${top}-${top + span - 1}. ` +
|
|
2266
|
+
`Insert after row ${top + span - 1} instead, or unmerge first.`);
|
|
2267
|
+
}
|
|
2268
|
+
}
|
|
2269
|
+
}
|
|
2030
2270
|
this.saveState();
|
|
2031
|
-
|
|
2032
|
-
|
|
2271
|
+
// Same column grid as the XML path (gridCellsForNewRow): one cell per
|
|
2272
|
+
// column position, taking the colAddr/colSpan of the cell that starts
|
|
2273
|
+
// there in the template row or the nearest row above. Sizing the row by
|
|
2274
|
+
// templateRow.cells.length left out a column covered by a vertical merge.
|
|
2275
|
+
// A cell with no colAddr (e.g. added by insertTableColumn, which does not
|
|
2276
|
+
// renumber) is placed by its position in the row, as gridCellsForNewRow
|
|
2277
|
+
// does for the XML. Dropping it made the new row one cell short.
|
|
2278
|
+
const placed = (r) => {
|
|
2279
|
+
let next = 0;
|
|
2280
|
+
return (table.rows[r]?.cells ?? []).map(c => {
|
|
2281
|
+
const span = c.colSpan ?? 1;
|
|
2282
|
+
const col = c.colAddr ?? next;
|
|
2283
|
+
next = col + span;
|
|
2284
|
+
return { col, span };
|
|
2285
|
+
});
|
|
2286
|
+
};
|
|
2287
|
+
const starts = (r) => new Map(placed(r).map(c => [c.col, c.span]));
|
|
2288
|
+
const colCount = Math.max(0, ...table.rows.map((_, r) => Math.max(0, ...placed(r).map(c => c.col + c.span))));
|
|
2289
|
+
const templateStarts = placed(afterRowIndex).map(c => c.col).sort((a, b) => a - b);
|
|
2290
|
+
const grid = [];
|
|
2291
|
+
for (let col = 0; col < colCount;) {
|
|
2292
|
+
let span;
|
|
2293
|
+
for (let r = afterRowIndex; r >= 0 && span === undefined; r--)
|
|
2294
|
+
span = starts(r).get(col);
|
|
2295
|
+
for (let r = afterRowIndex + 1; r < table.rows.length && span === undefined; r++)
|
|
2296
|
+
span = starts(r).get(col);
|
|
2297
|
+
if (span === undefined) {
|
|
2298
|
+
col++;
|
|
2299
|
+
continue;
|
|
2300
|
+
}
|
|
2301
|
+
const next = templateStarts.find(c => c > col);
|
|
2302
|
+
if (next !== undefined && col + span > next)
|
|
2303
|
+
span = next - col;
|
|
2304
|
+
grid.push({ colAddr: col, colSpan: Math.max(1, span) });
|
|
2305
|
+
col += Math.max(1, span);
|
|
2306
|
+
}
|
|
2033
2307
|
const newRow = {
|
|
2034
|
-
cells:
|
|
2308
|
+
cells: grid.map((g, i) => ({
|
|
2309
|
+
rowAddr: afterRowIndex + 1,
|
|
2310
|
+
colAddr: g.colAddr,
|
|
2311
|
+
rowSpan: 1,
|
|
2312
|
+
colSpan: g.colSpan,
|
|
2035
2313
|
paragraphs: [{
|
|
2036
2314
|
id: Math.random().toString(36).substring(2, 11),
|
|
2037
2315
|
runs: [{ text: cellTexts?.[i] || '' }],
|
|
2038
2316
|
}],
|
|
2039
2317
|
})),
|
|
2040
2318
|
};
|
|
2319
|
+
// Keep memory row addresses in step with the XML renumbering, so a later
|
|
2320
|
+
// merge/split/insert on this table reads the right rows.
|
|
2321
|
+
for (const row of table.rows) {
|
|
2322
|
+
for (const cell of row.cells) {
|
|
2323
|
+
if (cell.rowAddr !== undefined && cell.rowAddr > afterRowIndex)
|
|
2324
|
+
cell.rowAddr += 1;
|
|
2325
|
+
}
|
|
2326
|
+
}
|
|
2041
2327
|
table.rows.splice(afterRowIndex + 1, 0, newRow);
|
|
2042
|
-
this._pendingTableRowInserts
|
|
2328
|
+
this.queueTableOp(this._pendingTableRowInserts, {
|
|
2043
2329
|
sectionIndex,
|
|
2044
2330
|
tableIndex,
|
|
2045
2331
|
afterRowIndex,
|
|
@@ -2057,8 +2343,27 @@ class HwpxDocument {
|
|
|
2057
2343
|
return this.deleteTable(sectionIndex, tableIndex);
|
|
2058
2344
|
}
|
|
2059
2345
|
this.saveState();
|
|
2346
|
+
// Mirror applyTableRowDeletesToXml so later edits read the same addresses the
|
|
2347
|
+
// XML has after replay: a vertical merge from an earlier row that reaches the
|
|
2348
|
+
// deleted row loses one row, and cells below move up one row. Stale rowAddr
|
|
2349
|
+
// made the row-insert guard refuse an insert below a merge and allow one
|
|
2350
|
+
// through it (CodeRabbit, 2026-09-24).
|
|
2351
|
+
for (let r = 0; r < rowIndex; r++) {
|
|
2352
|
+
for (const cell of table.rows[r]?.cells ?? []) {
|
|
2353
|
+
const top = cell.rowAddr ?? r;
|
|
2354
|
+
const span = cell.rowSpan ?? 1;
|
|
2355
|
+
if (span > 1 && top + span > rowIndex)
|
|
2356
|
+
cell.rowSpan = span - 1;
|
|
2357
|
+
}
|
|
2358
|
+
}
|
|
2060
2359
|
table.rows.splice(rowIndex, 1);
|
|
2061
|
-
|
|
2360
|
+
for (const row of table.rows) {
|
|
2361
|
+
for (const cell of row.cells) {
|
|
2362
|
+
if (cell.rowAddr !== undefined && cell.rowAddr > rowIndex)
|
|
2363
|
+
cell.rowAddr -= 1;
|
|
2364
|
+
}
|
|
2365
|
+
}
|
|
2366
|
+
this.queueTableOp(this._pendingTableRowDeletes, {
|
|
2062
2367
|
sectionIndex,
|
|
2063
2368
|
tableIndex,
|
|
2064
2369
|
rowIndex,
|
|
@@ -2099,6 +2404,7 @@ class HwpxDocument {
|
|
|
2099
2404
|
});
|
|
2100
2405
|
// Remove from memory model
|
|
2101
2406
|
section.elements.splice(elementIndex, 1);
|
|
2407
|
+
this.markStructureChanged();
|
|
2102
2408
|
this.markModified();
|
|
2103
2409
|
return true;
|
|
2104
2410
|
}
|
|
@@ -2107,15 +2413,29 @@ class HwpxDocument {
|
|
|
2107
2413
|
if (!table)
|
|
2108
2414
|
return false;
|
|
2109
2415
|
this.saveState();
|
|
2416
|
+
// Keep memory addresses in step with the XML path (applyTableColumnInsertsToXml
|
|
2417
|
+
// gives the new cell colAddr afterColIndex+1 and shifts the cells after it).
|
|
2418
|
+
// A new cell with no colAddr in the middle of a row made the row read as
|
|
2419
|
+
// [0, (none), 1]: the grid for a later row insert counted column 1 twice and
|
|
2420
|
+
// the new row came out one cell short (CodeRabbit, 2026-09-24).
|
|
2110
2421
|
for (const row of table.rows) {
|
|
2422
|
+
for (const cell of row.cells) {
|
|
2423
|
+
if (cell.colAddr !== undefined && cell.colAddr > afterColIndex)
|
|
2424
|
+
cell.colAddr += 1;
|
|
2425
|
+
}
|
|
2426
|
+
const rowAddr = row.cells.find(c => c.rowAddr !== undefined)?.rowAddr;
|
|
2111
2427
|
row.cells.splice(afterColIndex + 1, 0, {
|
|
2428
|
+
colAddr: afterColIndex + 1,
|
|
2429
|
+
...(rowAddr !== undefined ? { rowAddr } : {}),
|
|
2430
|
+
colSpan: 1,
|
|
2431
|
+
rowSpan: 1,
|
|
2112
2432
|
paragraphs: [{
|
|
2113
2433
|
id: Math.random().toString(36).substring(2, 11),
|
|
2114
2434
|
runs: [{ text: '' }],
|
|
2115
2435
|
}],
|
|
2116
2436
|
});
|
|
2117
2437
|
}
|
|
2118
|
-
this._pendingTableColumnInserts
|
|
2438
|
+
this.queueTableOp(this._pendingTableColumnInserts, {
|
|
2119
2439
|
sectionIndex,
|
|
2120
2440
|
tableIndex,
|
|
2121
2441
|
afterColIndex,
|
|
@@ -2128,10 +2448,18 @@ class HwpxDocument {
|
|
|
2128
2448
|
if (!table || (table.rows[0]?.cells.length || 0) <= 1)
|
|
2129
2449
|
return false;
|
|
2130
2450
|
this.saveState();
|
|
2451
|
+
// Mirror applyTableColumnDeletesToXml: cells after the deleted column move one
|
|
2452
|
+
// column left. A write queued after the delete carries the cell's colAddr and
|
|
2453
|
+
// the XML is matched by it, so a stale address sent the text nowhere
|
|
2454
|
+
// (CodeRabbit, 2026-09-24; 0.3.3 dropped these writes too).
|
|
2131
2455
|
for (const row of table.rows) {
|
|
2132
2456
|
row.cells.splice(colIndex, 1);
|
|
2457
|
+
for (const cell of row.cells) {
|
|
2458
|
+
if (cell.colAddr !== undefined && cell.colAddr > colIndex)
|
|
2459
|
+
cell.colAddr -= 1;
|
|
2460
|
+
}
|
|
2133
2461
|
}
|
|
2134
|
-
this._pendingTableColumnDeletes
|
|
2462
|
+
this.queueTableOp(this._pendingTableColumnDeletes, {
|
|
2135
2463
|
sectionIndex,
|
|
2136
2464
|
tableIndex,
|
|
2137
2465
|
colIndex,
|
|
@@ -2317,12 +2645,13 @@ class HwpxDocument {
|
|
|
2317
2645
|
const cellText = cell.paragraphs.map(p => p.runs.map(r => r.text).join('')).join('\n');
|
|
2318
2646
|
// Use existing pending table cell update mechanism
|
|
2319
2647
|
this._pendingTableCellUpdates = this._pendingTableCellUpdates || [];
|
|
2320
|
-
this._pendingTableCellUpdates
|
|
2648
|
+
this.queueTableOp(this._pendingTableCellUpdates, {
|
|
2321
2649
|
sectionIndex,
|
|
2322
2650
|
tableIndex,
|
|
2323
2651
|
tableId,
|
|
2324
2652
|
row,
|
|
2325
2653
|
col,
|
|
2654
|
+
colAddr: cell.colAddr,
|
|
2326
2655
|
text: cellText,
|
|
2327
2656
|
});
|
|
2328
2657
|
this.markModified();
|
|
@@ -2389,15 +2718,24 @@ class HwpxDocument {
|
|
|
2389
2718
|
const srcElement = srcSection.elements[sourceParagraph];
|
|
2390
2719
|
if (!srcElement || srcElement.type !== 'paragraph')
|
|
2391
2720
|
return false;
|
|
2721
|
+
const source = this.resolveElementAnchor(sourceSection, sourceParagraph);
|
|
2722
|
+
if (!source)
|
|
2723
|
+
return false;
|
|
2724
|
+
const anchor = this.resolveElementAnchor(targetSection, targetAfter);
|
|
2392
2725
|
this.saveState();
|
|
2393
2726
|
const copy = JSON.parse(JSON.stringify(srcElement));
|
|
2394
|
-
|
|
2727
|
+
const paragraphId = Math.random().toString(36).substring(2, 11);
|
|
2728
|
+
copy.data.id = paragraphId;
|
|
2729
|
+
delete copy.data._xmlPosition;
|
|
2395
2730
|
tgtSection.elements.splice(targetAfter + 1, 0, copy);
|
|
2731
|
+
this.markStructureChanged();
|
|
2396
2732
|
this._pendingParagraphCopies.push({
|
|
2397
2733
|
sourceSection,
|
|
2398
|
-
sourceParagraph,
|
|
2399
2734
|
targetSection,
|
|
2400
|
-
|
|
2735
|
+
source,
|
|
2736
|
+
anchor,
|
|
2737
|
+
paragraphId,
|
|
2738
|
+
insertOrder: this._tableInsertCounter++,
|
|
2401
2739
|
});
|
|
2402
2740
|
this.markModified();
|
|
2403
2741
|
return true;
|
|
@@ -2410,6 +2748,9 @@ class HwpxDocument {
|
|
|
2410
2748
|
const srcElement = srcSection.elements[sourceParagraph];
|
|
2411
2749
|
if (!srcElement || srcElement.type !== 'paragraph')
|
|
2412
2750
|
return false;
|
|
2751
|
+
const source = this.resolveElementAnchor(sourceSection, sourceParagraph);
|
|
2752
|
+
if (!source)
|
|
2753
|
+
return false;
|
|
2413
2754
|
this.saveState();
|
|
2414
2755
|
srcSection.elements.splice(sourceParagraph, 1);
|
|
2415
2756
|
// Fix same-section index shift: if source was before target, adjust target down
|
|
@@ -2417,12 +2758,17 @@ class HwpxDocument {
|
|
|
2417
2758
|
if (sourceSection === targetSection && sourceParagraph < targetAfter) {
|
|
2418
2759
|
adjustedTargetAfter -= 1;
|
|
2419
2760
|
}
|
|
2761
|
+
// Resolve the destination now that the paragraph has left its old slot,
|
|
2762
|
+
// matching the XML at replay time (source node removed, then re-inserted).
|
|
2763
|
+
const anchor = this.resolveElementAnchor(targetSection, adjustedTargetAfter);
|
|
2420
2764
|
tgtSection.elements.splice(adjustedTargetAfter + 1, 0, srcElement);
|
|
2765
|
+
this.markStructureChanged();
|
|
2421
2766
|
this._pendingParagraphMoves.push({
|
|
2422
2767
|
sourceSection,
|
|
2423
|
-
sourceParagraph,
|
|
2424
2768
|
targetSection,
|
|
2425
|
-
|
|
2769
|
+
source,
|
|
2770
|
+
anchor,
|
|
2771
|
+
insertOrder: this._tableInsertCounter++,
|
|
2426
2772
|
});
|
|
2427
2773
|
this.markModified();
|
|
2428
2774
|
return true;
|
|
@@ -2587,8 +2933,11 @@ class HwpxDocument {
|
|
|
2587
2933
|
rows: tableRows,
|
|
2588
2934
|
width: defaultWidth,
|
|
2589
2935
|
};
|
|
2936
|
+
// Resolve the XML position before the new table joins the element list.
|
|
2937
|
+
const anchor = this.resolveElementAnchor(sectionIndex, afterElementIndex);
|
|
2590
2938
|
const newElement = { type: 'table', data: newTable };
|
|
2591
2939
|
section.elements.splice(afterElementIndex + 1, 0, newElement);
|
|
2940
|
+
this.markStructureChanged();
|
|
2592
2941
|
// Calculate table index
|
|
2593
2942
|
let tableIndex = 0;
|
|
2594
2943
|
for (let i = 0; i <= afterElementIndex + 1; i++) {
|
|
@@ -2599,10 +2948,10 @@ class HwpxDocument {
|
|
|
2599
2948
|
}
|
|
2600
2949
|
}
|
|
2601
2950
|
// Add to pending table inserts for XML generation
|
|
2602
|
-
// Store the original afterElementIndex and insertOrder for proper sequencing
|
|
2603
2951
|
this._pendingTableInserts.push({
|
|
2604
2952
|
sectionIndex,
|
|
2605
2953
|
afterElementIndex,
|
|
2954
|
+
anchor,
|
|
2606
2955
|
rows,
|
|
2607
2956
|
cols,
|
|
2608
2957
|
width: defaultWidth,
|
|
@@ -2661,7 +3010,7 @@ class HwpxDocument {
|
|
|
2661
3010
|
if (!this._pendingNestedTableInserts) {
|
|
2662
3011
|
this._pendingNestedTableInserts = [];
|
|
2663
3012
|
}
|
|
2664
|
-
this._pendingNestedTableInserts
|
|
3013
|
+
this.queueTableOp(this._pendingNestedTableInserts, {
|
|
2665
3014
|
sectionIndex,
|
|
2666
3015
|
parentTableIndex,
|
|
2667
3016
|
row,
|
|
@@ -2728,6 +3077,29 @@ class HwpxDocument {
|
|
|
2728
3077
|
console.warn(`[HwpxDocument] mergeCells: Single cell selected, no merge needed`);
|
|
2729
3078
|
return false;
|
|
2730
3079
|
}
|
|
3080
|
+
// A row whose every own cell falls inside the merge is saved as an <hp:tr>
|
|
3081
|
+
// with no <hp:tc>. 한/글 2024 gave no PDF for such a file (measured: a
|
|
3082
|
+
// full-width two-row merge and a vertical merge in a one-column table; the
|
|
3083
|
+
// same table merged short of full width converted), and a scan of 275 한/글
|
|
3084
|
+
// originals found no row without a cell. 0.3.3 wrote these files too.
|
|
3085
|
+
// Rows built in memory keep covered cells and rows read from a file do not,
|
|
3086
|
+
// so cells are placed by their own address (position only when it has none)
|
|
3087
|
+
// and a cell counts only if no other merged cell covers it.
|
|
3088
|
+
const placedCells = table.rows.flatMap((row, ri) => row.cells.map((cell, ci) => ({ cell, row: cell.rowAddr ?? ri, col: cell.colAddr ?? ci })));
|
|
3089
|
+
const masters = placedCells.filter(p => (p.cell.rowSpan ?? 1) > 1 || (p.cell.colSpan ?? 1) > 1);
|
|
3090
|
+
const coveredByOther = (p) => masters.some(m => m.cell !== p.cell &&
|
|
3091
|
+
p.row >= m.row && p.row < m.row + (m.cell.rowSpan ?? 1) &&
|
|
3092
|
+
p.col >= m.col && p.col < m.col + (m.cell.colSpan ?? 1));
|
|
3093
|
+
for (let r = startRow + 1; r <= endRow; r++) {
|
|
3094
|
+
const keepsCell = placedCells.some(p => p.row === r &&
|
|
3095
|
+
(p.col + (p.cell.colSpan ?? 1) - 1 < startCol || p.col > endCol) &&
|
|
3096
|
+
!coveredByOther(p));
|
|
3097
|
+
if (!keepsCell) {
|
|
3098
|
+
throw new Error(`Cannot merge (${startRow}, ${startCol})-(${endRow}, ${endCol}): row ${r} would have no ` +
|
|
3099
|
+
`cell of its own, and 한/글 does not open a table row without cells. Merge fewer ` +
|
|
3100
|
+
`columns so row ${r} keeps a cell, or delete row ${r} instead.`);
|
|
3101
|
+
}
|
|
3102
|
+
}
|
|
2731
3103
|
this.saveState();
|
|
2732
3104
|
// Calculate span values
|
|
2733
3105
|
const colSpan = endCol - startCol + 1;
|
|
@@ -2739,7 +3111,7 @@ class HwpxDocument {
|
|
|
2739
3111
|
masterCell.rowSpan = rowSpan;
|
|
2740
3112
|
}
|
|
2741
3113
|
// Add to pending merges for XML application during save
|
|
2742
|
-
this._pendingCellMerges
|
|
3114
|
+
this.queueTableOp(this._pendingCellMerges, {
|
|
2743
3115
|
sectionIndex,
|
|
2744
3116
|
tableIndex,
|
|
2745
3117
|
startRow,
|
|
@@ -2806,7 +3178,7 @@ class HwpxDocument {
|
|
|
2806
3178
|
cell.rowSpan = 1;
|
|
2807
3179
|
}
|
|
2808
3180
|
// Add to pending splits for XML application during save
|
|
2809
|
-
this._pendingCellSplits
|
|
3181
|
+
this.queueTableOp(this._pendingCellSplits, {
|
|
2810
3182
|
sectionIndex,
|
|
2811
3183
|
tableIndex,
|
|
2812
3184
|
row,
|
|
@@ -3115,6 +3487,7 @@ class HwpxDocument {
|
|
|
3115
3487
|
// Add image element to section
|
|
3116
3488
|
const newElement = { type: 'image', data: newImage };
|
|
3117
3489
|
section.elements.splice(afterElementIndex + 1, 0, newElement);
|
|
3490
|
+
this.markStructureChanged();
|
|
3118
3491
|
// Add to pending inserts for XML sync
|
|
3119
3492
|
this._pendingImageInserts.push({
|
|
3120
3493
|
sectionIndex,
|
|
@@ -3211,7 +3584,7 @@ class HwpxDocument {
|
|
|
3211
3584
|
// Get original image dimensions from binary data
|
|
3212
3585
|
const orgDimensions = this.getImageDimensions(imageData.data, imageData.mimeType);
|
|
3213
3586
|
// Add to pending cell image inserts
|
|
3214
|
-
this._pendingCellImageInserts
|
|
3587
|
+
this.queueTableOp(this._pendingCellImageInserts, {
|
|
3215
3588
|
sectionIndex,
|
|
3216
3589
|
tableIndex,
|
|
3217
3590
|
row,
|
|
@@ -3295,6 +3668,7 @@ class HwpxDocument {
|
|
|
3295
3668
|
const index = section.elements.findIndex(el => el.type === 'image' && el.data.id === imageId);
|
|
3296
3669
|
if (index !== -1) {
|
|
3297
3670
|
section.elements.splice(index, 1);
|
|
3671
|
+
this.markStructureChanged();
|
|
3298
3672
|
break;
|
|
3299
3673
|
}
|
|
3300
3674
|
}
|
|
@@ -3321,6 +3695,7 @@ class HwpxDocument {
|
|
|
3321
3695
|
};
|
|
3322
3696
|
const newElement = { type: 'line', data: newLine };
|
|
3323
3697
|
section.elements.push(newElement);
|
|
3698
|
+
this.markStructureChanged();
|
|
3324
3699
|
this.markModified();
|
|
3325
3700
|
return { id: lineId };
|
|
3326
3701
|
}
|
|
@@ -3341,6 +3716,7 @@ class HwpxDocument {
|
|
|
3341
3716
|
};
|
|
3342
3717
|
const newElement = { type: 'rect', data: newRect };
|
|
3343
3718
|
section.elements.push(newElement);
|
|
3719
|
+
this.markStructureChanged();
|
|
3344
3720
|
this.markModified();
|
|
3345
3721
|
return { id: rectId };
|
|
3346
3722
|
}
|
|
@@ -3361,6 +3737,7 @@ class HwpxDocument {
|
|
|
3361
3737
|
};
|
|
3362
3738
|
const newElement = { type: 'ellipse', data: newEllipse };
|
|
3363
3739
|
section.elements.push(newElement);
|
|
3740
|
+
this.markStructureChanged();
|
|
3364
3741
|
this.markModified();
|
|
3365
3742
|
return { id: ellipseId };
|
|
3366
3743
|
}
|
|
@@ -3381,6 +3758,7 @@ class HwpxDocument {
|
|
|
3381
3758
|
};
|
|
3382
3759
|
const newElement = { type: 'equation', data: newEquation };
|
|
3383
3760
|
section.elements.splice(afterElementIndex + 1, 0, newElement);
|
|
3761
|
+
this.markStructureChanged();
|
|
3384
3762
|
this.markModified();
|
|
3385
3763
|
return { id: equationId };
|
|
3386
3764
|
}
|
|
@@ -3485,13 +3863,18 @@ class HwpxDocument {
|
|
|
3485
3863
|
}));
|
|
3486
3864
|
}
|
|
3487
3865
|
insertSection(afterSectionIndex) {
|
|
3866
|
+
if (afterSectionIndex < -1 || afterSectionIndex >= this._content.sections.length) {
|
|
3867
|
+
throw new Error(`Cannot insert a section after ${afterSectionIndex}: document has ${this._content.sections.length} section(s).`);
|
|
3868
|
+
}
|
|
3488
3869
|
this.saveState();
|
|
3870
|
+
// The first paragraph of every section carries <hp:secPr>, so it must have
|
|
3871
|
+
// an XML id the anchors can find. '0' matches the section template below.
|
|
3489
3872
|
const newSection = {
|
|
3490
3873
|
id: Math.random().toString(36).substring(2, 11),
|
|
3491
3874
|
elements: [{
|
|
3492
3875
|
type: 'paragraph',
|
|
3493
3876
|
data: {
|
|
3494
|
-
id:
|
|
3877
|
+
id: '0',
|
|
3495
3878
|
runs: [{ text: '' }],
|
|
3496
3879
|
},
|
|
3497
3880
|
}],
|
|
@@ -3506,9 +3889,44 @@ class HwpxDocument {
|
|
|
3506
3889
|
};
|
|
3507
3890
|
const insertIndex = afterSectionIndex + 1;
|
|
3508
3891
|
this._content.sections.splice(insertIndex, 0, newSection);
|
|
3892
|
+
this.markStructureChanged();
|
|
3893
|
+
// insertSection used to change only the memory model: save wrote no
|
|
3894
|
+
// sectionN.xml, so a two-section document silently came back with one
|
|
3895
|
+
// section and everything added to the new section was lost (measured on
|
|
3896
|
+
// 0.3.3 with insert_section + insert_table, 2026-09-24).
|
|
3897
|
+
this._pendingSectionOps.push({ op: 'insert', at: insertIndex, templateFrom: Math.max(0, afterSectionIndex) });
|
|
3898
|
+
// Section files are created at the start of save, before every other
|
|
3899
|
+
// pending edit is replayed. Edits recorded earlier still name sections by
|
|
3900
|
+
// their old number; shift those at or after the insertion point so they
|
|
3901
|
+
// land in the same section after the renumbering (measured: an edit to the
|
|
3902
|
+
// old section 0, then insert_section(-1), wrote into the new section 0).
|
|
3903
|
+
this.shiftPendingSectionIndices(insertIndex, +1);
|
|
3509
3904
|
this.markModified();
|
|
3510
3905
|
return insertIndex;
|
|
3511
3906
|
}
|
|
3907
|
+
/**
|
|
3908
|
+
* Add `delta` to every section number held by a pending edit that is >= from.
|
|
3909
|
+
* Covers all pending arrays generically: any numeric field whose name is
|
|
3910
|
+
* sectionIndex or ends in "Section"/"SectionIndex" (source/target pairs).
|
|
3911
|
+
*/
|
|
3912
|
+
shiftPendingSectionIndices(from, delta) {
|
|
3913
|
+
const isSectionKey = (k) => k === 'sectionIndex' || /Section(Index)?$/.test(k);
|
|
3914
|
+
for (const key of Object.keys(this)) {
|
|
3915
|
+
if (!String(key).startsWith('_pending') || key === '_pendingSectionOps')
|
|
3916
|
+
continue;
|
|
3917
|
+
const list = this[key];
|
|
3918
|
+
if (!Array.isArray(list))
|
|
3919
|
+
continue;
|
|
3920
|
+
for (const item of list) {
|
|
3921
|
+
if (!item || typeof item !== 'object')
|
|
3922
|
+
continue;
|
|
3923
|
+
for (const [k, v] of Object.entries(item)) {
|
|
3924
|
+
if (isSectionKey(k) && typeof v === 'number' && v >= from)
|
|
3925
|
+
item[k] = v + delta;
|
|
3926
|
+
}
|
|
3927
|
+
}
|
|
3928
|
+
}
|
|
3929
|
+
}
|
|
3512
3930
|
deleteSection(sectionIndex) {
|
|
3513
3931
|
if (sectionIndex < 0 || sectionIndex >= this._content.sections.length)
|
|
3514
3932
|
return false;
|
|
@@ -3516,6 +3934,23 @@ class HwpxDocument {
|
|
|
3516
3934
|
return false; // Cannot delete the last section
|
|
3517
3935
|
this.saveState();
|
|
3518
3936
|
this._content.sections.splice(sectionIndex, 1);
|
|
3937
|
+
this.markStructureChanged();
|
|
3938
|
+
// Same persistence gap as insertSection had: the memory model lost the
|
|
3939
|
+
// section but save kept its file, so the deleted section came back on
|
|
3940
|
+
// reopen. Pending edits aimed at the deleted section are dropped; later
|
|
3941
|
+
// sections move down one number.
|
|
3942
|
+
for (const key of Object.keys(this)) {
|
|
3943
|
+
if (!String(key).startsWith('_pending') || key === '_pendingSectionOps')
|
|
3944
|
+
continue;
|
|
3945
|
+
const list = this[key];
|
|
3946
|
+
if (!Array.isArray(list))
|
|
3947
|
+
continue;
|
|
3948
|
+
const kept = list.filter(item => !(item && typeof item === 'object' &&
|
|
3949
|
+
Object.entries(item).some(([k, v]) => (k === 'sectionIndex' || /Section(Index)?$/.test(k)) && v === sectionIndex)));
|
|
3950
|
+
this[key] = kept;
|
|
3951
|
+
}
|
|
3952
|
+
this.shiftPendingSectionIndices(sectionIndex + 1, -1);
|
|
3953
|
+
this._pendingSectionOps.push({ op: 'delete', at: sectionIndex, templateFrom: 0 });
|
|
3519
3954
|
this.markModified();
|
|
3520
3955
|
return true;
|
|
3521
3956
|
}
|
|
@@ -3643,10 +4078,27 @@ class HwpxDocument {
|
|
|
3643
4078
|
async syncContentToZip() {
|
|
3644
4079
|
if (!this._zip)
|
|
3645
4080
|
return;
|
|
3646
|
-
//
|
|
3647
|
-
|
|
3648
|
-
|
|
4081
|
+
// New sections first: every later step addresses Contents/sectionN.xml by
|
|
4082
|
+
// the memory section index, so the files must already exist and be numbered
|
|
4083
|
+
// the same way.
|
|
4084
|
+
if (this._pendingSectionOps.length > 0) {
|
|
4085
|
+
await this.applySectionOpsToZip();
|
|
4086
|
+
this._pendingSectionOps = [];
|
|
4087
|
+
}
|
|
4088
|
+
// Replay paragraph/table inserts and paragraph copies/moves together, in
|
|
4089
|
+
// call order, before any text update. Text updates resolve their target in
|
|
4090
|
+
// the current XML, and other operations locate tables by index, so the
|
|
4091
|
+
// structure must already match the memory model.
|
|
4092
|
+
const hasStructuralEdits = this._pendingTableInserts.length > 0 ||
|
|
4093
|
+
this._pendingParagraphInserts.length > 0 ||
|
|
4094
|
+
this._pendingParagraphCopies.length > 0 ||
|
|
4095
|
+
this._pendingParagraphMoves.length > 0;
|
|
4096
|
+
if (hasStructuralEdits) {
|
|
4097
|
+
await this.applyStructuralInsertsToXml();
|
|
3649
4098
|
this._pendingTableInserts = [];
|
|
4099
|
+
this._pendingParagraphInserts = [];
|
|
4100
|
+
this._pendingParagraphCopies = [];
|
|
4101
|
+
this._pendingParagraphMoves = [];
|
|
3650
4102
|
}
|
|
3651
4103
|
// Apply table deletes
|
|
3652
4104
|
if (this._pendingTableDeletes && this._pendingTableDeletes.length > 0) {
|
|
@@ -3663,36 +4115,13 @@ class HwpxDocument {
|
|
|
3663
4115
|
await this.applyTableMovesToXml();
|
|
3664
4116
|
this._pendingTableMoves = [];
|
|
3665
4117
|
}
|
|
3666
|
-
//
|
|
3667
|
-
|
|
3668
|
-
|
|
3669
|
-
|
|
3670
|
-
|
|
3671
|
-
//
|
|
3672
|
-
|
|
3673
|
-
await this.applyTableCellUpdatesToXml();
|
|
3674
|
-
this._pendingTableCellUpdates = [];
|
|
3675
|
-
}
|
|
3676
|
-
// Apply cell merges
|
|
3677
|
-
if (this._pendingCellMerges && this._pendingCellMerges.length > 0) {
|
|
3678
|
-
await this.applyCellMergesToXml();
|
|
3679
|
-
this._pendingCellMerges = [];
|
|
3680
|
-
}
|
|
3681
|
-
// Apply cell splits
|
|
3682
|
-
if (this._pendingCellSplits && this._pendingCellSplits.length > 0) {
|
|
3683
|
-
await this.applyCellSplitsToXml();
|
|
3684
|
-
this._pendingCellSplits = [];
|
|
3685
|
-
}
|
|
3686
|
-
// Apply nested table inserts
|
|
3687
|
-
if (this._pendingNestedTableInserts && this._pendingNestedTableInserts.length > 0) {
|
|
3688
|
-
await this.applyNestedTableInsertsToXml();
|
|
3689
|
-
this._pendingNestedTableInserts = [];
|
|
3690
|
-
}
|
|
3691
|
-
// Apply cell image inserts
|
|
3692
|
-
if (this._pendingCellImageInserts && this._pendingCellImageInserts.length > 0) {
|
|
3693
|
-
await this.applyCellImageInsertsToXml();
|
|
3694
|
-
this._pendingCellImageInserts = [];
|
|
3695
|
-
}
|
|
4118
|
+
// Table edits that address cells or rows/columns by index, replayed in
|
|
4119
|
+
// CALL order. Each index is relative to the table as it was when that edit
|
|
4120
|
+
// was made; applying them by kind (all cell writes, then all row inserts,
|
|
4121
|
+
// then column inserts ...) wrote cell text into the pre-insert layout and
|
|
4122
|
+
// dropped text written to a new row or column (CodeRabbit, 2026-09-24; the
|
|
4123
|
+
// same 5 scenarios failed on 0.3.3).
|
|
4124
|
+
await this.applyTableOpsInCallOrder();
|
|
3696
4125
|
// Apply direct text updates (from updateParagraphText)
|
|
3697
4126
|
if (this._pendingDirectTextUpdates && this._pendingDirectTextUpdates.length > 0) {
|
|
3698
4127
|
await this.applyDirectTextUpdatesToXml();
|
|
@@ -3718,11 +4147,6 @@ class HwpxDocument {
|
|
|
3718
4147
|
await this.applyHangingIndentsToXml();
|
|
3719
4148
|
this._pendingHangingIndents = [];
|
|
3720
4149
|
}
|
|
3721
|
-
// Apply table cell hanging indent changes
|
|
3722
|
-
if (this._pendingTableCellHangingIndents && this._pendingTableCellHangingIndents.length > 0) {
|
|
3723
|
-
await this.applyTableCellHangingIndentsToXml();
|
|
3724
|
-
this._pendingTableCellHangingIndents = [];
|
|
3725
|
-
}
|
|
3726
4150
|
// Apply paragraph style changes (alignment, etc.)
|
|
3727
4151
|
if (this._pendingParagraphStyles && this._pendingParagraphStyles.length > 0) {
|
|
3728
4152
|
await this.applyParagraphStylesToXml();
|
|
@@ -3733,36 +4157,6 @@ class HwpxDocument {
|
|
|
3733
4157
|
await this.applyCharacterStylesToXml();
|
|
3734
4158
|
this._pendingCharacterStyles = [];
|
|
3735
4159
|
}
|
|
3736
|
-
// Apply table row inserts
|
|
3737
|
-
if (this._pendingTableRowInserts && this._pendingTableRowInserts.length > 0) {
|
|
3738
|
-
await this.applyTableRowInsertsToXml();
|
|
3739
|
-
this._pendingTableRowInserts = [];
|
|
3740
|
-
}
|
|
3741
|
-
// Apply table row deletes
|
|
3742
|
-
if (this._pendingTableRowDeletes && this._pendingTableRowDeletes.length > 0) {
|
|
3743
|
-
await this.applyTableRowDeletesToXml();
|
|
3744
|
-
this._pendingTableRowDeletes = [];
|
|
3745
|
-
}
|
|
3746
|
-
// Apply table column inserts
|
|
3747
|
-
if (this._pendingTableColumnInserts && this._pendingTableColumnInserts.length > 0) {
|
|
3748
|
-
await this.applyTableColumnInsertsToXml();
|
|
3749
|
-
this._pendingTableColumnInserts = [];
|
|
3750
|
-
}
|
|
3751
|
-
// Apply table column deletes
|
|
3752
|
-
if (this._pendingTableColumnDeletes && this._pendingTableColumnDeletes.length > 0) {
|
|
3753
|
-
await this.applyTableColumnDeletesToXml();
|
|
3754
|
-
this._pendingTableColumnDeletes = [];
|
|
3755
|
-
}
|
|
3756
|
-
// Apply paragraph copies
|
|
3757
|
-
if (this._pendingParagraphCopies && this._pendingParagraphCopies.length > 0) {
|
|
3758
|
-
await this.applyParagraphCopiesToXml();
|
|
3759
|
-
this._pendingParagraphCopies = [];
|
|
3760
|
-
}
|
|
3761
|
-
// Apply paragraph moves
|
|
3762
|
-
if (this._pendingParagraphMoves && this._pendingParagraphMoves.length > 0) {
|
|
3763
|
-
await this.applyParagraphMovesToXml();
|
|
3764
|
-
this._pendingParagraphMoves = [];
|
|
3765
|
-
}
|
|
3766
4160
|
// Apply header/footer updates
|
|
3767
4161
|
if (this._pendingHeaderUpdates && this._pendingHeaderUpdates.length > 0 ||
|
|
3768
4162
|
this._pendingFooterUpdates && this._pendingFooterUpdates.length > 0) {
|
|
@@ -3817,6 +4211,13 @@ class HwpxDocument {
|
|
|
3817
4211
|
* The cached positions are populated during parsing in HwpxParser.parseSection().
|
|
3818
4212
|
*/
|
|
3819
4213
|
getCachedXmlPosition(sectionIndex, elementIndex) {
|
|
4214
|
+
// Cached offsets point into the section XML as it was when parsed. Once any
|
|
4215
|
+
// element has been inserted, removed, copied or moved, the element index no
|
|
4216
|
+
// longer names the same XML node and every earlier offset may have shifted.
|
|
4217
|
+
// Using the cache then rewrites the wrong paragraph — measured: after
|
|
4218
|
+
// copyParagraph on a reopened document, the edit landed on the original.
|
|
4219
|
+
if (this._structureChanged)
|
|
4220
|
+
return undefined;
|
|
3820
4221
|
const section = this._content?.sections?.[sectionIndex];
|
|
3821
4222
|
if (!section)
|
|
3822
4223
|
return undefined;
|
|
@@ -3934,10 +4335,10 @@ class HwpxDocument {
|
|
|
3934
4335
|
}
|
|
3935
4336
|
// Clean up empty runs that may be left behind
|
|
3936
4337
|
// <hp:run charPrIDRef="0"><hp:t/></hp:run> or <hp:run charPrIDRef="0"></hp:run>
|
|
3937
|
-
xml = xml.replace(/<hp:run[^>]
|
|
4338
|
+
xml = xml.replace(/<hp:run(?:\s[^>]*)?>(\s*<hp:t\s*\/>)?\s*<\/hp:run>/g, '');
|
|
3938
4339
|
// Clean up empty paragraphs that only contained the image
|
|
3939
4340
|
// <hp:p ...><hp:linesegarray>...</hp:linesegarray></hp:p>
|
|
3940
|
-
xml = xml.replace(/<hp:p[^>]
|
|
4341
|
+
xml = xml.replace(/<hp:p(?:\s[^>]*)?>\s*(<hp:linesegarray[^>]*>[\s\S]*?<\/hp:linesegarray>)?\s*<\/hp:p>/g, '');
|
|
3941
4342
|
if (modified) {
|
|
3942
4343
|
this._zip.file(sectionPath, xml);
|
|
3943
4344
|
}
|
|
@@ -4054,143 +4455,345 @@ class HwpxDocument {
|
|
|
4054
4455
|
}
|
|
4055
4456
|
}
|
|
4056
4457
|
/**
|
|
4057
|
-
*
|
|
4058
|
-
*
|
|
4458
|
+
* Find the end offset of the section-level element an insert anchors to.
|
|
4459
|
+
*
|
|
4460
|
+
* Paragraphs are matched by their own <hp:p id>. Tables are matched by
|
|
4461
|
+
* <hp:tbl id>, either wrapped in a paragraph (the insert goes after that
|
|
4462
|
+
* paragraph) or placed directly in the section (move_table writes them that
|
|
4463
|
+
* way).
|
|
4464
|
+
*
|
|
4465
|
+
* The returned offset is always the end of a TOP-LEVEL element, because that
|
|
4466
|
+
* is the only place a new section-level element may go. But paragraph
|
|
4467
|
+
* occurrences are counted over exactly the paragraphs the parser puts in the
|
|
4468
|
+
* memory model (see parsedParagraphStarts), so they agree with the occurrence
|
|
4469
|
+
* resolveElementAnchor recorded. The parser also lifts paragraphs out of
|
|
4470
|
+
* headers, text boxes and shapes; Hancom reuses id="0" / id="2147483648"
|
|
4471
|
+
* there too. Counting only top-level paragraphs put 47 of 131 sampled Hancom
|
|
4472
|
+
* files' copies and moves on the wrong paragraph.
|
|
4473
|
+
*
|
|
4474
|
+
* Returns -1 if the anchor is not present in the current XML.
|
|
4475
|
+
*/
|
|
4476
|
+
findAnchorEnd(xml, anchor) {
|
|
4477
|
+
const topLevel = this.findTopLevelFullElements(xml);
|
|
4478
|
+
const idOf = (fragment) => fragment.slice(0, fragment.indexOf('>') + 1).match(/\bid="([^"]*)"/)?.[1];
|
|
4479
|
+
if (anchor.kind === 'paragraph') {
|
|
4480
|
+
const hit = this.findParsedParagraph(xml, anchor);
|
|
4481
|
+
if (!hit)
|
|
4482
|
+
return -1;
|
|
4483
|
+
// The anchor may sit inside a header/shape; new content goes after the
|
|
4484
|
+
// top-level element that contains it.
|
|
4485
|
+
const owner = topLevel.find(el => el.startIndex <= hit.start && hit.start < el.endIndex);
|
|
4486
|
+
return owner ? owner.endIndex : -1;
|
|
4487
|
+
}
|
|
4488
|
+
let seen = 0;
|
|
4489
|
+
for (const el of topLevel) {
|
|
4490
|
+
if (el.type === 'tbl') {
|
|
4491
|
+
if (idOf(el.xml) !== anchor.id)
|
|
4492
|
+
continue;
|
|
4493
|
+
}
|
|
4494
|
+
else if (!this.wrapsTopLevelTable(el.xml, anchor.id)) {
|
|
4495
|
+
continue;
|
|
4496
|
+
}
|
|
4497
|
+
if (seen === anchor.occurrence)
|
|
4498
|
+
return el.endIndex;
|
|
4499
|
+
seen++;
|
|
4500
|
+
}
|
|
4501
|
+
return -1;
|
|
4502
|
+
}
|
|
4503
|
+
/**
|
|
4504
|
+
* The exact XML range of the memory paragraph a paragraph anchor names,
|
|
4505
|
+
* found by id + occurrence among parsedParagraphStarts. For a paragraph in a
|
|
4506
|
+
* header or text box this is that paragraph alone, not its container.
|
|
4059
4507
|
*/
|
|
4060
|
-
|
|
4061
|
-
|
|
4062
|
-
|
|
4063
|
-
|
|
4064
|
-
|
|
4065
|
-
|
|
4066
|
-
|
|
4067
|
-
|
|
4068
|
-
|
|
4069
|
-
|
|
4070
|
-
|
|
4071
|
-
|
|
4072
|
-
cellWidth: insert.cellWidth,
|
|
4073
|
-
insertOrder: insert.insertOrder,
|
|
4074
|
-
tableId: insert.tableId,
|
|
4075
|
-
});
|
|
4076
|
-
insertsBySection.set(insert.sectionIndex, sectionInserts);
|
|
4508
|
+
findParsedParagraph(xml, anchor) {
|
|
4509
|
+
let seen = 0;
|
|
4510
|
+
for (const start of this.parsedParagraphStarts(xml)) {
|
|
4511
|
+
const openEnd = xml.indexOf('>', start) + 1;
|
|
4512
|
+
const id = xml.slice(start, openEnd).match(/\bid="([^"]*)"/)?.[1];
|
|
4513
|
+
if (id !== anchor.id)
|
|
4514
|
+
continue;
|
|
4515
|
+
if (seen === anchor.occurrence) {
|
|
4516
|
+
const end = this.findBalancedParagraphEnd(xml, start);
|
|
4517
|
+
return end === -1 ? null : { start, end };
|
|
4518
|
+
}
|
|
4519
|
+
seen++;
|
|
4077
4520
|
}
|
|
4078
|
-
|
|
4079
|
-
|
|
4080
|
-
|
|
4081
|
-
|
|
4082
|
-
|
|
4521
|
+
return null;
|
|
4522
|
+
}
|
|
4523
|
+
/**
|
|
4524
|
+
* Start offsets (in `xml`) of the paragraphs HwpxParser turns into memory
|
|
4525
|
+
* paragraphs, in document order. Mirrors HwpxParser.parseSection:
|
|
4526
|
+
*
|
|
4527
|
+
* - MEMO fields, footnotes and endnotes are ignored;
|
|
4528
|
+
* - paragraphs inside any table are skipped;
|
|
4529
|
+
* - a paragraph that holds a table is kept only if it still has <hp:t>
|
|
4530
|
+
* once its tables are removed.
|
|
4531
|
+
*
|
|
4532
|
+
* Offsets are mapped back to the original XML, so callers can slice it.
|
|
4533
|
+
*/
|
|
4534
|
+
parsedParagraphStarts(xml) {
|
|
4535
|
+
// Ranges the parser strips before it looks for paragraphs.
|
|
4536
|
+
const hidden = [];
|
|
4537
|
+
const hide = (re) => {
|
|
4538
|
+
for (const m of xml.matchAll(re))
|
|
4539
|
+
hidden.push([m.index, m.index + m[0].length]);
|
|
4540
|
+
};
|
|
4541
|
+
hide(/<hp:fieldBegin[^>]*type="MEMO"[^>]*>[\s\S]*?<\/hp:fieldBegin>/gi);
|
|
4542
|
+
hide(/<hp:footNote\b[^>]*>[\s\S]*?<\/hp:footNote>/gi);
|
|
4543
|
+
hide(/<hp:endNote\b[^>]*>[\s\S]*?<\/hp:endNote>/gi);
|
|
4544
|
+
const isHidden = (pos) => hidden.some(([a, b]) => pos >= a && pos < b);
|
|
4545
|
+
const tables = this.findAllTablesDeep(xml).filter(t => !isHidden(t.startIndex));
|
|
4546
|
+
const inTable = (pos) => tables.some(t => pos > t.startIndex && pos < t.endIndex);
|
|
4547
|
+
const starts = [];
|
|
4548
|
+
for (const m of xml.matchAll(/<hp:p\b(?=[\s>])[^>]*>/g)) {
|
|
4549
|
+
const start = m.index;
|
|
4550
|
+
if (isHidden(start) || inTable(start))
|
|
4083
4551
|
continue;
|
|
4084
|
-
|
|
4085
|
-
|
|
4086
|
-
|
|
4087
|
-
|
|
4088
|
-
|
|
4089
|
-
|
|
4090
|
-
|
|
4091
|
-
|
|
4092
|
-
|
|
4093
|
-
|
|
4094
|
-
|
|
4095
|
-
|
|
4096
|
-
|
|
4097
|
-
|
|
4098
|
-
|
|
4099
|
-
|
|
4100
|
-
|
|
4101
|
-
|
|
4102
|
-
|
|
4103
|
-
|
|
4104
|
-
|
|
4105
|
-
|
|
4106
|
-
|
|
4107
|
-
|
|
4108
|
-
|
|
4109
|
-
|
|
4110
|
-
|
|
4111
|
-
|
|
4112
|
-
|
|
4113
|
-
|
|
4114
|
-
|
|
4115
|
-
|
|
4116
|
-
|
|
4117
|
-
|
|
4118
|
-
tableXml += `<hp:cellAddr colAddr="${c}" rowAddr="${r}"/>`;
|
|
4119
|
-
tableXml += `<hp:cellSpan colSpan="1" rowSpan="1"/>`;
|
|
4120
|
-
tableXml += `<hp:cellSz width="${insert.cellWidth}" height="${rowHeight}"/>`;
|
|
4121
|
-
tableXml += `<hp:cellMargin left="141" right="141" top="141" bottom="141"/>`;
|
|
4122
|
-
tableXml += `</hp:tc>`;
|
|
4123
|
-
}
|
|
4124
|
-
tableXml += `</hp:tr>`;
|
|
4125
|
-
}
|
|
4126
|
-
tableXml += `</hp:tbl>`;
|
|
4127
|
-
// Find the position to insert the table
|
|
4128
|
-
// We need to insert after a paragraph element
|
|
4129
|
-
// Find all <hp:p> elements at the root level (not inside tables)
|
|
4130
|
-
const paragraphMatches = [...xml.matchAll(/<hp:p\s[^>]*>.*?<\/hp:p>/gs)];
|
|
4131
|
-
// Filter to find only top-level paragraphs (not inside <hp:tbl> or <hp:subList>)
|
|
4132
|
-
// For simplicity, insert after the first paragraph if afterElementIndex is 0
|
|
4133
|
-
// or find the appropriate position
|
|
4134
|
-
let insertPosition = -1;
|
|
4135
|
-
let elementCount = -1;
|
|
4136
|
-
let searchPos = 0;
|
|
4137
|
-
// Find paragraphs and tables at root level using balanced bracket matching
|
|
4138
|
-
while (searchPos < xml.length) {
|
|
4139
|
-
// Look for next <hp:p or <hp:tbl
|
|
4140
|
-
const nextP = xml.indexOf('<hp:p ', searchPos);
|
|
4141
|
-
const nextTbl = xml.indexOf('<hp:tbl ', searchPos);
|
|
4142
|
-
let nextPos = -1;
|
|
4143
|
-
let isTable = false;
|
|
4144
|
-
if (nextP !== -1 && (nextTbl === -1 || nextP < nextTbl)) {
|
|
4145
|
-
nextPos = nextP;
|
|
4146
|
-
isTable = false;
|
|
4147
|
-
}
|
|
4148
|
-
else if (nextTbl !== -1) {
|
|
4149
|
-
nextPos = nextTbl;
|
|
4150
|
-
isTable = true;
|
|
4151
|
-
}
|
|
4152
|
-
if (nextPos === -1)
|
|
4153
|
-
break;
|
|
4154
|
-
// Check if this is inside a subList (nested)
|
|
4155
|
-
const beforeText = xml.substring(Math.max(0, nextPos - HwpxDocument.NESTED_CHECK_LOOKBACK), nextPos);
|
|
4156
|
-
const subListOpen = beforeText.lastIndexOf('<hp:subList');
|
|
4157
|
-
const subListClose = beforeText.lastIndexOf('</hp:subList>');
|
|
4158
|
-
const isNested = subListOpen > subListClose;
|
|
4159
|
-
if (!isNested) {
|
|
4160
|
-
elementCount++;
|
|
4161
|
-
// Find the end of this element using balanced bracket matching
|
|
4162
|
-
const endPos = isTable
|
|
4163
|
-
? HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:tbl', '</hp:tbl>')
|
|
4164
|
-
: HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:p ', '</hp:p>');
|
|
4165
|
-
if (endPos === -1) {
|
|
4166
|
-
searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
|
|
4167
|
-
continue;
|
|
4168
|
-
}
|
|
4169
|
-
if (elementCount === insert.afterElementIndex) {
|
|
4170
|
-
insertPosition = endPos;
|
|
4171
|
-
break;
|
|
4172
|
-
}
|
|
4173
|
-
searchPos = endPos;
|
|
4174
|
-
}
|
|
4175
|
-
else {
|
|
4176
|
-
searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
|
|
4177
|
-
}
|
|
4552
|
+
const end = this.findBalancedParagraphEnd(xml, start);
|
|
4553
|
+
if (end === -1)
|
|
4554
|
+
continue;
|
|
4555
|
+
// Remove only the outermost tables in this paragraph. A nested table
|
|
4556
|
+
// is already inside one of them; cutting it again with its original
|
|
4557
|
+
// offsets would slice the wrong text out of the shortened string.
|
|
4558
|
+
const own = tables.filter(t => t.startIndex >= start && t.endIndex <= end &&
|
|
4559
|
+
!tables.some(o => o !== t && o.startIndex >= start && o.startIndex < t.startIndex && o.endIndex > t.endIndex));
|
|
4560
|
+
if (own.length > 0) {
|
|
4561
|
+
let rest = xml.slice(start, end);
|
|
4562
|
+
for (const t of [...own].sort((a, b) => b.startIndex - a.startIndex)) {
|
|
4563
|
+
rest = rest.slice(0, t.startIndex - start) + rest.slice(t.endIndex - start);
|
|
4564
|
+
}
|
|
4565
|
+
if (!/<hp:t\b[^>]*>/.test(rest))
|
|
4566
|
+
continue;
|
|
4567
|
+
}
|
|
4568
|
+
starts.push(start);
|
|
4569
|
+
}
|
|
4570
|
+
return starts;
|
|
4571
|
+
}
|
|
4572
|
+
/** Every <hp:tbl> range at any depth (outer tables before their nested ones). */
|
|
4573
|
+
findAllTablesDeep(xml) {
|
|
4574
|
+
const out = [];
|
|
4575
|
+
for (const m of xml.matchAll(/<hp:tbl\b/g)) {
|
|
4576
|
+
let depth = 1;
|
|
4577
|
+
let pos = m.index + 7;
|
|
4578
|
+
while (depth > 0 && pos < xml.length) {
|
|
4579
|
+
const nextOpen = xml.indexOf('<hp:tbl', pos);
|
|
4580
|
+
const nextClose = xml.indexOf('</hp:tbl>', pos);
|
|
4581
|
+
if (nextClose === -1)
|
|
4582
|
+
break;
|
|
4583
|
+
if (nextOpen !== -1 && nextOpen < nextClose) {
|
|
4584
|
+
depth++;
|
|
4585
|
+
pos = nextOpen + 7;
|
|
4178
4586
|
}
|
|
4179
|
-
|
|
4180
|
-
|
|
4181
|
-
|
|
4182
|
-
if (secEnd !== -1) {
|
|
4183
|
-
insertPosition = secEnd;
|
|
4184
|
-
}
|
|
4587
|
+
else {
|
|
4588
|
+
depth--;
|
|
4589
|
+
pos = nextClose + 9;
|
|
4185
4590
|
}
|
|
4186
|
-
|
|
4187
|
-
|
|
4188
|
-
|
|
4189
|
-
|
|
4190
|
-
|
|
4591
|
+
}
|
|
4592
|
+
if (depth === 0)
|
|
4593
|
+
out.push({ startIndex: m.index, endIndex: pos });
|
|
4594
|
+
}
|
|
4595
|
+
return out;
|
|
4596
|
+
}
|
|
4597
|
+
/** End offset of the paragraph opening at `start`, counting nested <hp:p>. */
|
|
4598
|
+
findBalancedParagraphEnd(xml, start) {
|
|
4599
|
+
const openRe = /<hp:p\b(?=[\s>/])[^>]*>/g;
|
|
4600
|
+
let depth = 0;
|
|
4601
|
+
let pos = start;
|
|
4602
|
+
while (pos < xml.length) {
|
|
4603
|
+
openRe.lastIndex = pos;
|
|
4604
|
+
const open = openRe.exec(xml);
|
|
4605
|
+
const close = xml.indexOf('</hp:p>', pos);
|
|
4606
|
+
if (close === -1)
|
|
4607
|
+
return -1;
|
|
4608
|
+
if (open && open.index < close) {
|
|
4609
|
+
if (!open[0].endsWith('/>'))
|
|
4610
|
+
depth++;
|
|
4611
|
+
pos = open.index + open[0].length;
|
|
4612
|
+
}
|
|
4613
|
+
else {
|
|
4614
|
+
depth--;
|
|
4615
|
+
pos = close + 7;
|
|
4616
|
+
if (depth === 0)
|
|
4617
|
+
return pos;
|
|
4618
|
+
}
|
|
4619
|
+
}
|
|
4620
|
+
return -1;
|
|
4621
|
+
}
|
|
4622
|
+
/** True if this paragraph directly (not via a nested table) holds <hp:tbl id>. */
|
|
4623
|
+
wrapsTopLevelTable(paragraphXml, tableId) {
|
|
4624
|
+
const escaped = tableId.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
4625
|
+
const match = paragraphXml.match(new RegExp(`<(?:hp|hs|hc):tbl\\b[^>]*\\bid="${escaped}"`));
|
|
4626
|
+
if (!match || match.index === undefined)
|
|
4627
|
+
return false;
|
|
4628
|
+
const before = paragraphXml.slice(0, match.index);
|
|
4629
|
+
const opens = (before.match(/<(?:hp|hs|hc):tbl\b/g) || []).length;
|
|
4630
|
+
const closes = (before.match(/<\/(?:hp|hs|hc):tbl>/g) || []).length;
|
|
4631
|
+
return opens === closes;
|
|
4632
|
+
}
|
|
4633
|
+
/**
|
|
4634
|
+
* Offset for an insert with no anchor ("before everything"). The first
|
|
4635
|
+
* paragraph carries <hp:secPr> (page and section settings) and must stay
|
|
4636
|
+
* first, so new content goes right after it.
|
|
4637
|
+
*/
|
|
4638
|
+
findSectionHeadEnd(xml) {
|
|
4639
|
+
const topLevel = this.findTopLevelFullElements(xml);
|
|
4640
|
+
const first = topLevel.find(el => el.type === 'p');
|
|
4641
|
+
if (first)
|
|
4642
|
+
return first.endIndex;
|
|
4643
|
+
const secOpen = xml.match(/<(?:hs|hp):sec[^>]*>/);
|
|
4644
|
+
return secOpen && secOpen.index !== undefined ? secOpen.index + secOpen[0].length : -1;
|
|
4645
|
+
}
|
|
4646
|
+
/** Build the XML for a table inserted by insertTable, wrapped in its own paragraph. */
|
|
4647
|
+
buildInsertedTableXml(insert, nextId) {
|
|
4648
|
+
const rowHeight = 1000; // hwpunit
|
|
4649
|
+
const tableHeight = rowHeight * insert.rows;
|
|
4650
|
+
let tableXml = `<hp:tbl id="${insert.tableId}" zOrder="0" numberingType="TABLE" textWrap="TOP_AND_BOTTOM" textFlow="BOTH_SIDES" lock="0" dropcapstyle="None" pageBreak="CELL" repeatHeader="0" rowCnt="${insert.rows}" colCnt="${insert.cols}" cellSpacing="0" borderFillIDRef="2" noAdjust="0">`;
|
|
4651
|
+
tableXml += `<hp:sz width="${insert.width}" widthRelTo="ABSOLUTE" height="${tableHeight}" heightRelTo="ABSOLUTE" protect="0"/>`;
|
|
4652
|
+
tableXml += `<hp:pos treatAsChar="1" affectLSpacing="0" flowWithText="1" allowOverlap="0" holdAnchorAndSO="0" vertRelTo="PARA" horzRelTo="PARA" vertAlign="TOP" horzAlign="LEFT" vertOffset="0" horzOffset="0"/>`;
|
|
4653
|
+
tableXml += `<hp:outMargin left="141" right="141" top="141" bottom="141"/>`;
|
|
4654
|
+
// Cells below use hasMargin="0", which tells Hancom to pad them with this
|
|
4655
|
+
// table-level inMargin and ignore their own cellMargin. A zero inMargin put
|
|
4656
|
+
// text flush against the cell border (measured 0pt). 510/510/141/141 is the
|
|
4657
|
+
// most common value in Hancom-saved tables (1,868 surveyed).
|
|
4658
|
+
tableXml += `<hp:inMargin left="510" right="510" top="141" bottom="141"/>`;
|
|
4659
|
+
for (let r = 0; r < insert.rows; r++) {
|
|
4660
|
+
tableXml += `<hp:tr>`;
|
|
4661
|
+
for (let c = 0; c < insert.cols; c++) {
|
|
4662
|
+
tableXml += `<hp:tc name="" header="0" hasMargin="0" protect="0" editable="0" dirty="0" borderFillIDRef="2">`;
|
|
4663
|
+
tableXml += `<hp:subList id="" textDirection="HORIZONTAL" lineWrap="BREAK" vertAlign="CENTER" linkListIDRef="0" linkListNextIDRef="0" textWidth="0" textHeight="0" hasTextRef="0" hasNumRef="0">`;
|
|
4664
|
+
tableXml += `<hp:p id="${nextId()}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0">`;
|
|
4665
|
+
tableXml += `<hp:run charPrIDRef="0"><hp:t></hp:t></hp:run>`;
|
|
4666
|
+
tableXml += `</hp:p>`;
|
|
4667
|
+
tableXml += `</hp:subList>`;
|
|
4668
|
+
tableXml += `<hp:cellAddr colAddr="${c}" rowAddr="${r}"/>`;
|
|
4669
|
+
tableXml += `<hp:cellSpan colSpan="1" rowSpan="1"/>`;
|
|
4670
|
+
tableXml += `<hp:cellSz width="${insert.cellWidth}" height="${rowHeight}"/>`;
|
|
4671
|
+
tableXml += `<hp:cellMargin left="510" right="510" top="141" bottom="141"/>`;
|
|
4672
|
+
tableXml += `</hp:tc>`;
|
|
4673
|
+
}
|
|
4674
|
+
tableXml += `</hp:tr>`;
|
|
4675
|
+
}
|
|
4676
|
+
tableXml += `</hp:tbl>`;
|
|
4677
|
+
return `<hp:p id="${nextId()}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0">${tableXml}<hp:t></hp:t></hp:run></hp:p>`;
|
|
4678
|
+
}
|
|
4679
|
+
/**
|
|
4680
|
+
* Replay structural edits — paragraph/table inserts and paragraph
|
|
4681
|
+
* copies/moves — into the section XML in the order the calls were made.
|
|
4682
|
+
*
|
|
4683
|
+
* Every edit carries id-based anchors resolved at call time, so it lands
|
|
4684
|
+
* after the same element in the XML that it followed in memory. Replaying in
|
|
4685
|
+
* call order means each anchor already exists (or has already moved) by the
|
|
4686
|
+
* time a later edit needs it. Copies/moves may cross sections, so all
|
|
4687
|
+
* touched sections are held in memory and written once at the end.
|
|
4688
|
+
*/
|
|
4689
|
+
async applyStructuralInsertsToXml() {
|
|
4690
|
+
if (!this._zip)
|
|
4691
|
+
return;
|
|
4692
|
+
const edits = [
|
|
4693
|
+
...this._pendingParagraphInserts.map(i => ({
|
|
4694
|
+
kind: 'paragraph', order: i.insertOrder, sectionIndex: i.sectionIndex,
|
|
4695
|
+
anchor: i.anchor, paragraphId: i.paragraphId, text: i.text,
|
|
4696
|
+
})),
|
|
4697
|
+
...this._pendingTableInserts.map(i => ({
|
|
4698
|
+
kind: 'table', order: i.insertOrder, sectionIndex: i.sectionIndex,
|
|
4699
|
+
anchor: i.anchor, rows: i.rows, cols: i.cols, width: i.width,
|
|
4700
|
+
cellWidth: i.cellWidth, tableId: i.tableId,
|
|
4701
|
+
})),
|
|
4702
|
+
...this._pendingParagraphCopies.map(c => ({
|
|
4703
|
+
kind: 'copy', order: c.insertOrder, sectionIndex: c.targetSection,
|
|
4704
|
+
sourceSection: c.sourceSection, source: c.source, anchor: c.anchor, paragraphId: c.paragraphId,
|
|
4705
|
+
})),
|
|
4706
|
+
...this._pendingParagraphMoves.map(m => ({
|
|
4707
|
+
kind: 'move', order: m.insertOrder, sectionIndex: m.targetSection,
|
|
4708
|
+
sourceSection: m.sourceSection, source: m.source, anchor: m.anchor,
|
|
4709
|
+
})),
|
|
4710
|
+
].sort((a, b) => a.order - b.order);
|
|
4711
|
+
if (edits.length === 0)
|
|
4712
|
+
return;
|
|
4713
|
+
const sections = new Map();
|
|
4714
|
+
const load = async (index) => {
|
|
4715
|
+
if (sections.has(index))
|
|
4716
|
+
return sections.get(index);
|
|
4717
|
+
const file = this._zip.file(`Contents/section${index}.xml`);
|
|
4718
|
+
if (!file)
|
|
4719
|
+
return undefined;
|
|
4720
|
+
const xml = await file.async('string');
|
|
4721
|
+
sections.set(index, xml);
|
|
4722
|
+
return xml;
|
|
4723
|
+
};
|
|
4724
|
+
// Numeric ids for generated cell/wrapper paragraphs must not collide.
|
|
4725
|
+
const maxIdBySection = new Map();
|
|
4726
|
+
const nextIdFor = (index, xml) => {
|
|
4727
|
+
if (!maxIdBySection.has(index)) {
|
|
4728
|
+
let maxId = 0;
|
|
4729
|
+
for (const m of xml.matchAll(/\bid="(\d+)"/g)) {
|
|
4730
|
+
const n = parseInt(m[1], 10);
|
|
4731
|
+
if (n < 2147483648 && n > maxId)
|
|
4732
|
+
maxId = n;
|
|
4733
|
+
}
|
|
4734
|
+
maxIdBySection.set(index, maxId);
|
|
4735
|
+
}
|
|
4736
|
+
return () => {
|
|
4737
|
+
const next = maxIdBySection.get(index) + 1;
|
|
4738
|
+
maxIdBySection.set(index, next);
|
|
4739
|
+
return next;
|
|
4740
|
+
};
|
|
4741
|
+
};
|
|
4742
|
+
const placeAfter = (xml, anchor) => {
|
|
4743
|
+
const position = anchor ? this.findAnchorEnd(xml, anchor) : this.findSectionHeadEnd(xml);
|
|
4744
|
+
if (position !== -1)
|
|
4745
|
+
return position;
|
|
4746
|
+
// The anchor vanished (e.g. deleted later in the same session) —
|
|
4747
|
+
// append rather than drop the user's content.
|
|
4748
|
+
return Math.max(xml.lastIndexOf('</hs:sec>'), xml.lastIndexOf('</hp:sec>'));
|
|
4749
|
+
};
|
|
4750
|
+
for (const edit of edits) {
|
|
4751
|
+
if (edit.kind === 'copy' || edit.kind === 'move') {
|
|
4752
|
+
const srcXml = await load(edit.sourceSection);
|
|
4753
|
+
if (srcXml === undefined)
|
|
4754
|
+
continue;
|
|
4755
|
+
// Only a section-level paragraph can be copied or moved as a unit; a
|
|
4756
|
+
// paragraph inside a header or text box would drag its container along.
|
|
4757
|
+
const found = this.findParsedParagraph(srcXml, edit.source);
|
|
4758
|
+
if (!found)
|
|
4759
|
+
continue;
|
|
4760
|
+
const srcEl = this.findTopLevelFullElements(srcXml)
|
|
4761
|
+
.find(el => el.type === 'p' && el.startIndex === found.start && el.endIndex === found.end);
|
|
4762
|
+
if (!srcEl)
|
|
4763
|
+
continue;
|
|
4764
|
+
let fragment = srcEl.xml;
|
|
4765
|
+
if (edit.kind === 'copy') {
|
|
4766
|
+
// Same id as the memory copy, so later edits anchored on it find it.
|
|
4767
|
+
fragment = fragment.replace(/^<(hp|hs):p\b([^>]*?)\bid="[^"]*"/, `<$1:p$2id="${edit.paragraphId}"`);
|
|
4768
|
+
// The clone inherits the source's fixed <hp:lineseg> geometry; reset
|
|
4769
|
+
// it so replacement text of a different length does not overlap.
|
|
4770
|
+
fragment = this.resetLinesegInXml(fragment);
|
|
4771
|
+
}
|
|
4772
|
+
else {
|
|
4773
|
+
sections.set(edit.sourceSection, srcXml.slice(0, srcEl.startIndex) + srcXml.slice(srcEl.endIndex));
|
|
4191
4774
|
}
|
|
4775
|
+
const tgtXml = await load(edit.sectionIndex);
|
|
4776
|
+
if (tgtXml === undefined)
|
|
4777
|
+
continue;
|
|
4778
|
+
const position = placeAfter(tgtXml, edit.anchor);
|
|
4779
|
+
if (position === -1)
|
|
4780
|
+
continue;
|
|
4781
|
+
sections.set(edit.sectionIndex, tgtXml.slice(0, position) + fragment + tgtXml.slice(position));
|
|
4782
|
+
continue;
|
|
4192
4783
|
}
|
|
4193
|
-
|
|
4784
|
+
const xml = await load(edit.sectionIndex);
|
|
4785
|
+
if (xml === undefined)
|
|
4786
|
+
continue;
|
|
4787
|
+
const position = placeAfter(xml, edit.anchor);
|
|
4788
|
+
if (position === -1)
|
|
4789
|
+
continue;
|
|
4790
|
+
const newXml = edit.kind === 'paragraph'
|
|
4791
|
+
? `<hp:p id="${edit.paragraphId}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0"><hp:t>${this.escapeXml(edit.text)}</hp:t></hp:run></hp:p>`
|
|
4792
|
+
: this.buildInsertedTableXml(edit, nextIdFor(edit.sectionIndex, xml));
|
|
4793
|
+
sections.set(edit.sectionIndex, xml.slice(0, position) + newXml + xml.slice(position));
|
|
4794
|
+
}
|
|
4795
|
+
for (const [index, xml] of sections) {
|
|
4796
|
+
this._zip.file(`Contents/section${index}.xml`, xml);
|
|
4194
4797
|
}
|
|
4195
4798
|
}
|
|
4196
4799
|
/**
|
|
@@ -4277,12 +4880,13 @@ class HwpxDocument {
|
|
|
4277
4880
|
idMap.set(oldId, newId);
|
|
4278
4881
|
}
|
|
4279
4882
|
}
|
|
4280
|
-
// Second pass: replace
|
|
4281
|
-
|
|
4282
|
-
|
|
4283
|
-
|
|
4284
|
-
|
|
4285
|
-
|
|
4883
|
+
// Second pass: replace every id in one scan. Building a RegExp per old id
|
|
4884
|
+
// broke on ids with regex metacharacters, and replacing ids one at a time
|
|
4885
|
+
// could rewrite an id that an earlier replacement had just produced.
|
|
4886
|
+
return xml.replace(/id="([^"]+)"/g, (whole, oldId) => {
|
|
4887
|
+
const newId = idMap.get(oldId);
|
|
4888
|
+
return newId === undefined ? whole : `id="${newId}"`;
|
|
4889
|
+
});
|
|
4286
4890
|
}
|
|
4287
4891
|
/**
|
|
4288
4892
|
* Find the position to insert an element after a given element index.
|
|
@@ -4300,7 +4904,7 @@ class HwpxDocument {
|
|
|
4300
4904
|
// Find all root-level elements (paragraphs, tables)
|
|
4301
4905
|
const elements = [];
|
|
4302
4906
|
// Find paragraphs (not inside subList)
|
|
4303
|
-
const pRegex = /<hp:p[^>]
|
|
4907
|
+
const pRegex = /<hp:p(?:\s[^>]*)?>[\s\S]*?<\/hp:p>/g;
|
|
4304
4908
|
let match;
|
|
4305
4909
|
// Find tables
|
|
4306
4910
|
const tables = this.findAllTables(xml);
|
|
@@ -4314,7 +4918,7 @@ class HwpxDocument {
|
|
|
4314
4918
|
const end = start + match[0].length;
|
|
4315
4919
|
// Check if this paragraph is inside a table (inside subList)
|
|
4316
4920
|
const beforeMatch = xml.substring(0, start);
|
|
4317
|
-
const subListOpen = (beforeMatch.match(/<hp:subList[^>]
|
|
4921
|
+
const subListOpen = (beforeMatch.match(/<hp:subList(?:\s[^>]*)?>/g) || []).length;
|
|
4318
4922
|
const subListClose = (beforeMatch.match(/<\/hp:subList>/g) || []).length;
|
|
4319
4923
|
if (subListOpen === subListClose) {
|
|
4320
4924
|
// This is a root-level paragraph
|
|
@@ -4330,111 +4934,6 @@ class HwpxDocument {
|
|
|
4330
4934
|
// Return position after the element at afterIndex
|
|
4331
4935
|
return elements[afterIndex].end;
|
|
4332
4936
|
}
|
|
4333
|
-
/**
|
|
4334
|
-
* Apply paragraph inserts to XML.
|
|
4335
|
-
* Inserts new paragraphs at the specified positions.
|
|
4336
|
-
*/
|
|
4337
|
-
async applyParagraphInsertsToXml() {
|
|
4338
|
-
if (!this._zip)
|
|
4339
|
-
return;
|
|
4340
|
-
// Group inserts by section
|
|
4341
|
-
const insertsBySection = new Map();
|
|
4342
|
-
for (const insert of this._pendingParagraphInserts) {
|
|
4343
|
-
const sectionInserts = insertsBySection.get(insert.sectionIndex) || [];
|
|
4344
|
-
sectionInserts.push({
|
|
4345
|
-
afterElementIndex: insert.afterElementIndex,
|
|
4346
|
-
paragraphId: insert.paragraphId,
|
|
4347
|
-
text: insert.text,
|
|
4348
|
-
});
|
|
4349
|
-
insertsBySection.set(insert.sectionIndex, sectionInserts);
|
|
4350
|
-
}
|
|
4351
|
-
// Process each section
|
|
4352
|
-
for (const [sectionIndex, inserts] of insertsBySection) {
|
|
4353
|
-
const sectionPath = `Contents/section${sectionIndex}.xml`;
|
|
4354
|
-
const file = this._zip.file(sectionPath);
|
|
4355
|
-
if (!file)
|
|
4356
|
-
continue;
|
|
4357
|
-
let xml = await file.async('string');
|
|
4358
|
-
// Sort inserts by afterElementIndex in ascending order
|
|
4359
|
-
// This ensures each insert happens at the correct position as XML grows
|
|
4360
|
-
const sortedInserts = [...inserts].sort((a, b) => a.afterElementIndex - b.afterElementIndex);
|
|
4361
|
-
for (const insert of sortedInserts) {
|
|
4362
|
-
// Escape text for XML
|
|
4363
|
-
const escapedText = this.escapeXml(insert.text);
|
|
4364
|
-
// Build paragraph XML
|
|
4365
|
-
const paragraphXml = `<hp:p id="${insert.paragraphId}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0"><hp:t>${escapedText}</hp:t></hp:run></hp:p>`;
|
|
4366
|
-
// Find the position to insert
|
|
4367
|
-
let insertPosition = -1;
|
|
4368
|
-
let elementCount = -1;
|
|
4369
|
-
let searchPos = 0;
|
|
4370
|
-
// Find paragraphs and tables at root level using balanced bracket matching
|
|
4371
|
-
while (searchPos < xml.length) {
|
|
4372
|
-
// Look for next <hp:p or <hp:tbl
|
|
4373
|
-
const nextP = xml.indexOf('<hp:p ', searchPos);
|
|
4374
|
-
const nextTbl = xml.indexOf('<hp:tbl ', searchPos);
|
|
4375
|
-
let nextPos = -1;
|
|
4376
|
-
let isTable = false;
|
|
4377
|
-
if (nextP !== -1 && (nextTbl === -1 || nextP < nextTbl)) {
|
|
4378
|
-
nextPos = nextP;
|
|
4379
|
-
isTable = false;
|
|
4380
|
-
}
|
|
4381
|
-
else if (nextTbl !== -1) {
|
|
4382
|
-
nextPos = nextTbl;
|
|
4383
|
-
isTable = true;
|
|
4384
|
-
}
|
|
4385
|
-
if (nextPos === -1)
|
|
4386
|
-
break;
|
|
4387
|
-
// Check if this is inside a subList (nested)
|
|
4388
|
-
const beforeText = xml.substring(Math.max(0, nextPos - HwpxDocument.NESTED_CHECK_LOOKBACK), nextPos);
|
|
4389
|
-
const subListOpen = beforeText.lastIndexOf('<hp:subList');
|
|
4390
|
-
const subListClose = beforeText.lastIndexOf('</hp:subList>');
|
|
4391
|
-
const isNested = subListOpen > subListClose;
|
|
4392
|
-
if (!isNested) {
|
|
4393
|
-
elementCount++;
|
|
4394
|
-
// Find the end of this element using balanced bracket matching
|
|
4395
|
-
const endPos = isTable
|
|
4396
|
-
? HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:tbl', '</hp:tbl>')
|
|
4397
|
-
: HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:p ', '</hp:p>');
|
|
4398
|
-
if (endPos === -1) {
|
|
4399
|
-
searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
|
|
4400
|
-
continue;
|
|
4401
|
-
}
|
|
4402
|
-
if (elementCount === insert.afterElementIndex) {
|
|
4403
|
-
insertPosition = endPos;
|
|
4404
|
-
break;
|
|
4405
|
-
}
|
|
4406
|
-
searchPos = endPos;
|
|
4407
|
-
}
|
|
4408
|
-
else {
|
|
4409
|
-
searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
|
|
4410
|
-
}
|
|
4411
|
-
}
|
|
4412
|
-
// If afterElementIndex is -1, insert after the first paragraph (which contains secPr)
|
|
4413
|
-
// IMPORTANT: <hp:secPr> must remain in the first paragraph for the document to be valid
|
|
4414
|
-
if (insert.afterElementIndex === -1) {
|
|
4415
|
-
// Find the end of the first <hp:p> element (which contains <hp:secPr>)
|
|
4416
|
-
const firstPStart = xml.indexOf('<hp:p');
|
|
4417
|
-
if (firstPStart !== -1) {
|
|
4418
|
-
const firstPEnd = xml.indexOf('</hp:p>', firstPStart);
|
|
4419
|
-
if (firstPEnd !== -1) {
|
|
4420
|
-
insertPosition = firstPEnd + '</hp:p>'.length;
|
|
4421
|
-
}
|
|
4422
|
-
}
|
|
4423
|
-
}
|
|
4424
|
-
// If position not found, insert at end of section (before </hs:sec>)
|
|
4425
|
-
if (insertPosition === -1) {
|
|
4426
|
-
const secEnd = xml.lastIndexOf('</hs:sec>');
|
|
4427
|
-
if (secEnd !== -1) {
|
|
4428
|
-
insertPosition = secEnd;
|
|
4429
|
-
}
|
|
4430
|
-
}
|
|
4431
|
-
if (insertPosition !== -1) {
|
|
4432
|
-
xml = xml.substring(0, insertPosition) + paragraphXml + xml.substring(insertPosition);
|
|
4433
|
-
}
|
|
4434
|
-
}
|
|
4435
|
-
this._zip.file(sectionPath, xml);
|
|
4436
|
-
}
|
|
4437
|
-
}
|
|
4438
4937
|
/**
|
|
4439
4938
|
* Apply nested table inserts to XML.
|
|
4440
4939
|
* Inserts a new table inside a cell of an existing table.
|
|
@@ -4506,8 +5005,11 @@ class HwpxDocument {
|
|
|
4506
5005
|
if (insert.col >= cells.length)
|
|
4507
5006
|
continue;
|
|
4508
5007
|
const cellXml = cells[insert.col].xml;
|
|
4509
|
-
//
|
|
4510
|
-
|
|
5008
|
+
// Size the nested table to the parent cell. A fixed per-cell width
|
|
5009
|
+
// ignored the parent and pushed columns past its border (measured:
|
|
5010
|
+
// 3 × 8000 = 24000 inside a 21260-wide cell).
|
|
5011
|
+
const innerWidth = this.getCellInnerWidth(cellXml, tableXml);
|
|
5012
|
+
const nestedTableXml = this.generateNestedTableXml(insert.nestedRows, insert.nestedCols, insert.data, innerWidth);
|
|
4511
5013
|
// Insert nested table into cell
|
|
4512
5014
|
const updatedCellXml = this.insertNestedTableIntoCell(cellXml, nestedTableXml);
|
|
4513
5015
|
// Update the row with the new cell
|
|
@@ -4538,14 +5040,54 @@ class HwpxDocument {
|
|
|
4538
5040
|
/**
|
|
4539
5041
|
* Generate XML for a nested table.
|
|
4540
5042
|
*/
|
|
4541
|
-
|
|
5043
|
+
/**
|
|
5044
|
+
* Usable width inside a table cell, in hwpunit.
|
|
5045
|
+
*
|
|
5046
|
+
* A cell with hasMargin="0" takes its padding from the table's inMargin, so
|
|
5047
|
+
* the cell's own cellMargin is only authoritative when hasMargin="1".
|
|
5048
|
+
*
|
|
5049
|
+
* Every lookup is scoped to the cell's (or table's) own markup. A nested
|
|
5050
|
+
* table inside the cell carries its own cellSz/cellMargin/inMargin, and a
|
|
5051
|
+
* first-match regex over the whole cell would read those instead — which
|
|
5052
|
+
* sized a second nested table to the first one's column (7086 vs 21260).
|
|
5053
|
+
*/
|
|
5054
|
+
getCellInnerWidth(cellXml, tableXml) {
|
|
5055
|
+
// hp:tc children are subList → cellAddr → cellSpan → cellSz → cellMargin,
|
|
5056
|
+
// so the cell's own properties are everything after its last </hp:subList>.
|
|
5057
|
+
const subListEnd = cellXml.lastIndexOf('</hp:subList>');
|
|
5058
|
+
const cellProps = subListEnd === -1 ? cellXml : cellXml.slice(subListEnd);
|
|
5059
|
+
const size = cellProps.match(/<hp:cellSz width="(\d+)"/);
|
|
5060
|
+
if (!size)
|
|
5061
|
+
return null;
|
|
5062
|
+
const width = parseInt(size[1], 10);
|
|
5063
|
+
const openTag = cellXml.slice(0, cellXml.indexOf('>') + 1);
|
|
5064
|
+
const usesOwnMargin = /\bhasMargin="1"/.test(openTag);
|
|
5065
|
+
// Table-level inMargin precedes the first row.
|
|
5066
|
+
const firstRow = tableXml.indexOf('<hp:tr');
|
|
5067
|
+
const tableHead = firstRow === -1 ? tableXml : tableXml.slice(0, firstRow);
|
|
5068
|
+
const margin = usesOwnMargin
|
|
5069
|
+
? cellProps.match(/<hp:cellMargin left="(\d+)" right="(\d+)"/)
|
|
5070
|
+
: tableHead.match(/<hp:inMargin left="(\d+)" right="(\d+)"/);
|
|
5071
|
+
const padding = margin ? parseInt(margin[1], 10) + parseInt(margin[2], 10) : 0;
|
|
5072
|
+
return Math.max(width - padding, 0);
|
|
5073
|
+
}
|
|
5074
|
+
/**
|
|
5075
|
+
* Generate XML for a nested table.
|
|
5076
|
+
*
|
|
5077
|
+
* @param innerWidth Usable width of the parent cell in hwpunit. The nested
|
|
5078
|
+
* table is sized to fit it exactly; columns share the width evenly.
|
|
5079
|
+
*/
|
|
5080
|
+
generateNestedTableXml(rows, cols, data, innerWidth = null) {
|
|
4542
5081
|
// Generate unique ID
|
|
4543
5082
|
const id = Math.floor(Math.random() * 2000000000) + 100000000;
|
|
4544
5083
|
const zOrder = Math.floor(Math.random() * 100);
|
|
4545
|
-
// Calculate sizes (
|
|
4546
|
-
|
|
5084
|
+
// Calculate sizes (hwpunit, 100 = 1pt). Without a parent width fall back to
|
|
5085
|
+
// the previous fixed column width so standalone callers keep working.
|
|
5086
|
+
const tableWidth = innerWidth !== null && innerWidth > 0 ? innerWidth : 8000 * cols;
|
|
5087
|
+
const baseCellWidth = Math.floor(tableWidth / cols);
|
|
5088
|
+
// Give the rounding remainder to the last column so the columns sum to tableWidth.
|
|
5089
|
+
const cellWidthAt = (c) => (c === cols - 1 ? tableWidth - baseCellWidth * (cols - 1) : baseCellWidth);
|
|
4547
5090
|
const cellHeight = 1400; // ~14mm per cell
|
|
4548
|
-
const tableWidth = cellWidth * cols;
|
|
4549
5091
|
const tableHeight = cellHeight * rows;
|
|
4550
5092
|
let xml = `<hp:tbl id="${id}" zOrder="${zOrder}" numberingType="TABLE" textWrap="TOP_AND_BOTTOM" textFlow="BOTH_SIDES" lock="0" dropcapstyle="None" pageBreak="NONE" repeatHeader="0" rowCnt="${rows}" colCnt="${cols}" cellSpacing="0" borderFillIDRef="2" noAdjust="0">`;
|
|
4551
5093
|
// Size element
|
|
@@ -4578,7 +5120,7 @@ class HwpxDocument {
|
|
|
4578
5120
|
xml += `</hp:subList>`;
|
|
4579
5121
|
xml += `<hp:cellAddr colAddr="${c}" rowAddr="${r}"/>`;
|
|
4580
5122
|
xml += `<hp:cellSpan colSpan="1" rowSpan="1"/>`;
|
|
4581
|
-
xml += `<hp:cellSz width="${
|
|
5123
|
+
xml += `<hp:cellSz width="${cellWidthAt(c)}" height="${cellHeight}"/>`;
|
|
4582
5124
|
xml += `<hp:cellMargin left="141" right="141" top="141" bottom="141"/>`;
|
|
4583
5125
|
xml += `</hp:tc>`;
|
|
4584
5126
|
}
|
|
@@ -4593,13 +5135,13 @@ class HwpxDocument {
|
|
|
4593
5135
|
*/
|
|
4594
5136
|
insertNestedTableIntoCell(cellXml, nestedTableXml) {
|
|
4595
5137
|
// Find the subList in the cell
|
|
4596
|
-
const subListMatch = cellXml.match(/<hp:subList[^>]
|
|
5138
|
+
const subListMatch = cellXml.match(/<hp:subList(?:\s[^>]*)?>/);
|
|
4597
5139
|
if (!subListMatch) {
|
|
4598
5140
|
// No subList, try to add to paragraph directly
|
|
4599
|
-
const pMatch = cellXml.match(/<hp:p[^>]
|
|
5141
|
+
const pMatch = cellXml.match(/<hp:p(?:\s[^>]*)?>/);
|
|
4600
5142
|
if (pMatch) {
|
|
4601
5143
|
const insertPos = cellXml.indexOf(pMatch[0]) + pMatch[0].length;
|
|
4602
|
-
const runXml = `<hp:run charPrIDRef="0"
|
|
5144
|
+
const runXml = `<hp:run charPrIDRef="0">${nestedTableXml}<hp:t/></hp:run>`;
|
|
4603
5145
|
return cellXml.substring(0, insertPos) + runXml + cellXml.substring(insertPos);
|
|
4604
5146
|
}
|
|
4605
5147
|
return cellXml;
|
|
@@ -4618,8 +5160,10 @@ class HwpxDocument {
|
|
|
4618
5160
|
return cellXml;
|
|
4619
5161
|
// Find the end of the opening <hp:p ...> tag
|
|
4620
5162
|
const pTagEnd = cellXml.indexOf('>', pStart) + 1;
|
|
4621
|
-
// Create new run with nested table
|
|
4622
|
-
|
|
5163
|
+
// Create new run with nested table. No leading text: a space before an
|
|
5164
|
+
// inline (treatAsChar) table that fills the cell width forces the table onto
|
|
5165
|
+
// a second line and leaves an empty first line above it (measured in Hancom).
|
|
5166
|
+
const runXml = `<hp:run charPrIDRef="0">${nestedTableXml}<hp:t/></hp:run>`;
|
|
4623
5167
|
// Insert after the opening <hp:p> tag
|
|
4624
5168
|
return cellXml.substring(0, pTagEnd) + runXml + cellXml.substring(pTagEnd);
|
|
4625
5169
|
}
|
|
@@ -4904,7 +5448,7 @@ class HwpxDocument {
|
|
|
4904
5448
|
}
|
|
4905
5449
|
// If no cell before masterCol, insert at the beginning of row content
|
|
4906
5450
|
if (insertPoint === -1) {
|
|
4907
|
-
const trMatch = updatedRowXml.match(/<(hp|hs):tr[^>]
|
|
5451
|
+
const trMatch = updatedRowXml.match(/<(hp|hs):tr(?:\s[^>]*)?>/);
|
|
4908
5452
|
if (trMatch) {
|
|
4909
5453
|
insertPoint = trMatch[0].length;
|
|
4910
5454
|
}
|
|
@@ -4941,6 +5485,54 @@ class HwpxDocument {
|
|
|
4941
5485
|
</hp:subList>
|
|
4942
5486
|
</hp:tc>`;
|
|
4943
5487
|
}
|
|
5488
|
+
/**
|
|
5489
|
+
* Replay every pending table edit (cell text, merge/split, nested table,
|
|
5490
|
+
* cell image, cell hanging indent, row/column insert/delete) in call order.
|
|
5491
|
+
*
|
|
5492
|
+
* Each index an edit carries is relative to the table as it was when the
|
|
5493
|
+
* edit was made. Applying by kind (all cell writes, then all row inserts,
|
|
5494
|
+
* then all column inserts ...) wrote text into the pre-insert layout; and
|
|
5495
|
+
* the row appliers sort their own queue by index, which reorders two
|
|
5496
|
+
* inserts or two deletes on the same table. So edits that change a table's
|
|
5497
|
+
* row/column layout run one at a time. Runs of layout-preserving edits
|
|
5498
|
+
* (cell text, indents, images, nested tables) go to their applier together.
|
|
5499
|
+
*/
|
|
5500
|
+
async applyTableOpsInCallOrder() {
|
|
5501
|
+
const kinds = [
|
|
5502
|
+
{ layout: false, take: () => this._pendingTableCellUpdates, put: o => { this._pendingTableCellUpdates = o; }, apply: () => this.applyTableCellUpdatesToXml() },
|
|
5503
|
+
{ layout: true, take: () => this._pendingCellMerges, put: o => { this._pendingCellMerges = o; }, apply: () => this.applyCellMergesToXml() },
|
|
5504
|
+
{ layout: true, take: () => this._pendingCellSplits, put: o => { this._pendingCellSplits = o; }, apply: () => this.applyCellSplitsToXml() },
|
|
5505
|
+
{ layout: false, take: () => this._pendingNestedTableInserts, put: o => { this._pendingNestedTableInserts = o; }, apply: () => this.applyNestedTableInsertsToXml() },
|
|
5506
|
+
{ layout: false, take: () => this._pendingCellImageInserts, put: o => { this._pendingCellImageInserts = o; }, apply: () => this.applyCellImageInsertsToXml() },
|
|
5507
|
+
{ layout: false, take: () => this._pendingTableCellHangingIndents, put: o => { this._pendingTableCellHangingIndents = o; }, apply: () => this.applyTableCellHangingIndentsToXml() },
|
|
5508
|
+
{ layout: true, take: () => this._pendingTableRowInserts, put: o => { this._pendingTableRowInserts = o; }, apply: () => this.applyTableRowInsertsToXml() },
|
|
5509
|
+
{ layout: true, take: () => this._pendingTableRowDeletes, put: o => { this._pendingTableRowDeletes = o; }, apply: () => this.applyTableRowDeletesToXml() },
|
|
5510
|
+
{ layout: true, take: () => this._pendingTableColumnInserts, put: o => { this._pendingTableColumnInserts = o; }, apply: () => this.applyTableColumnInsertsToXml() },
|
|
5511
|
+
{ layout: true, take: () => this._pendingTableColumnDeletes, put: o => { this._pendingTableColumnDeletes = o; }, apply: () => this.applyTableColumnDeletesToXml() },
|
|
5512
|
+
];
|
|
5513
|
+
// Every push goes through queueTableOp, so every op has a sequence number;
|
|
5514
|
+
// a missing one would sort last and keep its queue position.
|
|
5515
|
+
const all = [];
|
|
5516
|
+
for (const kind of kinds) {
|
|
5517
|
+
kind.take().forEach((op, pos) => all.push({ kind, op, seq: this._tableOpSeq.get(op) ?? Number.MAX_SAFE_INTEGER, pos }));
|
|
5518
|
+
kind.put([]);
|
|
5519
|
+
}
|
|
5520
|
+
all.sort((a, b) => a.seq - b.seq || a.pos - b.pos);
|
|
5521
|
+
for (let i = 0; i < all.length;) {
|
|
5522
|
+
const kind = all[i].kind;
|
|
5523
|
+
const batch = [all[i++].op];
|
|
5524
|
+
if (!kind.layout)
|
|
5525
|
+
while (i < all.length && all[i].kind === kind)
|
|
5526
|
+
batch.push(all[i++].op);
|
|
5527
|
+
kind.put(batch);
|
|
5528
|
+
try {
|
|
5529
|
+
await kind.apply();
|
|
5530
|
+
}
|
|
5531
|
+
finally {
|
|
5532
|
+
kind.put([]);
|
|
5533
|
+
}
|
|
5534
|
+
}
|
|
5535
|
+
}
|
|
4944
5536
|
/**
|
|
4945
5537
|
* Apply table cell updates to XML while preserving original structure.
|
|
4946
5538
|
* This function modifies only the text content of specific cells,
|
|
@@ -4958,7 +5550,7 @@ class HwpxDocument {
|
|
|
4958
5550
|
const updatesBySection = new Map();
|
|
4959
5551
|
for (const update of this._pendingTableCellUpdates) {
|
|
4960
5552
|
const sectionUpdates = updatesBySection.get(update.sectionIndex) || [];
|
|
4961
|
-
sectionUpdates.push({ tableId: update.tableId, row: update.row, col: update.col, text: update.text, charShapeId: update.charShapeId });
|
|
5553
|
+
sectionUpdates.push({ tableId: update.tableId, row: update.row, col: update.col, colAddr: update.colAddr, text: update.text, charShapeId: update.charShapeId });
|
|
4962
5554
|
updatesBySection.set(update.sectionIndex, sectionUpdates);
|
|
4963
5555
|
}
|
|
4964
5556
|
// Process each section that has updates
|
|
@@ -4974,7 +5566,7 @@ class HwpxDocument {
|
|
|
4974
5566
|
const updatesByTableId = new Map();
|
|
4975
5567
|
for (const update of updates) {
|
|
4976
5568
|
const tableUpdates = updatesByTableId.get(update.tableId) || [];
|
|
4977
|
-
tableUpdates.push({ row: update.row, col: update.col, text: update.text, charShapeId: update.charShapeId });
|
|
5569
|
+
tableUpdates.push({ row: update.row, col: update.col, colAddr: update.colAddr, text: update.text, charShapeId: update.charShapeId });
|
|
4978
5570
|
updatesByTableId.set(update.tableId, tableUpdates);
|
|
4979
5571
|
}
|
|
4980
5572
|
// Process each table that has updates (by ID)
|
|
@@ -5225,12 +5817,14 @@ class HwpxDocument {
|
|
|
5225
5817
|
* Find a table by its ID in XML.
|
|
5226
5818
|
*/
|
|
5227
5819
|
findTableById(xml, tableId) {
|
|
5228
|
-
// Match table with specific ID
|
|
5229
|
-
|
|
5820
|
+
// Match table with specific ID. The id comes from document XML, so it is
|
|
5821
|
+
// escaped: an id holding '.', '(' or '+' matched another table or threw.
|
|
5822
|
+
const id = this.escapeRegex(tableId);
|
|
5823
|
+
const tableStartRegex = new RegExp(`<(?:hp|hs|hc):tbl\\s[^>]*\\bid="${id}"[^>]*>`, 'g');
|
|
5230
5824
|
const match = tableStartRegex.exec(xml);
|
|
5231
5825
|
if (!match) {
|
|
5232
5826
|
// Try alternate ID format (id='...' instead of id="...")
|
|
5233
|
-
const altRegex = new RegExp(`<(?:hp|hs|hc):tbl[^>]*\\bid='${
|
|
5827
|
+
const altRegex = new RegExp(`<(?:hp|hs|hc):tbl\\s[^>]*\\bid='${id}'[^>]*>`, 'g');
|
|
5234
5828
|
const altMatch = altRegex.exec(xml);
|
|
5235
5829
|
if (!altMatch)
|
|
5236
5830
|
return null;
|
|
@@ -5369,7 +5963,7 @@ class HwpxDocument {
|
|
|
5369
5963
|
findAllTables(xml) {
|
|
5370
5964
|
const tables = [];
|
|
5371
5965
|
// Match both hp:tbl and hs:tbl (different namespace prefixes)
|
|
5372
|
-
const tableStartRegex = /<(?:hp|hs|hc):tbl[^>]
|
|
5966
|
+
const tableStartRegex = /<(?:hp|hs|hc):tbl(?:\s[^>]*)?>/g;
|
|
5373
5967
|
let match;
|
|
5374
5968
|
while ((match = tableStartRegex.exec(xml)) !== null) {
|
|
5375
5969
|
const startIndex = match.index;
|
|
@@ -5474,7 +6068,7 @@ class HwpxDocument {
|
|
|
5474
6068
|
if (!updatesByRow.has(update.row)) {
|
|
5475
6069
|
updatesByRow.set(update.row, []);
|
|
5476
6070
|
}
|
|
5477
|
-
updatesByRow.get(update.row).push({ col: update.col, text: update.text, charShapeId: update.charShapeId });
|
|
6071
|
+
updatesByRow.get(update.row).push({ col: update.col, colAddr: update.colAddr, text: update.text, charShapeId: update.charShapeId });
|
|
5478
6072
|
}
|
|
5479
6073
|
// Sort row indices descending to process from end to start (avoid index shifting)
|
|
5480
6074
|
const sortedRowIndices = Array.from(updatesByRow.keys()).sort((a, b) => b - a);
|
|
@@ -5520,17 +6114,28 @@ class HwpxDocument {
|
|
|
5520
6114
|
let result = rowXml;
|
|
5521
6115
|
// Find all cells in this row using depth tracking to handle nested tables correctly
|
|
5522
6116
|
const cells = this.findAllElementsWithDepth(rowXml, 'tc');
|
|
6117
|
+
// Resolve each update to its <hp:tc> index. The cell's grid column
|
|
6118
|
+
// (colAddr) is authoritative: after a merge the XML row no longer has the
|
|
6119
|
+
// covered cells that memory still lists, so the memory position `col`
|
|
6120
|
+
// points one cell too far. `col` is used only when the write has no
|
|
6121
|
+
// colAddr or the row carries no addresses.
|
|
6122
|
+
const cellCols = cells.map(c => this.cellOwnAttr(c.xml, 'colAddr')?.value);
|
|
6123
|
+
const indexOf = (u) => {
|
|
6124
|
+
if (u.colAddr !== undefined && cellCols.some(a => a !== undefined))
|
|
6125
|
+
return cellCols.indexOf(u.colAddr);
|
|
6126
|
+
return u.col < cells.length ? u.col : -1;
|
|
6127
|
+
};
|
|
5523
6128
|
// Deduplicate updates for the same cell (keep last value)
|
|
5524
6129
|
// This prevents stale index issues when the same cell is updated multiple times
|
|
5525
6130
|
const uniqueUpdates = new Map();
|
|
5526
6131
|
for (const update of updates) {
|
|
5527
|
-
|
|
6132
|
+
const at = indexOf(update);
|
|
6133
|
+
if (at >= 0)
|
|
6134
|
+
uniqueUpdates.set(at, { ...update, col: at });
|
|
5528
6135
|
}
|
|
5529
6136
|
// Sort updates by col descending to process from right to left (avoid index shifting)
|
|
5530
6137
|
const sortedUpdates = Array.from(uniqueUpdates.values()).sort((a, b) => b.col - a.col);
|
|
5531
6138
|
for (const update of sortedUpdates) {
|
|
5532
|
-
if (update.col >= cells.length)
|
|
5533
|
-
continue;
|
|
5534
6139
|
const cellData = cells[update.col];
|
|
5535
6140
|
// Validate cell before update - capture nested table structure
|
|
5536
6141
|
const cellTblOpen = (cellData.xml.match(/<(?:hp|hs|hc):tbl[\s>]/g) || []).length;
|
|
@@ -5606,7 +6211,7 @@ class HwpxDocument {
|
|
|
5606
6211
|
xml = xml.replace(/(<(?:hp|hs|hc):run\s+)charPrIDRef="[^"]*"/, `$1charPrIDRef="${charShapeId}"`);
|
|
5607
6212
|
}
|
|
5608
6213
|
// Pattern 1: Cell has existing <hp:t> or <hs:t> or <hc:t> tags with content
|
|
5609
|
-
const tTagPattern = /(<(?:hp|hs|hc):t[^>]
|
|
6214
|
+
const tTagPattern = /(<(?:hp|hs|hc):t(?:\s[^>]*)?>)([^<]*)(<\/(?:hp|hs|hc):t>)/g;
|
|
5610
6215
|
let foundText = false;
|
|
5611
6216
|
let result = xml.replace(tTagPattern, (match, openTag, _oldText, closeTag, offset) => {
|
|
5612
6217
|
// Only replace the first text occurrence
|
|
@@ -5619,14 +6224,14 @@ class HwpxDocument {
|
|
|
5619
6224
|
if (foundText)
|
|
5620
6225
|
return this.resetLinesegInXml(result);
|
|
5621
6226
|
// Pattern 2: Cell has empty <hp:t/> or <hp:t></hp:t> tags
|
|
5622
|
-
const emptyTTagPattern = /<((?:hp|hs|hc):t)([^>]
|
|
6227
|
+
const emptyTTagPattern = /<((?:hp|hs|hc):t)((?:\s[^>]*?)?)\s*\/>/;
|
|
5623
6228
|
const emptyTMatch = xml.match(emptyTTagPattern);
|
|
5624
6229
|
if (emptyTMatch) {
|
|
5625
6230
|
const updated = xml.replace(emptyTTagPattern, `<${emptyTMatch[1]}${emptyTMatch[2]}>${escapedText}</${emptyTMatch[1]}>`);
|
|
5626
6231
|
return this.resetLinesegInXml(updated);
|
|
5627
6232
|
}
|
|
5628
6233
|
// Pattern 3a: Self-closing <hp:run .../> - expand to full run with text
|
|
5629
|
-
const selfClosingRunPattern = /<((?:hp|hs|hc):run)([^>]
|
|
6234
|
+
const selfClosingRunPattern = /<((?:hp|hs|hc):run)((?:\s[^>]*?)?)\s*\/>/;
|
|
5630
6235
|
const selfClosingRunMatch = xml.match(selfClosingRunPattern);
|
|
5631
6236
|
if (selfClosingRunMatch) {
|
|
5632
6237
|
const tagName = selfClosingRunMatch[1]; // e.g., "hp:run"
|
|
@@ -5645,7 +6250,7 @@ class HwpxDocument {
|
|
|
5645
6250
|
return this.resetLinesegInXml(updated);
|
|
5646
6251
|
}
|
|
5647
6252
|
// Pattern 3b: Cell has <hp:run> but no <hp:t> - add text inside run
|
|
5648
|
-
const runPattern = /(<(?:hp|hs|hc):run[^>]
|
|
6253
|
+
const runPattern = /(<(?:hp|hs|hc):run(?:\s[^>]*)?>)([\s\S]*?)(<\/(?:hp|hs|hc):run>)/;
|
|
5649
6254
|
const runMatch = xml.match(runPattern);
|
|
5650
6255
|
if (runMatch) {
|
|
5651
6256
|
const prefix = runMatch[1].match(/<(hp|hs|hc):run/)?.[1] || 'hp';
|
|
@@ -5654,7 +6259,7 @@ class HwpxDocument {
|
|
|
5654
6259
|
return this.resetLinesegInXml(updated);
|
|
5655
6260
|
}
|
|
5656
6261
|
// Pattern 4: Cell has <hp:subList><hp:p> structure - find the paragraph and add text
|
|
5657
|
-
const subListPattern = /(<(?:hp|hs|hc):subList[^>]
|
|
6262
|
+
const subListPattern = /(<(?:hp|hs|hc):subList(?:\s[^>]*)?>[\s\S]*?<(?:hp|hs|hc):p(?:\s[^>]*)?>)([\s\S]*?)(<\/(?:hp|hs|hc):p>)/;
|
|
5658
6263
|
const subListMatch = xml.match(subListPattern);
|
|
5659
6264
|
if (subListMatch) {
|
|
5660
6265
|
const prefix = subListMatch[1].match(/<(hp|hs|hc):subList/)?.[1] || 'hp';
|
|
@@ -5667,7 +6272,7 @@ class HwpxDocument {
|
|
|
5667
6272
|
}
|
|
5668
6273
|
}
|
|
5669
6274
|
// Pattern 5: Cell has only <hp:p> without subList
|
|
5670
|
-
const pPattern = /(<(?:hp|hs|hc):p[^>]
|
|
6275
|
+
const pPattern = /(<(?:hp|hs|hc):p(?:\s[^>]*)?>)([\s\S]*?)(<\/(?:hp|hs|hc):p>)/;
|
|
5671
6276
|
const pMatch = xml.match(pPattern);
|
|
5672
6277
|
if (pMatch) {
|
|
5673
6278
|
const prefix = pMatch[1].match(/<(hp|hs|hc):p/)?.[1] || 'hp';
|
|
@@ -5689,7 +6294,7 @@ class HwpxDocument {
|
|
|
5689
6294
|
const charAttr = charShapeId !== undefined ? ` charPrIDRef="${charShapeId}"` : ' charPrIDRef="0"';
|
|
5690
6295
|
let xml = cellXml;
|
|
5691
6296
|
// Find the subList element to replace paragraph content
|
|
5692
|
-
const subListStartMatch = xml.match(/<(hp|hs|hc):subList[^>]
|
|
6297
|
+
const subListStartMatch = xml.match(/<(hp|hs|hc):subList(?:\s[^>]*)?>/);
|
|
5693
6298
|
if (subListStartMatch) {
|
|
5694
6299
|
const prefix = subListStartMatch[1];
|
|
5695
6300
|
const startTag = subListStartMatch[0];
|
|
@@ -5723,7 +6328,7 @@ class HwpxDocument {
|
|
|
5723
6328
|
// Preserve nested tables
|
|
5724
6329
|
const nestedTables = this.extractNestedTables(subListContent, prefix);
|
|
5725
6330
|
// Extract paraPrIDRef and styleIDRef from existing paragraph
|
|
5726
|
-
const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
|
|
6331
|
+
const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p\s[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
|
|
5727
6332
|
const paraPrIDRef = existingPMatch?.[1] || '0';
|
|
5728
6333
|
const styleIDRef = existingPMatch?.[2] || '0';
|
|
5729
6334
|
const paraId = Math.floor(Math.random() * 2147483647);
|
|
@@ -5735,7 +6340,7 @@ class HwpxDocument {
|
|
|
5735
6340
|
}
|
|
5736
6341
|
}
|
|
5737
6342
|
// Fallback: try to find paragraph directly
|
|
5738
|
-
const pStartMatch = xml.match(/<(hp|hs|hc):p[^>]
|
|
6343
|
+
const pStartMatch = xml.match(/<(hp|hs|hc):p(?:\s[^>]*)?>/);
|
|
5739
6344
|
if (pStartMatch) {
|
|
5740
6345
|
const prefix = pStartMatch[1];
|
|
5741
6346
|
const attrMatch = pStartMatch[0].match(/<(?:hp|hs|hc):p([^>]*)>/);
|
|
@@ -5763,7 +6368,7 @@ class HwpxDocument {
|
|
|
5763
6368
|
if (depth === 0) {
|
|
5764
6369
|
lastParagraphEnd = searchIndex;
|
|
5765
6370
|
const remainingXml = xml.substring(searchIndex);
|
|
5766
|
-
const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p[^>]
|
|
6371
|
+
const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p(?:\s[^>]*)?>/);
|
|
5767
6372
|
if (!nextPMatch)
|
|
5768
6373
|
break;
|
|
5769
6374
|
}
|
|
@@ -5794,7 +6399,7 @@ class HwpxDocument {
|
|
|
5794
6399
|
const charAttr = charShapeId !== undefined ? ` charPrIDRef="${charShapeId}"` : ' charPrIDRef="0"';
|
|
5795
6400
|
// Find the OUTER subList element with balanced tag matching
|
|
5796
6401
|
// This is crucial because cells can contain nested tables with their own subLists
|
|
5797
|
-
const subListStartMatch = cellXml.match(/<(hp|hs|hc):subList[^>]
|
|
6402
|
+
const subListStartMatch = cellXml.match(/<(hp|hs|hc):subList(?:\s[^>]*)?>/);
|
|
5798
6403
|
if (subListStartMatch) {
|
|
5799
6404
|
const prefix = subListStartMatch[1];
|
|
5800
6405
|
const startTag = subListStartMatch[0];
|
|
@@ -5832,7 +6437,7 @@ class HwpxDocument {
|
|
|
5832
6437
|
// IMPORTANT: Check for nested tables in subList content - preserve them!
|
|
5833
6438
|
const nestedTables = this.extractNestedTables(subListContent, prefix);
|
|
5834
6439
|
// Extract paraPrIDRef and styleIDRef from existing paragraph if available
|
|
5835
|
-
const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
|
|
6440
|
+
const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p\s[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
|
|
5836
6441
|
const paraPrIDRef = existingPMatch?.[1] || '0';
|
|
5837
6442
|
const styleIDRef = existingPMatch?.[2] || '0';
|
|
5838
6443
|
// Generate multiple paragraphs with chunked runs for long lines
|
|
@@ -5849,7 +6454,7 @@ class HwpxDocument {
|
|
|
5849
6454
|
}
|
|
5850
6455
|
// If no subList found, try to find just paragraphs and replace
|
|
5851
6456
|
// Use balanced matching for paragraphs too, since they can contain nested tables
|
|
5852
|
-
const pStartMatch = cellXml.match(/<(hp|hs|hc):p[^>]
|
|
6457
|
+
const pStartMatch = cellXml.match(/<(hp|hs|hc):p(?:\s[^>]*)?>/);
|
|
5853
6458
|
if (pStartMatch) {
|
|
5854
6459
|
const prefix = pStartMatch[1];
|
|
5855
6460
|
const firstPStart = cellXml.indexOf(pStartMatch[0]);
|
|
@@ -5883,7 +6488,7 @@ class HwpxDocument {
|
|
|
5883
6488
|
lastParagraphEnd = searchIndex;
|
|
5884
6489
|
// Check if there's another paragraph at top level
|
|
5885
6490
|
const remainingXml = cellXml.substring(searchIndex);
|
|
5886
|
-
const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p[^>]
|
|
6491
|
+
const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p(?:\s[^>]*)?>/);
|
|
5887
6492
|
if (!nextPMatch) {
|
|
5888
6493
|
// No more top-level paragraphs
|
|
5889
6494
|
break;
|
|
@@ -6008,7 +6613,7 @@ class HwpxDocument {
|
|
|
6008
6613
|
else {
|
|
6009
6614
|
// Text not found, fall back to first paragraph
|
|
6010
6615
|
console.warn(`[HwpxDocument] afterText "${insert.afterText}" not found in cell, using first paragraph`);
|
|
6011
|
-
const paragraphMatch = targetCell.xml.match(/<hp:p[^>]
|
|
6616
|
+
const paragraphMatch = targetCell.xml.match(/<hp:p(?:\s[^>]*)?>/);
|
|
6012
6617
|
if (!paragraphMatch)
|
|
6013
6618
|
continue;
|
|
6014
6619
|
insertPosition = targetCell.xml.indexOf(paragraphMatch[0]) + paragraphMatch[0].length;
|
|
@@ -6016,7 +6621,7 @@ class HwpxDocument {
|
|
|
6016
6621
|
}
|
|
6017
6622
|
else {
|
|
6018
6623
|
// Default: find the first <hp:p> in the cell and insert the image inside it
|
|
6019
|
-
const paragraphMatch = targetCell.xml.match(/<hp:p[^>]
|
|
6624
|
+
const paragraphMatch = targetCell.xml.match(/<hp:p(?:\s[^>]*)?>/);
|
|
6020
6625
|
if (!paragraphMatch)
|
|
6021
6626
|
continue;
|
|
6022
6627
|
insertPosition = targetCell.xml.indexOf(paragraphMatch[0]) + paragraphMatch[0].length;
|
|
@@ -6106,9 +6711,31 @@ class HwpxDocument {
|
|
|
6106
6711
|
async applyDirectTextUpdatesToXml() {
|
|
6107
6712
|
if (!this._zip)
|
|
6108
6713
|
return;
|
|
6714
|
+
// Re-anchor every update on the memory paragraph it edits. The element
|
|
6715
|
+
// index and id-occurrence recorded at call time are stale once a later
|
|
6716
|
+
// insert/delete/copy/move reshapes the section: the frozen occurrence then
|
|
6717
|
+
// names another same-id paragraph (measured: [A,B,C] all id="0", edit B,
|
|
6718
|
+
// move C to the front → A was rewritten). At this point the memory model
|
|
6719
|
+
// matches the XML, whose structural edits were already replayed.
|
|
6720
|
+
for (const update of this._pendingDirectTextUpdates) {
|
|
6721
|
+
if (!update.paragraph)
|
|
6722
|
+
continue;
|
|
6723
|
+
const elements = this._content.sections[update.sectionIndex]?.elements ?? [];
|
|
6724
|
+
const now = elements.findIndex(e => e.type === 'paragraph' && e.data === update.paragraph);
|
|
6725
|
+
if (now === -1) {
|
|
6726
|
+
// The paragraph was deleted after the edit; there is nothing to write.
|
|
6727
|
+
update.elementIndex = -1;
|
|
6728
|
+
continue;
|
|
6729
|
+
}
|
|
6730
|
+
update.elementIndex = now;
|
|
6731
|
+
update.paragraphId = update.paragraph.id || '';
|
|
6732
|
+
update.paragraphOccurrence = this.getParagraphOccurrence(update.sectionIndex, now, update.paragraphId);
|
|
6733
|
+
}
|
|
6109
6734
|
// Group updates by sectionIndex, then by elementIndex
|
|
6110
6735
|
const updatesBySectionAndElement = new Map();
|
|
6111
6736
|
for (const update of this._pendingDirectTextUpdates) {
|
|
6737
|
+
if (update.elementIndex < 0)
|
|
6738
|
+
continue;
|
|
6112
6739
|
let sectionMap = updatesBySectionAndElement.get(update.sectionIndex);
|
|
6113
6740
|
if (!sectionMap) {
|
|
6114
6741
|
sectionMap = new Map();
|
|
@@ -6127,25 +6754,38 @@ class HwpxDocument {
|
|
|
6127
6754
|
if (!file)
|
|
6128
6755
|
continue;
|
|
6129
6756
|
let xml = await file.async('string');
|
|
6130
|
-
// STEP 1: Pre-compute target paragraph
|
|
6131
|
-
//
|
|
6757
|
+
// STEP 1: Pre-compute target paragraph ranges BEFORE any modifications.
|
|
6758
|
+
//
|
|
6759
|
+
// Each memory paragraph is mapped to its XML paragraph with the parser's
|
|
6760
|
+
// own rule (parsedParagraphStarts), computed once per section. The offsets
|
|
6761
|
+
// the parser cached at load time are not used: they pair memory paragraphs
|
|
6762
|
+
// with a DIFFERENT list (top-level paragraphs of the raw XML), which drifts
|
|
6763
|
+
// wherever the parser lifts paragraphs out of headers, text boxes or
|
|
6764
|
+
// endnotes. Measured on 325 Hancom-saved sections: 16,271 of 70,677 cached
|
|
6765
|
+
// offsets pointed at another paragraph, and an edit then reported success
|
|
6766
|
+
// while its text went to — or vanished into — the wrong paragraph.
|
|
6132
6767
|
const paragraphTargets = new Map();
|
|
6768
|
+
const starts = this.parsedParagraphStarts(xml);
|
|
6769
|
+
const elements = this._content.sections[sectionIdx]?.elements ?? [];
|
|
6770
|
+
const slotOf = new Map();
|
|
6771
|
+
let slot = 0;
|
|
6772
|
+
elements.forEach((el, i) => {
|
|
6773
|
+
if (this.anchorKeyOf(el)?.kind === 'paragraph')
|
|
6774
|
+
slotOf.set(i, slot++);
|
|
6775
|
+
});
|
|
6776
|
+
const aligned = slot === starts.length;
|
|
6133
6777
|
for (const [elementIndex, updates] of elementMap) {
|
|
6134
|
-
|
|
6135
|
-
|
|
6136
|
-
|
|
6137
|
-
|
|
6138
|
-
|
|
6139
|
-
|
|
6140
|
-
paragraphTargets.set(elementIndex, {
|
|
6141
|
-
start: cachedPosition.start,
|
|
6142
|
-
end: cachedPosition.end,
|
|
6143
|
-
xml: cachedXml
|
|
6144
|
-
});
|
|
6778
|
+
const k = slotOf.get(elementIndex);
|
|
6779
|
+
if (aligned && k !== undefined) {
|
|
6780
|
+
const start = starts[k];
|
|
6781
|
+
const end = this.findBalancedParagraphEnd(xml, start);
|
|
6782
|
+
if (end !== -1) {
|
|
6783
|
+
paragraphTargets.set(elementIndex, { start, end, xml: xml.slice(start, end) });
|
|
6145
6784
|
continue;
|
|
6146
6785
|
}
|
|
6147
6786
|
}
|
|
6148
|
-
//
|
|
6787
|
+
// Memory and XML disagree on the paragraph count (should not happen for
|
|
6788
|
+
// parser-produced documents); fall back to id + occurrence search.
|
|
6149
6789
|
const paragraphId = updates[0]?.paragraphId || '';
|
|
6150
6790
|
const paragraphOccurrence = updates[0]?.paragraphOccurrence ?? 0;
|
|
6151
6791
|
const target = this.findTargetParagraphForUpdate(xml, sectionIdx, elementIndex, updates, paragraphId, paragraphOccurrence);
|
|
@@ -6170,11 +6810,11 @@ class HwpxDocument {
|
|
|
6170
6810
|
// Sort by runIndex to process in order
|
|
6171
6811
|
updates.sort((a, b) => a.runIndex - b.runIndex);
|
|
6172
6812
|
// Apply the update directly using pre-computed target location
|
|
6173
|
-
if (updates.length > 1) {
|
|
6813
|
+
if (updates.length > 1 || /<hp:t\b/.test(target.xml)) {
|
|
6174
6814
|
xml = this.replaceRunsInParagraphDirect(xml, target, updates);
|
|
6175
6815
|
}
|
|
6176
6816
|
else {
|
|
6177
|
-
//
|
|
6817
|
+
// Empty runs without text tags need a new hp:t element.
|
|
6178
6818
|
xml = this.replaceTextInElementDirect(xml, target, updates[0].oldText, updates[0].newText);
|
|
6179
6819
|
}
|
|
6180
6820
|
}
|
|
@@ -6449,10 +7089,10 @@ class HwpxDocument {
|
|
|
6449
7089
|
}
|
|
6450
7090
|
else if (/<hp:t\b[^>]*>/.test(run.xml)) {
|
|
6451
7091
|
// Has <hp:t>...</hp:t> tags - replace content of FIRST one only (no g flag)
|
|
6452
|
-
newRunXml = run.xml.replace(/(<hp:t[^>]
|
|
7092
|
+
newRunXml = run.xml.replace(/(<hp:t(?:\s[^>]*)?>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
|
|
6453
7093
|
// Remove any additional <hp:t>...</hp:t> tags to prevent duplication
|
|
6454
7094
|
let firstReplaced = false;
|
|
6455
|
-
newRunXml = newRunXml.replace(/<hp:t[^>]
|
|
7095
|
+
newRunXml = newRunXml.replace(/<hp:t(?:\s[^>]*)?>[^<]*<\/hp:t>/g, (match) => {
|
|
6456
7096
|
if (!firstReplaced) {
|
|
6457
7097
|
firstReplaced = true;
|
|
6458
7098
|
return match; // Keep the first one
|
|
@@ -6483,27 +7123,14 @@ class HwpxDocument {
|
|
|
6483
7123
|
return xml.slice(0, targetInOriginal.start) + newParagraphXml + xml.slice(targetInOriginal.end);
|
|
6484
7124
|
}
|
|
6485
7125
|
/**
|
|
6486
|
-
*
|
|
6487
|
-
*
|
|
7126
|
+
* Occurrence index of the paragraph at `elementIndex` among paragraphs with
|
|
7127
|
+
* the same id — counted with the same rule as insert anchors
|
|
7128
|
+
* (resolveElementAnchor), so a text update finds the paragraph that
|
|
7129
|
+
* findParsedParagraph resolves. Divider paragraphs parsed as 'hr' count.
|
|
6488
7130
|
*/
|
|
6489
7131
|
getParagraphOccurrence(sectionIndex, elementIndex, paragraphId) {
|
|
6490
|
-
|
|
6491
|
-
|
|
6492
|
-
return 0;
|
|
6493
|
-
const section = this._content.sections[sectionIndex];
|
|
6494
|
-
if (!section || !section.elements)
|
|
6495
|
-
return 0;
|
|
6496
|
-
let occurrenceCount = 0;
|
|
6497
|
-
for (let i = 0; i < elementIndex; i++) {
|
|
6498
|
-
const element = section.elements[i];
|
|
6499
|
-
if (element && element.type === 'paragraph') { // Use 'paragraph' not 'p'
|
|
6500
|
-
const para = element.data;
|
|
6501
|
-
if (para.id === paragraphId) {
|
|
6502
|
-
occurrenceCount++;
|
|
6503
|
-
}
|
|
6504
|
-
}
|
|
6505
|
-
}
|
|
6506
|
-
return occurrenceCount;
|
|
7132
|
+
const anchor = this.resolveElementAnchor(sectionIndex, elementIndex);
|
|
7133
|
+
return anchor && anchor.kind === 'paragraph' && anchor.id === paragraphId ? anchor.occurrence : 0;
|
|
6507
7134
|
}
|
|
6508
7135
|
/**
|
|
6509
7136
|
* Find paragraph by its ID attribute and occurrence index.
|
|
@@ -6799,18 +7426,22 @@ class HwpxDocument {
|
|
|
6799
7426
|
// TIER 1: ID-based lookup (most reliable)
|
|
6800
7427
|
// TIER 2: Index-based lookup with text validation
|
|
6801
7428
|
// TIER 3: Fuzzy text matching fallback
|
|
6802
|
-
// TIER 1:
|
|
6803
|
-
// Problem: XML counting includes nested paragraphs (inside tables),
|
|
6804
|
-
// but _content.sections.elements only has top-level elements.
|
|
6805
|
-
// This mismatch causes wrong paragraph selection.
|
|
6806
|
-
// Solution: Skip ID-based lookup and use index-based (TIER 2) instead.
|
|
7429
|
+
// TIER 1: id + occurrence, counted with the parser's paragraph rule.
|
|
6807
7430
|
//
|
|
6808
|
-
//
|
|
6809
|
-
//
|
|
6810
|
-
//
|
|
6811
|
-
//
|
|
6812
|
-
//
|
|
6813
|
-
//
|
|
7431
|
+
// This was disabled because counting every <hp:p> in the XML included cell
|
|
7432
|
+
// paragraphs the memory model does not have. findAnchorEnd counts only
|
|
7433
|
+
// top-level paragraphs the parser keeps, so the occurrence recorded from
|
|
7434
|
+
// the memory model names the same node. Index lookup (TIER 2) is wrong
|
|
7435
|
+
// after a copy: its ±2 text search finds the original first because the
|
|
7436
|
+
// copy carries the same text, and the edit lands on the original.
|
|
7437
|
+
if (paragraphId) {
|
|
7438
|
+
// The paragraph's OWN range. For a paragraph inside a header or text
|
|
7439
|
+
// box, the enclosing top-level paragraph would rewrite the whole body.
|
|
7440
|
+
const hit = this.findParsedParagraph(xml, { kind: 'paragraph', id: paragraphId, occurrence: paragraphOccurrence ?? 0 });
|
|
7441
|
+
if (hit) {
|
|
7442
|
+
return { start: hit.start, end: hit.end, xml: xml.slice(hit.start, hit.end) };
|
|
7443
|
+
}
|
|
7444
|
+
}
|
|
6814
7445
|
// Calculate paragraph index using _content.sections.elements (same source as elementIndex)
|
|
6815
7446
|
// This ensures consistency between elementIndex and paragraph counting
|
|
6816
7447
|
let topLevelParagraphIndex = 0;
|
|
@@ -6875,68 +7506,151 @@ class HwpxDocument {
|
|
|
6875
7506
|
for (const update of updates) {
|
|
6876
7507
|
updateMap.set(update.runIndex, update.newText);
|
|
6877
7508
|
}
|
|
6878
|
-
//
|
|
6879
|
-
//
|
|
6880
|
-
|
|
6881
|
-
|
|
6882
|
-
|
|
6883
|
-
|
|
6884
|
-
|
|
6885
|
-
|
|
6886
|
-
|
|
6887
|
-
|
|
6888
|
-
|
|
6889
|
-
|
|
6890
|
-
|
|
6891
|
-
|
|
6892
|
-
|
|
6893
|
-
|
|
6894
|
-
|
|
6895
|
-
|
|
7509
|
+
// The paragraph's OWN runs only — its direct children. A paragraph that
|
|
7510
|
+
// holds a table, text box, footnote or endnote also contains the runs of
|
|
7511
|
+
// every paragraph inside those containers. Counting those as its own made
|
|
7512
|
+
// "run N" land in a table cell or endnote: the reported success wrote the
|
|
7513
|
+
// new text into a nested paragraph (or into nothing) and cut the rest.
|
|
7514
|
+
// Measured: 39 of 60 Hancom files lost the text this way (2026-09-24).
|
|
7515
|
+
const runs = this.findDirectChildRuns(paragraphXml);
|
|
7516
|
+
// Filter to only runs that have <hp:t> content (matching memory model behavior)
|
|
7517
|
+
// Memory model only counts runs with text, not runs with only <hp:ctrl> etc.
|
|
7518
|
+
const textRuns = runs.filter(run => /<hp:t\b/.test(this.ownRunText(run.xml)));
|
|
7519
|
+
// The parser creates a model run per non-empty hp:t, not per hp:run.
|
|
7520
|
+
// Merge those updates back into their shared XML run without losing a suffix.
|
|
7521
|
+
const xmlRunUpdates = new Map();
|
|
7522
|
+
let modelRunIndex = 0;
|
|
7523
|
+
for (let i = 0; i < textRuns.length; i++) {
|
|
7524
|
+
const textNodes = [...this.ownRunText(textRuns[i].xml).matchAll(/<hp:t\b[^>]*>([^<]+)<\/hp:t>/g)];
|
|
7525
|
+
const count = Math.max(1, textNodes.length);
|
|
7526
|
+
let changed = false;
|
|
7527
|
+
let escapedText = '';
|
|
7528
|
+
for (let offset = 0; offset < count; offset++) {
|
|
7529
|
+
const index = modelRunIndex + offset;
|
|
7530
|
+
if (updateMap.has(index)) {
|
|
7531
|
+
escapedText += this.escapeXml(updateMap.get(index));
|
|
7532
|
+
changed = true;
|
|
6896
7533
|
}
|
|
6897
7534
|
else {
|
|
6898
|
-
|
|
6899
|
-
if (depth === 0) {
|
|
6900
|
-
const runEnd = nextClose + '</hp:run>'.length;
|
|
6901
|
-
runs.push({
|
|
6902
|
-
start: runStart,
|
|
6903
|
-
end: runEnd,
|
|
6904
|
-
xml: paragraphXml.slice(runStart, runEnd)
|
|
6905
|
-
});
|
|
6906
|
-
}
|
|
6907
|
-
pos = nextClose + 9;
|
|
7535
|
+
escapedText += textNodes[offset]?.[1] || '';
|
|
6908
7536
|
}
|
|
6909
7537
|
}
|
|
7538
|
+
if (changed)
|
|
7539
|
+
xmlRunUpdates.set(i, escapedText);
|
|
7540
|
+
modelRunIndex += count;
|
|
6910
7541
|
}
|
|
6911
|
-
// Filter to only runs that have <hp:t> content (matching memory model behavior)
|
|
6912
|
-
// Memory model only counts runs with text, not runs with only <hp:ctrl> etc.
|
|
6913
|
-
const textRuns = runs.filter(run => /<hp:t\b/.test(run.xml) || /<hp:t\s*\/>/.test(run.xml));
|
|
6914
7542
|
// Process text runs in reverse order to maintain positions
|
|
6915
7543
|
for (let i = textRuns.length - 1; i >= 0; i--) {
|
|
6916
|
-
if (!
|
|
7544
|
+
if (!xmlRunUpdates.has(i))
|
|
6917
7545
|
continue;
|
|
6918
7546
|
const run = textRuns[i];
|
|
6919
|
-
const
|
|
6920
|
-
|
|
6921
|
-
|
|
6922
|
-
//
|
|
6923
|
-
|
|
6924
|
-
|
|
6925
|
-
|
|
6926
|
-
|
|
6927
|
-
|
|
6928
|
-
|
|
6929
|
-
newRunXml = newRunXml.replace(/(<hp:t\b[^>]*>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
|
|
6930
|
-
}
|
|
6931
|
-
else {
|
|
6932
|
-
// No hp:t tag - add one after the opening hp:run tag
|
|
6933
|
-
newRunXml = newRunXml.replace(/(<hp:run\b[^>]*>)/, `$1<hp:t>${escapedNew}</hp:t>`);
|
|
6934
|
-
}
|
|
7547
|
+
const escapedNew = xmlRunUpdates.get(i);
|
|
7548
|
+
// Write each XML run's combined text once, preserving text-tag attributes.
|
|
7549
|
+
// Only the run's own <hp:t> are rewritten; text inside a table, equation
|
|
7550
|
+
// or text box that sits in the same run is left untouched.
|
|
7551
|
+
let textWritten = false;
|
|
7552
|
+
const newRunXml = this.mapOwnRunText(run.xml, tXml => tXml.replace(/<hp:t\b([^>]*?)\/>|<hp:t\b([^>]*)>[^<]*<\/hp:t>/g, (_match, selfClosingAttrs, attrs) => {
|
|
7553
|
+
const text = textWritten ? '' : escapedNew;
|
|
7554
|
+
textWritten = true;
|
|
7555
|
+
return `<hp:t${selfClosingAttrs ?? attrs ?? ''}>${text}</hp:t>`;
|
|
7556
|
+
}));
|
|
6935
7557
|
// Replace in paragraph XML
|
|
6936
7558
|
paragraphXml = paragraphXml.slice(0, run.start) + newRunXml + paragraphXml.slice(run.end);
|
|
6937
7559
|
}
|
|
6938
7560
|
return xml.slice(0, target.start) + paragraphXml + xml.slice(target.end);
|
|
6939
7561
|
}
|
|
7562
|
+
/** Direct <hp:run> children of a paragraph (runs of nested paragraphs excluded). */
|
|
7563
|
+
findDirectChildRuns(paragraphXml) {
|
|
7564
|
+
const runs = [];
|
|
7565
|
+
const openEnd = paragraphXml.indexOf('>') + 1;
|
|
7566
|
+
let pos = openEnd;
|
|
7567
|
+
let depth = 0; // nesting depth of <hp:p> inside this paragraph
|
|
7568
|
+
const tagRe = /<(\/?)hp:(p|run)\b[^>]*?(\/?)>/g;
|
|
7569
|
+
tagRe.lastIndex = pos;
|
|
7570
|
+
let runStart = -1;
|
|
7571
|
+
let m;
|
|
7572
|
+
while ((m = tagRe.exec(paragraphXml)) !== null) {
|
|
7573
|
+
const [whole, closing, name, selfClosing] = m;
|
|
7574
|
+
if (name === 'p') {
|
|
7575
|
+
if (selfClosing)
|
|
7576
|
+
continue;
|
|
7577
|
+
if (closing) {
|
|
7578
|
+
if (depth === 0)
|
|
7579
|
+
break; // end of this paragraph
|
|
7580
|
+
depth--;
|
|
7581
|
+
}
|
|
7582
|
+
else {
|
|
7583
|
+
depth++;
|
|
7584
|
+
}
|
|
7585
|
+
continue;
|
|
7586
|
+
}
|
|
7587
|
+
if (depth !== 0)
|
|
7588
|
+
continue; // a run of a nested paragraph
|
|
7589
|
+
if (selfClosing) {
|
|
7590
|
+
runs.push({ start: m.index, end: m.index + whole.length, xml: whole });
|
|
7591
|
+
}
|
|
7592
|
+
else if (!closing) {
|
|
7593
|
+
runStart = m.index;
|
|
7594
|
+
}
|
|
7595
|
+
else if (runStart !== -1) {
|
|
7596
|
+
const end = m.index + whole.length;
|
|
7597
|
+
runs.push({ start: runStart, end, xml: paragraphXml.slice(runStart, end) });
|
|
7598
|
+
runStart = -1;
|
|
7599
|
+
}
|
|
7600
|
+
}
|
|
7601
|
+
return runs;
|
|
7602
|
+
}
|
|
7603
|
+
/**
|
|
7604
|
+
* A run's own markup with every nested container (table, equation, text box,
|
|
7605
|
+
* note…) blanked out, so its <hp:t> are the run's own text only.
|
|
7606
|
+
*/
|
|
7607
|
+
ownRunText(runXml) {
|
|
7608
|
+
return this.mapOwnRunText(runXml, s => s, true);
|
|
7609
|
+
}
|
|
7610
|
+
/**
|
|
7611
|
+
* Apply `fn` to the parts of a run that are its own text, leaving nested
|
|
7612
|
+
* containers byte-for-byte intact. With `blank`, nested containers are
|
|
7613
|
+
* replaced by an empty marker instead (for reading).
|
|
7614
|
+
*/
|
|
7615
|
+
mapOwnRunText(runXml, fn, blank = false) {
|
|
7616
|
+
let out = '';
|
|
7617
|
+
let pos = 0;
|
|
7618
|
+
while (pos < runXml.length) {
|
|
7619
|
+
const rest = runXml.slice(pos);
|
|
7620
|
+
const m = rest.match(HwpxDocument.NESTED_CONTENT);
|
|
7621
|
+
if (!m || m.index === undefined) {
|
|
7622
|
+
out += fn(rest);
|
|
7623
|
+
break;
|
|
7624
|
+
}
|
|
7625
|
+
const openAt = pos + m.index;
|
|
7626
|
+
const name = m[1];
|
|
7627
|
+
out += fn(runXml.slice(pos, openAt));
|
|
7628
|
+
const end = this.findElementEnd(runXml, openAt, name);
|
|
7629
|
+
out += blank ? '<NESTED/>' : runXml.slice(openAt, end);
|
|
7630
|
+
pos = end;
|
|
7631
|
+
}
|
|
7632
|
+
return out;
|
|
7633
|
+
}
|
|
7634
|
+
/** End offset of the <hp:name> element opening at `start` (handles nesting and self-closing). */
|
|
7635
|
+
findElementEnd(xml, start, name) {
|
|
7636
|
+
const tagEnd = xml.indexOf('>', start);
|
|
7637
|
+
if (tagEnd === -1)
|
|
7638
|
+
return xml.length;
|
|
7639
|
+
if (xml[tagEnd - 1] === '/')
|
|
7640
|
+
return tagEnd + 1;
|
|
7641
|
+
const re = new RegExp(`<(/?)hp:${name}\\b[^>]*?(/?)>`, 'g');
|
|
7642
|
+
re.lastIndex = tagEnd + 1;
|
|
7643
|
+
let depth = 1;
|
|
7644
|
+
let m;
|
|
7645
|
+
while ((m = re.exec(xml)) !== null) {
|
|
7646
|
+
if (m[2])
|
|
7647
|
+
continue;
|
|
7648
|
+
depth += m[1] ? -1 : 1;
|
|
7649
|
+
if (depth === 0)
|
|
7650
|
+
return m.index + m[0].length;
|
|
7651
|
+
}
|
|
7652
|
+
return xml.length;
|
|
7653
|
+
}
|
|
6940
7654
|
/**
|
|
6941
7655
|
* Replace text in a single run directly using pre-computed target location.
|
|
6942
7656
|
* Simpler version for single-run updates.
|
|
@@ -6950,9 +7664,9 @@ class HwpxDocument {
|
|
|
6950
7664
|
// Self-closing: <hp:t/> -> <hp:t>newText</hp:t>
|
|
6951
7665
|
paragraphXml = paragraphXml.replace(/<hp:t\s*\/>/, `<hp:t>${escapedNew}</hp:t>`);
|
|
6952
7666
|
}
|
|
6953
|
-
else if (/<hp:t[^>]
|
|
7667
|
+
else if (/<hp:t(?:\s[^>]*)?>/.test(paragraphXml)) {
|
|
6954
7668
|
// Has content or empty: <hp:t>...</hp:t> -> <hp:t>newText</hp:t>
|
|
6955
|
-
paragraphXml = paragraphXml.replace(/(<hp:t[^>]
|
|
7669
|
+
paragraphXml = paragraphXml.replace(/(<hp:t(?:\s[^>]*)?>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
|
|
6956
7670
|
}
|
|
6957
7671
|
else if (/<hp:run\b[^>]*>/.test(paragraphXml)) {
|
|
6958
7672
|
// No <hp:t> tag exists - add one after the <hp:run> opening tag
|
|
@@ -7012,7 +7726,7 @@ class HwpxDocument {
|
|
|
7012
7726
|
continue;
|
|
7013
7727
|
// Found the right paragraph! Replace the text
|
|
7014
7728
|
// Replace within <hp:t> tags
|
|
7015
|
-
const pattern1 = new RegExp(`(<hp:t[^>]
|
|
7729
|
+
const pattern1 = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}`);
|
|
7016
7730
|
let newParagraphContent = paragraphContent.replace(pattern1, `$1${escapedNew}`);
|
|
7017
7731
|
// Also try standalone text replacement
|
|
7018
7732
|
const pattern2 = new RegExp(`>${this.escapeRegex(escapedOld)}<`);
|
|
@@ -7275,9 +7989,9 @@ class HwpxDocument {
|
|
|
7275
7989
|
// Case 1: Self-closing <hp:t/> - replace with full tag containing new text
|
|
7276
7990
|
newElementContent = elementContent.replace(/<hp:t\s*\/>/, `<hp:t>${escapedNew}</hp:t>`);
|
|
7277
7991
|
}
|
|
7278
|
-
else if (oldText === '' && /<hp:t[^>]
|
|
7992
|
+
else if (oldText === '' && /<hp:t(?:\s[^>]*)?><\/hp:t>/.test(elementContent)) {
|
|
7279
7993
|
// Case 2: Empty <hp:t></hp:t> - fill with new text
|
|
7280
|
-
newElementContent = elementContent.replace(/(<hp:t[^>]
|
|
7994
|
+
newElementContent = elementContent.replace(/(<hp:t(?:\s[^>]*)?>)<\/hp:t>/, `$1${escapedNew}</hp:t>`);
|
|
7281
7995
|
}
|
|
7282
7996
|
else if (oldText === '' && !/<hp:t\b[^>]*>/.test(elementContent)) {
|
|
7283
7997
|
// Case 3: No hp:t tag at all - add one after the first hp:run opening tag
|
|
@@ -7285,7 +7999,7 @@ class HwpxDocument {
|
|
|
7285
7999
|
}
|
|
7286
8000
|
else {
|
|
7287
8001
|
// Case 4: Normal case - replace text within <hp:t> tags (first match only)
|
|
7288
|
-
const pattern1 = new RegExp(`(<hp:t[^>]
|
|
8002
|
+
const pattern1 = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}`);
|
|
7289
8003
|
newElementContent = elementContent.replace(pattern1, `$1${escapedNew}`);
|
|
7290
8004
|
// Also try standalone text replacement if pattern1 didn't match
|
|
7291
8005
|
if (newElementContent === elementContent) {
|
|
@@ -7358,7 +8072,7 @@ class HwpxDocument {
|
|
|
7358
8072
|
if (runIndex >= runs.length) {
|
|
7359
8073
|
// Run index out of bounds, try to replace in any run
|
|
7360
8074
|
// Replace text within <hp:t> tags (first match only)
|
|
7361
|
-
const pattern1 = new RegExp(`(<hp:t[^>]
|
|
8075
|
+
const pattern1 = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}`);
|
|
7362
8076
|
let newParagraphContent = paragraphContent.replace(pattern1, `$1${escapedNew}`);
|
|
7363
8077
|
// Also try standalone text replacement
|
|
7364
8078
|
if (newParagraphContent === paragraphContent) {
|
|
@@ -7371,7 +8085,7 @@ class HwpxDocument {
|
|
|
7371
8085
|
const targetRun = runs[runIndex];
|
|
7372
8086
|
let newRunContent = targetRun.content;
|
|
7373
8087
|
// Replace within <hp:t> tags in this run
|
|
7374
|
-
const tPattern = new RegExp(`(<hp:t[^>]
|
|
8088
|
+
const tPattern = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}(</hp:t>)`);
|
|
7375
8089
|
newRunContent = newRunContent.replace(tPattern, `$1${escapedNew}$2`);
|
|
7376
8090
|
// If no match, try simpler pattern
|
|
7377
8091
|
if (newRunContent === targetRun.content) {
|
|
@@ -7553,7 +8267,7 @@ class HwpxDocument {
|
|
|
7553
8267
|
return tblMatch;
|
|
7554
8268
|
}
|
|
7555
8269
|
let rowIndex = 0;
|
|
7556
|
-
return tblMatch.replace(/<hp:tr[^>]
|
|
8270
|
+
return tblMatch.replace(/<hp:tr(?:\s[^>]*)?>([\s\S]*?)<\/hp:tr>/g, (rowMatch) => {
|
|
7557
8271
|
if (rowIndex >= table.rows.length) {
|
|
7558
8272
|
rowIndex++;
|
|
7559
8273
|
return rowMatch;
|
|
@@ -8330,6 +9044,84 @@ class HwpxDocument {
|
|
|
8330
9044
|
contentHpf = contentHpf.substring(0, insertPos) + newItem + contentHpf.substring(insertPos);
|
|
8331
9045
|
this._zip.file('Contents/content.hpf', contentHpf);
|
|
8332
9046
|
}
|
|
9047
|
+
/**
|
|
9048
|
+
* Apply section inserts/deletes to Contents/sectionN.xml, in call order.
|
|
9049
|
+
*
|
|
9050
|
+
* File numbers must keep matching memory section indices, so an insert
|
|
9051
|
+
* renames later files up one (section1 → section2 …) and a delete removes
|
|
9052
|
+
* its file and renames later files down one. content.hpf gets a manifest
|
|
9053
|
+
* item and a spine itemref per section, and header.xml's secCnt follows.
|
|
9054
|
+
*/
|
|
9055
|
+
async applySectionOpsToZip() {
|
|
9056
|
+
if (!this._zip)
|
|
9057
|
+
return;
|
|
9058
|
+
const secPath = (i) => `Contents/section${i}.xml`;
|
|
9059
|
+
const countFiles = () => Object.keys(this._zip.files).filter(n => /^Contents\/section\d+\.xml$/.test(n)).length;
|
|
9060
|
+
const move = async (from, to) => {
|
|
9061
|
+
const f = this._zip.file(secPath(from));
|
|
9062
|
+
if (!f)
|
|
9063
|
+
return;
|
|
9064
|
+
this._zip.file(secPath(to), await f.async('string'));
|
|
9065
|
+
this._zip.remove(secPath(from));
|
|
9066
|
+
};
|
|
9067
|
+
for (const op of this._pendingSectionOps) {
|
|
9068
|
+
const fileCount = countFiles();
|
|
9069
|
+
if (op.op === 'delete') {
|
|
9070
|
+
if (op.at >= fileCount || fileCount <= 1)
|
|
9071
|
+
continue;
|
|
9072
|
+
this._zip.remove(secPath(op.at));
|
|
9073
|
+
for (let i = op.at + 1; i < fileCount; i++)
|
|
9074
|
+
await move(i, i - 1);
|
|
9075
|
+
continue;
|
|
9076
|
+
}
|
|
9077
|
+
// Insert: shift later files up, highest first.
|
|
9078
|
+
for (let i = fileCount - 1; i >= op.at; i--)
|
|
9079
|
+
await move(i, i + 1);
|
|
9080
|
+
// Build the new section from the template section's <hs:sec> wrapper and
|
|
9081
|
+
// its first paragraph's <hp:secPr> (page size, margins, numbering).
|
|
9082
|
+
const templateIndex = op.templateFrom >= op.at ? op.templateFrom + 1 : op.templateFrom;
|
|
9083
|
+
const template = await this._zip.file(secPath(templateIndex))?.async('string');
|
|
9084
|
+
this._zip.file(secPath(op.at), this.buildEmptySectionXml(template));
|
|
9085
|
+
}
|
|
9086
|
+
// Manifest + spine: one item per section file, in order.
|
|
9087
|
+
const hpfFile = this._zip.file('Contents/content.hpf');
|
|
9088
|
+
const total = countFiles();
|
|
9089
|
+
if (hpfFile) {
|
|
9090
|
+
let hpf = await hpfFile.async('string');
|
|
9091
|
+
hpf = hpf.replace(/<opf:item\b[^>]*\bid="section\d+"[^>]*\/>\s*/g, '');
|
|
9092
|
+
hpf = hpf.replace(/<opf:itemref\b[^>]*\bidref="section\d+"[^>]*\/>\s*/g, '');
|
|
9093
|
+
const items = Array.from({ length: total }, (_, i) => `<opf:item id="section${i}" href="Contents/section${i}.xml" media-type="application/xml"/>`).join('');
|
|
9094
|
+
const refs = Array.from({ length: total }, (_, i) => `<opf:itemref idref="section${i}" linear="yes"/>`).join('');
|
|
9095
|
+
hpf = hpf.replace('</opf:manifest>', items + '</opf:manifest>');
|
|
9096
|
+
hpf = hpf.replace('</opf:spine>', refs + '</opf:spine>');
|
|
9097
|
+
this._zip.file('Contents/content.hpf', hpf);
|
|
9098
|
+
}
|
|
9099
|
+
const headerFile = this._zip.file('Contents/header.xml');
|
|
9100
|
+
if (headerFile) {
|
|
9101
|
+
const header = await headerFile.async('string');
|
|
9102
|
+
this._zip.file('Contents/header.xml', header.replace(/\bsecCnt="\d+"/, `secCnt="${total}"`));
|
|
9103
|
+
}
|
|
9104
|
+
}
|
|
9105
|
+
/** A section XML holding one empty paragraph with the template's <hp:secPr>. */
|
|
9106
|
+
buildEmptySectionXml(template) {
|
|
9107
|
+
const declaration = '<?xml version="1.0" encoding="UTF-8" standalone="yes" ?>';
|
|
9108
|
+
const secOpen = template?.match(/<hs:sec\b[^>]*>/)?.[0]
|
|
9109
|
+
?? '<hs:sec xmlns:hp="http://www.hancom.co.kr/hwpml/2011/paragraph" xmlns:hs="http://www.hancom.co.kr/hwpml/2011/section">';
|
|
9110
|
+
let secPr = '';
|
|
9111
|
+
if (template) {
|
|
9112
|
+
const at = template.indexOf('<hp:secPr');
|
|
9113
|
+
if (at !== -1)
|
|
9114
|
+
secPr = template.slice(at, this.findElementEnd(template, at, 'secPr'));
|
|
9115
|
+
}
|
|
9116
|
+
// A fresh column definition follows secPr in Hancom's own first paragraph.
|
|
9117
|
+
const colPr = template?.match(/<hp:ctrl>\s*<hp:colPr\b[^>]*\/>\s*<\/hp:ctrl>/)?.[0] ?? '';
|
|
9118
|
+
return `${declaration}${secOpen}` +
|
|
9119
|
+
`<hp:p id="0" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0">` +
|
|
9120
|
+
`<hp:run charPrIDRef="0">${secPr}${colPr}</hp:run>` +
|
|
9121
|
+
`<hp:run charPrIDRef="0"><hp:t></hp:t></hp:run>` +
|
|
9122
|
+
`<hp:linesegarray><hp:lineseg textpos="0" vertpos="0" vertsize="1000" textheight="1000" baseline="850" spacing="600" horzpos="0" horzsize="0" flags="393216"/></hp:linesegarray>` +
|
|
9123
|
+
`</hp:p></hs:sec>`;
|
|
9124
|
+
}
|
|
8333
9125
|
/**
|
|
8334
9126
|
* Add hp:pic tag to section XML
|
|
8335
9127
|
*/
|
|
@@ -8432,7 +9224,7 @@ class HwpxDocument {
|
|
|
8432
9224
|
for (const table of tables) {
|
|
8433
9225
|
const tableXml = xml.substring(table.startIndex, table.endIndex);
|
|
8434
9226
|
// Find cells in this table
|
|
8435
|
-
const cellMatches = [...tableXml.matchAll(/<(?:hp|hs):tc[^>]
|
|
9227
|
+
const cellMatches = [...tableXml.matchAll(/<(?:hp|hs):tc(?:\s[^>]*)?>([\s\S]*?)<\/(?:hp|hs):tc>/g)];
|
|
8436
9228
|
for (const cellMatch of cellMatches) {
|
|
8437
9229
|
const cellContent = cellMatch[1];
|
|
8438
9230
|
const textContent = this.extractTextFromCellXml(cellContent);
|
|
@@ -8464,7 +9256,7 @@ class HwpxDocument {
|
|
|
8464
9256
|
*/
|
|
8465
9257
|
findAllParagraphsInCell(cellXml) {
|
|
8466
9258
|
const paragraphs = [];
|
|
8467
|
-
const paragraphRegex = /<hp:p[^>]
|
|
9259
|
+
const paragraphRegex = /<hp:p(?:\s[^>]*)?>[\s\S]*?<\/hp:p>/g;
|
|
8468
9260
|
let match;
|
|
8469
9261
|
while ((match = paragraphRegex.exec(cellXml)) !== null) {
|
|
8470
9262
|
paragraphs.push({
|
|
@@ -8647,7 +9439,7 @@ class HwpxDocument {
|
|
|
8647
9439
|
findTblTagIssues(xml) {
|
|
8648
9440
|
const issues = [];
|
|
8649
9441
|
// Track table tag positions
|
|
8650
|
-
const tblOpenRegex = /<(?:hp|hs|hc):tbl[^>]
|
|
9442
|
+
const tblOpenRegex = /<(?:hp|hs|hc):tbl(?:\s[^>]*)?>/g;
|
|
8651
9443
|
const tblCloseRegex = /<\/(?:hp|hs|hc):tbl>/g;
|
|
8652
9444
|
const allPositions = [];
|
|
8653
9445
|
let match;
|
|
@@ -8700,11 +9492,11 @@ class HwpxDocument {
|
|
|
8700
9492
|
checkNestingErrors(xml) {
|
|
8701
9493
|
const issues = [];
|
|
8702
9494
|
// Check for tc outside of tr
|
|
8703
|
-
const tcOutsideTr = /<(?:hp|hs|hc):tc[^>]
|
|
9495
|
+
const tcOutsideTr = /<(?:hp|hs|hc):tc(?:\s[^>]*)?>(?:(?!<(?:hp|hs|hc):tr(?:\s[^>]*)?>).)*?<\/(?:hp|hs|hc):tc>/gs;
|
|
8704
9496
|
// This is simplified - a full check would need proper nesting validation
|
|
8705
9497
|
// Check for tr outside of tbl
|
|
8706
|
-
const trPattern = /<(?:hp|hs|hc):tr[^>]
|
|
8707
|
-
const tblPattern = /<(?:hp|hs|hc):tbl[^>]
|
|
9498
|
+
const trPattern = /<(?:hp|hs|hc):tr(?:\s[^>]*)?>/g;
|
|
9499
|
+
const tblPattern = /<(?:hp|hs|hc):tbl(?:\s[^>]*)?>/g;
|
|
8708
9500
|
// Simple check: count if tr appears without preceding tbl
|
|
8709
9501
|
let match;
|
|
8710
9502
|
let lastTblPos = -1;
|
|
@@ -10247,7 +11039,7 @@ class HwpxDocument {
|
|
|
10247
11039
|
return null;
|
|
10248
11040
|
const targetRowData = rows[targetRow];
|
|
10249
11041
|
// Extract content inside the row (between <hp:tr...> and </hp:tr>)
|
|
10250
|
-
const rowOpenTagMatch = targetRowData.xml.match(/^<(?:hp|hs|hc):tr[^>]
|
|
11042
|
+
const rowOpenTagMatch = targetRowData.xml.match(/^<(?:hp|hs|hc):tr(?:\s[^>]*)?>/);
|
|
10251
11043
|
if (!rowOpenTagMatch)
|
|
10252
11044
|
return null;
|
|
10253
11045
|
const rowContentStart = rowOpenTagMatch[0].length;
|
|
@@ -10261,7 +11053,7 @@ class HwpxDocument {
|
|
|
10261
11053
|
return null;
|
|
10262
11054
|
const targetCellData = cells[targetCol];
|
|
10263
11055
|
// Extract content inside the cell (between <hp:tc...> and </hp:tc>)
|
|
10264
|
-
const cellOpenTagMatch = targetCellData.xml.match(/^<(?:hp|hs|hc):tc[^>]
|
|
11056
|
+
const cellOpenTagMatch = targetCellData.xml.match(/^<(?:hp|hs|hc):tc(?:\s[^>]*)?>/);
|
|
10265
11057
|
if (!cellOpenTagMatch)
|
|
10266
11058
|
return null;
|
|
10267
11059
|
const cellContentStart = cellOpenTagMatch[0].length;
|
|
@@ -10284,6 +11076,219 @@ class HwpxDocument {
|
|
|
10284
11076
|
// ============================================================
|
|
10285
11077
|
// Table Row Insert/Delete XML Persistence
|
|
10286
11078
|
// ============================================================
|
|
11079
|
+
/**
|
|
11080
|
+
* Scale this table's column widths so they sum to its <hp:sz width>.
|
|
11081
|
+
*
|
|
11082
|
+
* Column widths are read from cells whose colSpan is 1 (the first one seen
|
|
11083
|
+
* per colAddr). Every cell then gets the sum of the scaled widths of the
|
|
11084
|
+
* columns it spans, so merged cells stay aligned. Rounding leftovers go to
|
|
11085
|
+
* the last column so the total is exact. Nested tables are not touched.
|
|
11086
|
+
*/
|
|
11087
|
+
fitColumnsToTableWidth(tableXml) {
|
|
11088
|
+
const tableWidth = parseInt(tableXml.match(/^<hp:tbl\b[\s\S]*?<hp:sz width="(\d+)"/)?.[1] ?? '', 10);
|
|
11089
|
+
const colCnt = parseInt(tableXml.match(/^<hp:tbl\b[^>]*\bcolCnt="(\d+)"/)?.[1] ?? '', 10);
|
|
11090
|
+
if (!tableWidth || !colCnt)
|
|
11091
|
+
return tableXml;
|
|
11092
|
+
const rows = this.findAllElementsWithDepth(tableXml, 'tr');
|
|
11093
|
+
const own = [];
|
|
11094
|
+
rows.forEach((row, r) => {
|
|
11095
|
+
for (const cell of this.findAllElementsWithDepth(row.xml, 'tc')) {
|
|
11096
|
+
const tail = cell.xml.lastIndexOf('</hp:subList>');
|
|
11097
|
+
const from = tail === -1 ? 0 : tail;
|
|
11098
|
+
const props = cell.xml.slice(from);
|
|
11099
|
+
const col = parseInt(props.match(/<hp:cellAddr\b[^>]*\bcolAddr="(\d+)"/)?.[1] ?? '-1', 10);
|
|
11100
|
+
const span = parseInt(props.match(/<hp:cellSpan\b[^>]*\bcolSpan="(\d+)"/)?.[1] ?? '1', 10);
|
|
11101
|
+
const sz = props.match(/(<hp:cellSz\b[^>]*\bwidth=")(\d+)(")/);
|
|
11102
|
+
if (col < 0 || !sz || sz.index === undefined)
|
|
11103
|
+
continue;
|
|
11104
|
+
own.push({ row: r, cell, col, span, width: parseInt(sz[2], 10), at: from + sz.index + sz[1].length });
|
|
11105
|
+
}
|
|
11106
|
+
});
|
|
11107
|
+
const widths = new Array(colCnt).fill(0);
|
|
11108
|
+
for (const o of own)
|
|
11109
|
+
if (o.span === 1 && o.col < colCnt && widths[o.col] === 0)
|
|
11110
|
+
widths[o.col] = o.width;
|
|
11111
|
+
if (widths.some(w => w === 0))
|
|
11112
|
+
return tableXml; // cannot derive every column safely
|
|
11113
|
+
const sum = widths.reduce((a, b) => a + b, 0);
|
|
11114
|
+
if (sum === tableWidth)
|
|
11115
|
+
return tableXml;
|
|
11116
|
+
const scaled = widths.map(w => Math.floor((w * tableWidth) / sum));
|
|
11117
|
+
scaled[colCnt - 1] += tableWidth - scaled.reduce((a, b) => a + b, 0);
|
|
11118
|
+
let out = tableXml;
|
|
11119
|
+
for (let r = rows.length - 1; r >= 0; r--) {
|
|
11120
|
+
let rowXml = rows[r].xml;
|
|
11121
|
+
const cellsInRow = own.filter(o => o.row === r).sort((a, b) => b.cell.startIndex - a.cell.startIndex);
|
|
11122
|
+
for (const o of cellsInRow) {
|
|
11123
|
+
const w = scaled.slice(o.col, o.col + o.span).reduce((a, b) => a + b, 0);
|
|
11124
|
+
const newCell = o.cell.xml.slice(0, o.at) + String(w) + o.cell.xml.slice(o.at + String(o.width).length);
|
|
11125
|
+
rowXml = rowXml.slice(0, o.cell.startIndex) + newCell + rowXml.slice(o.cell.endIndex);
|
|
11126
|
+
}
|
|
11127
|
+
out = out.slice(0, rows[r].startIndex) + rowXml + out.slice(rows[r].endIndex);
|
|
11128
|
+
}
|
|
11129
|
+
return out;
|
|
11130
|
+
}
|
|
11131
|
+
/**
|
|
11132
|
+
* Locate one of a cell's OWN address/span attributes (`colAddr`, `rowAddr`,
|
|
11133
|
+
* `colSpan`, `rowSpan`) in `cellXml`, returning the value and the absolute
|
|
11134
|
+
* index of its digits so callers can rewrite it in place.
|
|
11135
|
+
*
|
|
11136
|
+
* Hancom writes them on `<hp:cellAddr>`/`<hp:cellSpan>` after the cell's
|
|
11137
|
+
* sub-list (209/209 corpus files). Hand-made files may put them on the
|
|
11138
|
+
* `<hp:tc>` start tag instead, which the parser also accepts. A nested
|
|
11139
|
+
* table's cells live inside the sub-list, so only the tail is searched for
|
|
11140
|
+
* the child form and only the start tag for the attribute form.
|
|
11141
|
+
*/
|
|
11142
|
+
cellOwnAttr(cellXml, name) {
|
|
11143
|
+
const child = name.endsWith('Addr') ? 'cellAddr' : 'cellSpan';
|
|
11144
|
+
const tail = cellXml.lastIndexOf('</hp:subList>');
|
|
11145
|
+
const from = tail === -1 ? 0 : tail;
|
|
11146
|
+
const own = new RegExp(`(<hp:${child}\\b[^>]*\\b${name}=")(\\d+)"`).exec(cellXml.slice(from));
|
|
11147
|
+
if (own) {
|
|
11148
|
+
return { value: parseInt(own[2], 10), at: from + own.index + own[1].length, length: own[2].length };
|
|
11149
|
+
}
|
|
11150
|
+
const startTag = cellXml.slice(0, cellXml.indexOf('>') + 1);
|
|
11151
|
+
const attr = new RegExp(`(\\s${name}=")(\\d+)"`).exec(startTag);
|
|
11152
|
+
if (attr) {
|
|
11153
|
+
return { value: parseInt(attr[2], 10), at: attr.index + attr[1].length, length: attr[2].length };
|
|
11154
|
+
}
|
|
11155
|
+
return null;
|
|
11156
|
+
}
|
|
11157
|
+
/** Rewrite one of a cell's own attributes (see cellOwnAttr); no-op if absent. */
|
|
11158
|
+
setCellOwnAttr(cellXml, name, value) {
|
|
11159
|
+
const a = this.cellOwnAttr(cellXml, name);
|
|
11160
|
+
return a ? cellXml.slice(0, a.at) + String(value) + cellXml.slice(a.at + a.length) : cellXml;
|
|
11161
|
+
}
|
|
11162
|
+
/** A cell's own <hp:cellSz width> (after its sub-list, so never a nested table's). */
|
|
11163
|
+
cellOwnWidth(cellXml) {
|
|
11164
|
+
const tail = cellXml.lastIndexOf('</hp:subList>');
|
|
11165
|
+
const m = cellXml.slice(tail === -1 ? 0 : tail).match(/<hp:cellSz\b[^>]*\bwidth="(\d+)"/);
|
|
11166
|
+
return m ? parseInt(m[1], 10) : null;
|
|
11167
|
+
}
|
|
11168
|
+
/** Rewrite a cell's own <hp:cellSz width>; no-op if the cell has none or width <= 0. */
|
|
11169
|
+
setCellOwnWidth(cellXml, width) {
|
|
11170
|
+
if (width <= 0)
|
|
11171
|
+
return cellXml;
|
|
11172
|
+
const tail = cellXml.lastIndexOf('</hp:subList>');
|
|
11173
|
+
const from = tail === -1 ? 0 : tail;
|
|
11174
|
+
const m = /(<hp:cellSz\b[^>]*\bwidth=")(\d+)"/.exec(cellXml.slice(from));
|
|
11175
|
+
if (!m)
|
|
11176
|
+
return cellXml;
|
|
11177
|
+
const at = from + m.index + m[1].length;
|
|
11178
|
+
return cellXml.slice(0, at) + String(width) + cellXml.slice(at + m[2].length);
|
|
11179
|
+
}
|
|
11180
|
+
/**
|
|
11181
|
+
* Add `delta` to the rowAddr of every cell of THIS table whose rowAddr is
|
|
11182
|
+
* >= fromRow. Nested tables inside cells keep their own addresses.
|
|
11183
|
+
*/
|
|
11184
|
+
shiftTableRowAddrs(tableXml, fromRow, delta) {
|
|
11185
|
+
let out = tableXml;
|
|
11186
|
+
const rows = this.findAllElementsWithDepth(out, 'tr');
|
|
11187
|
+
for (let r = rows.length - 1; r >= 0; r--) {
|
|
11188
|
+
const row = rows[r];
|
|
11189
|
+
const cells = this.findAllElementsWithDepth(row.xml, 'tc');
|
|
11190
|
+
let rowXml = row.xml;
|
|
11191
|
+
for (let c = cells.length - 1; c >= 0; c--) {
|
|
11192
|
+
const cell = cells[c];
|
|
11193
|
+
const addr = this.cellOwnAttr(cell.xml, 'rowAddr');
|
|
11194
|
+
if (!addr || addr.value < fromRow)
|
|
11195
|
+
continue;
|
|
11196
|
+
const newCell = this.setCellOwnAttr(cell.xml, 'rowAddr', addr.value + delta);
|
|
11197
|
+
rowXml = rowXml.slice(0, cell.startIndex) + newCell + rowXml.slice(cell.endIndex);
|
|
11198
|
+
}
|
|
11199
|
+
if (rowXml !== row.xml)
|
|
11200
|
+
out = out.slice(0, row.startIndex) + rowXml + out.slice(row.endIndex);
|
|
11201
|
+
}
|
|
11202
|
+
return out;
|
|
11203
|
+
}
|
|
11204
|
+
/**
|
|
11205
|
+
* Clone a table cell for a newly inserted row: same cell attributes, same
|
|
11206
|
+
* first-paragraph formatting, but a single paragraph holding `text`.
|
|
11207
|
+
*
|
|
11208
|
+
* Nested tables and extra paragraphs are dropped. The first run's
|
|
11209
|
+
* charPrIDRef is kept so the new text matches the template cell's font.
|
|
11210
|
+
*/
|
|
11211
|
+
cloneCellWithText(cellXml, text) {
|
|
11212
|
+
const subListOpen = cellXml.match(/<(hp|hs):subList\b[^>]*>/);
|
|
11213
|
+
const subListCloseIdx = cellXml.lastIndexOf('</hp:subList>') !== -1
|
|
11214
|
+
? cellXml.lastIndexOf('</hp:subList>')
|
|
11215
|
+
: cellXml.lastIndexOf('</hs:subList>');
|
|
11216
|
+
if (!subListOpen || subListOpen.index === undefined || subListCloseIdx === -1) {
|
|
11217
|
+
// No sub-list to rebuild — fall back to blanking the text in place.
|
|
11218
|
+
return this.resetLinesegInXml(cellXml.replace(T_TAG_WITH_CONTENT, '<$1:t$2></$1:t>'));
|
|
11219
|
+
}
|
|
11220
|
+
const prefix = subListOpen[1];
|
|
11221
|
+
const inner = cellXml.slice(subListOpen.index + subListOpen[0].length, subListCloseIdx);
|
|
11222
|
+
const firstPara = inner.match(new RegExp(`<${prefix}:p\\b[^>]*>`));
|
|
11223
|
+
const paraOpen = firstPara
|
|
11224
|
+
? firstPara[0]
|
|
11225
|
+
: `<${prefix}:p id="0" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0">`;
|
|
11226
|
+
const firstRun = inner.match(new RegExp(`<${prefix}:run\\b[^>]*charPrIDRef="(\\d+)"`));
|
|
11227
|
+
const charPr = firstRun ? firstRun[1] : '0';
|
|
11228
|
+
const paragraph = `${paraOpen}<${prefix}:run charPrIDRef="${charPr}"><${prefix}:t>${this.escapeXml(text)}</${prefix}:t></${prefix}:run>` +
|
|
11229
|
+
`<${prefix}:linesegarray><${prefix}:lineseg textpos="0" vertpos="0" vertsize="1000" textheight="1000" baseline="850" spacing="600" horzpos="0" horzsize="0" flags="0"/></${prefix}:linesegarray>` +
|
|
11230
|
+
`</${prefix}:p>`;
|
|
11231
|
+
return cellXml.slice(0, subListOpen.index + subListOpen[0].length) + paragraph + cellXml.slice(subListCloseIdx);
|
|
11232
|
+
}
|
|
11233
|
+
/**
|
|
11234
|
+
* Source cells for a new row inserted after `afterRow`, one per column
|
|
11235
|
+
* position, in column order, covering every column 0..colCnt-1 exactly once.
|
|
11236
|
+
*
|
|
11237
|
+
* For each column: the cell that STARTS there in the template row (keeping
|
|
11238
|
+
* its colSpan so horizontal merges carry over), otherwise the nearest row
|
|
11239
|
+
* above whose own cell starts there. A column no row starts is skipped by the
|
|
11240
|
+
* colSpan of the cell covering it. Returned XML still carries the source
|
|
11241
|
+
* addresses; the caller rewrites rowAddr/rowSpan.
|
|
11242
|
+
*/
|
|
11243
|
+
gridCellsForNewRow(rows, afterRow) {
|
|
11244
|
+
const ownProps = (cellXml) => ({
|
|
11245
|
+
col: this.cellOwnAttr(cellXml, 'colAddr')?.value ?? -1,
|
|
11246
|
+
span: this.cellOwnAttr(cellXml, 'colSpan')?.value ?? 1,
|
|
11247
|
+
});
|
|
11248
|
+
// Cells with no address anywhere are placed by position in their row.
|
|
11249
|
+
const rowCells = rows.map(r => {
|
|
11250
|
+
let next = 0;
|
|
11251
|
+
return this.findAllElementsWithDepth(r.xml, 'tc').map(c => {
|
|
11252
|
+
const p = ownProps(c.xml);
|
|
11253
|
+
const col = p.col >= 0 ? p.col : next;
|
|
11254
|
+
next = col + p.span;
|
|
11255
|
+
return { xml: c.xml, col, span: p.span };
|
|
11256
|
+
});
|
|
11257
|
+
});
|
|
11258
|
+
const colCount = Math.max(0, ...rowCells.flat().map(c => c.col + c.span));
|
|
11259
|
+
const out = [];
|
|
11260
|
+
for (let col = 0; col < colCount;) {
|
|
11261
|
+
let pick;
|
|
11262
|
+
for (let r = afterRow; r >= 0 && !pick; r--)
|
|
11263
|
+
pick = rowCells[r].find(c => c.col === col);
|
|
11264
|
+
// Nothing above starts here (should not happen in a well-formed table):
|
|
11265
|
+
// fall back to any row below so the grid still has no hole.
|
|
11266
|
+
for (let r = afterRow + 1; r < rowCells.length && !pick; r++)
|
|
11267
|
+
pick = rowCells[r].find(c => c.col === col);
|
|
11268
|
+
if (!pick) {
|
|
11269
|
+
col++;
|
|
11270
|
+
continue;
|
|
11271
|
+
}
|
|
11272
|
+
// A cell borrowed from a row above may span columns the template row
|
|
11273
|
+
// splits; keep the template row's split by clamping to the next column
|
|
11274
|
+
// that the template row starts.
|
|
11275
|
+
let span = Math.max(1, pick.span);
|
|
11276
|
+
const nextTemplateStart = rowCells[afterRow].map(c => c.col).filter(c => c > col).sort((a, b) => a - b)[0];
|
|
11277
|
+
if (nextTemplateStart !== undefined && col + span > nextTemplateStart)
|
|
11278
|
+
span = nextTemplateStart - col;
|
|
11279
|
+
// Narrow through the cell's OWN attributes: a nested table's cells come
|
|
11280
|
+
// first in the XML, so replacing the first <hp:cellSpan> changed the
|
|
11281
|
+
// nested cell and left this one overlapping the next template cell.
|
|
11282
|
+
// Its width shrinks to the columns it still covers, so the row keeps
|
|
11283
|
+
// the table width.
|
|
11284
|
+
const xml = span === pick.span
|
|
11285
|
+
? pick.xml
|
|
11286
|
+
: this.setCellOwnWidth(this.setCellOwnAttr(pick.xml, 'colSpan', span), Math.round((this.cellOwnWidth(pick.xml) ?? 0) * span / pick.span));
|
|
11287
|
+
out.push(xml);
|
|
11288
|
+
col += span;
|
|
11289
|
+
}
|
|
11290
|
+
return out;
|
|
11291
|
+
}
|
|
10287
11292
|
async applyTableRowInsertsToXml() {
|
|
10288
11293
|
if (!this._zip)
|
|
10289
11294
|
return;
|
|
@@ -10310,29 +11315,40 @@ class HwpxDocument {
|
|
|
10310
11315
|
if (insert.afterRowIndex >= rows.length)
|
|
10311
11316
|
continue;
|
|
10312
11317
|
const templateRow = rows[insert.afterRowIndex];
|
|
10313
|
-
//
|
|
10314
|
-
|
|
10315
|
-
//
|
|
10316
|
-
|
|
10317
|
-
//
|
|
11318
|
+
// Build the new row from the table's COLUMN GRID, not from the template
|
|
11319
|
+
// row's cells. A row just below a vertical merge has no <hp:tc> for the
|
|
11320
|
+
// merged column (the master above covers it), so cloning its cells gave
|
|
11321
|
+
// the new row a hole there: colCnt=3 but only columns 1-2 present
|
|
11322
|
+
// (CodeRabbit, 2026-09-24). For each column position we take the cell
|
|
11323
|
+
// that starts there in the template row, or — if the template row has
|
|
11324
|
+
// none — the nearest row above that does, cloned as a single-row cell.
|
|
11325
|
+
//
|
|
11326
|
+
// Each new cell keeps its source's formatting but only its FIRST
|
|
11327
|
+
// paragraph, emptied: cloning every paragraph copied multi-line cells
|
|
11328
|
+
// (e.g. "○ a\n○ b\n- c") as three empty lines, so Hancom sized the row
|
|
11329
|
+
// for three lines and the one line of new text sat at the top.
|
|
10318
11330
|
const newRowAddr = insert.afterRowIndex + 1;
|
|
10319
|
-
|
|
10320
|
-
|
|
10321
|
-
|
|
10322
|
-
|
|
10323
|
-
|
|
10324
|
-
|
|
10325
|
-
|
|
10326
|
-
|
|
10327
|
-
|
|
10328
|
-
|
|
10329
|
-
|
|
10330
|
-
|
|
10331
|
-
|
|
10332
|
-
|
|
10333
|
-
|
|
10334
|
-
|
|
10335
|
-
|
|
11331
|
+
const newRowCells = this.gridCellsForNewRow(rows, insert.afterRowIndex);
|
|
11332
|
+
const trOpen = templateRow.xml.slice(0, templateRow.xml.indexOf('>') + 1);
|
|
11333
|
+
let newRowXml = trOpen + newRowCells.map((cellXml, i) => {
|
|
11334
|
+
const text = insert.cellTexts?.[i] ?? '';
|
|
11335
|
+
// New cells sit on row afterRowIndex+1 and span one row each.
|
|
11336
|
+
const cell = this.setCellOwnAttr(this.cloneCellWithText(cellXml, text), 'rowAddr', newRowAddr);
|
|
11337
|
+
return this.setCellOwnAttr(cell, 'rowSpan', 1);
|
|
11338
|
+
}).join('') + '</hp:tr>';
|
|
11339
|
+
// Shift every existing cell below the insertion point down one row.
|
|
11340
|
+
// Without this the next row kept rowAddr=afterRowIndex+1 — the same as
|
|
11341
|
+
// the new row — and Hancom 2020 hung opening the file (reported
|
|
11342
|
+
// 2026-09-24; renumbering rowAddr by <hp:tr> order made it open).
|
|
11343
|
+
// Only the table's OWN cells are touched: a nested table in a cell has
|
|
11344
|
+
// its own row addresses. The delete path does the mirror of this.
|
|
11345
|
+
const shiftedTableXml = this.shiftTableRowAddrs(tableXml, newRowAddr, +1);
|
|
11346
|
+
// Insert after the template row (positions unchanged by the shift above:
|
|
11347
|
+
// it rewrites digits in place only after re-finding rows).
|
|
11348
|
+
const rowsAfterShift = this.findAllElementsWithDepth(shiftedTableXml, 'tr');
|
|
11349
|
+
const anchorRow = rowsAfterShift[insert.afterRowIndex];
|
|
11350
|
+
const insertPos = anchorRow.startIndex + anchorRow.xml.length;
|
|
11351
|
+
const newTableXml = shiftedTableXml.substring(0, insertPos) + '\n' + newRowXml + shiftedTableXml.substring(insertPos);
|
|
10336
11352
|
// Update rowCnt attribute
|
|
10337
11353
|
const updatedTableXml = newTableXml.replace(/rowCnt="(\d+)"/, (_m, cnt) => `rowCnt="${parseInt(cnt) + 1}"`);
|
|
10338
11354
|
xml = xml.substring(0, tables[insert.tableIndex].startIndex) + updatedTableXml + xml.substring(tables[insert.tableIndex].endIndex);
|
|
@@ -10480,9 +11496,11 @@ class HwpxDocument {
|
|
|
10480
11496
|
}
|
|
10481
11497
|
if (!templateCell)
|
|
10482
11498
|
continue;
|
|
10483
|
-
// Clone template and clear text
|
|
11499
|
+
// Clone template and clear text.
|
|
11500
|
+
// The tag-name boundary in T_TAG_WITH_CONTENT keeps <hp:tc> structure intact.
|
|
10484
11501
|
let newCellXml = templateCell.xml;
|
|
10485
|
-
newCellXml = newCellXml.replace(
|
|
11502
|
+
newCellXml = newCellXml.replace(T_TAG_WITH_CONTENT, '<$1:t$2></$1:t>');
|
|
11503
|
+
newCellXml = this.resetLinesegInXml(newCellXml);
|
|
10486
11504
|
// Update colAddr to afterColIndex + 1
|
|
10487
11505
|
newCellXml = newCellXml.replace(/colAddr="(\d+)"/, `colAddr="${insert.afterColIndex + 1}"`);
|
|
10488
11506
|
// Also update <hp:cellAddr colAddr="..."> inside the cell
|
|
@@ -10508,6 +11526,13 @@ class HwpxDocument {
|
|
|
10508
11526
|
}
|
|
10509
11527
|
// Update colCnt
|
|
10510
11528
|
tableXml = tableXml.replace(/colCnt="(\d+)"/, (_m, cnt) => `colCnt="${parseInt(cnt) + 1}"`);
|
|
11529
|
+
// Keep the table inside its original width. The new column cloned the
|
|
11530
|
+
// template column's width, so the columns summed to more than the
|
|
11531
|
+
// table: reported 2026-09-24, 4 × 11765 + 11765 = 58825 > body 51024
|
|
11532
|
+
// while <hp:sz width> still said 47060, and the table ran past the
|
|
11533
|
+
// right margin. Scale every column by the same factor so the total is
|
|
11534
|
+
// exactly the table's width again.
|
|
11535
|
+
tableXml = this.fitColumnsToTableWidth(tableXml);
|
|
10511
11536
|
xml = xml.substring(0, tables[insert.tableIndex].startIndex) + tableXml + xml.substring(tables[insert.tableIndex].endIndex);
|
|
10512
11537
|
}
|
|
10513
11538
|
this._zip.file(sectionPath, xml);
|
|
@@ -10582,12 +11607,43 @@ class HwpxDocument {
|
|
|
10582
11607
|
const prefix = prefixMatch[1];
|
|
10583
11608
|
const tag = prefixMatch[2];
|
|
10584
11609
|
const closeTag = `</${prefix}:${tag}>`;
|
|
10585
|
-
//
|
|
11610
|
+
// Paragraphs DO nest: a paragraph that holds a table contains the
|
|
11611
|
+
// paragraphs of every cell. Taking the first </hp:p> cut a table-wrapper
|
|
11612
|
+
// paragraph off inside its first cell, so anything placed "after" it
|
|
11613
|
+
// landed inside that cell (measured: text inserted after a table
|
|
11614
|
+
// appeared in the table's first cell).
|
|
10586
11615
|
if (tag === 'p') {
|
|
10587
|
-
|
|
10588
|
-
|
|
10589
|
-
|
|
10590
|
-
|
|
11616
|
+
const openTag = `<${prefix}:p`;
|
|
11617
|
+
let depth = 1;
|
|
11618
|
+
let pos = elem.start + elem.tagLength;
|
|
11619
|
+
let endIndex = -1;
|
|
11620
|
+
while (depth > 0 && pos < sectionXml.length) {
|
|
11621
|
+
const nextClose = sectionXml.indexOf(closeTag, pos);
|
|
11622
|
+
if (nextClose === -1)
|
|
11623
|
+
break;
|
|
11624
|
+
// Count only real <hp:p ...> / <hp:p> opens, not <hp:pic>, <hp:pos>, ...
|
|
11625
|
+
let nextOpen = sectionXml.indexOf(openTag, pos);
|
|
11626
|
+
while (nextOpen !== -1 && nextOpen < nextClose) {
|
|
11627
|
+
const after = sectionXml[nextOpen + openTag.length];
|
|
11628
|
+
if (after === ' ' || after === '>' || after === '/')
|
|
11629
|
+
break;
|
|
11630
|
+
nextOpen = sectionXml.indexOf(openTag, nextOpen + 1);
|
|
11631
|
+
}
|
|
11632
|
+
if (nextOpen !== -1 && nextOpen < nextClose) {
|
|
11633
|
+
const tagEnd = sectionXml.indexOf('>', nextOpen);
|
|
11634
|
+
// A self-closing <hp:p/> does not change depth.
|
|
11635
|
+
if (sectionXml[tagEnd - 1] !== '/')
|
|
11636
|
+
depth++;
|
|
11637
|
+
pos = tagEnd + 1;
|
|
11638
|
+
}
|
|
11639
|
+
else {
|
|
11640
|
+
depth--;
|
|
11641
|
+
pos = nextClose + closeTag.length;
|
|
11642
|
+
if (depth === 0)
|
|
11643
|
+
endIndex = pos;
|
|
11644
|
+
}
|
|
11645
|
+
}
|
|
11646
|
+
if (endIndex !== -1) {
|
|
10591
11647
|
results.push({
|
|
10592
11648
|
xml: sectionXml.substring(elem.start, endIndex),
|
|
10593
11649
|
startIndex: elem.start,
|
|
@@ -10628,102 +11684,6 @@ class HwpxDocument {
|
|
|
10628
11684
|
}
|
|
10629
11685
|
return results;
|
|
10630
11686
|
}
|
|
10631
|
-
async applyParagraphCopiesToXml() {
|
|
10632
|
-
if (!this._zip)
|
|
10633
|
-
return;
|
|
10634
|
-
for (const copy of this._pendingParagraphCopies) {
|
|
10635
|
-
const srcPath = `Contents/section${copy.sourceSection}.xml`;
|
|
10636
|
-
const srcXml = await this._zip.file(srcPath)?.async('string');
|
|
10637
|
-
if (!srcXml)
|
|
10638
|
-
continue;
|
|
10639
|
-
const srcElements = this.findTopLevelFullElements(srcXml);
|
|
10640
|
-
if (copy.sourceParagraph >= srcElements.length)
|
|
10641
|
-
continue;
|
|
10642
|
-
const srcElem = srcElements[copy.sourceParagraph];
|
|
10643
|
-
if (srcElem.type !== 'p')
|
|
10644
|
-
continue;
|
|
10645
|
-
// Clone and regenerate ID
|
|
10646
|
-
let clonedXml = srcElem.xml;
|
|
10647
|
-
const newId = Math.random().toString(36).substring(2, 11);
|
|
10648
|
-
clonedXml = clonedXml.replace(/<(hp|hs):p\s+([^>]*?)id="[^"]*"/, `<$1:p $2id="${newId}"`);
|
|
10649
|
-
// Read target section
|
|
10650
|
-
const tgtPath = `Contents/section${copy.targetSection}.xml`;
|
|
10651
|
-
let tgtXml = await this._zip.file(tgtPath)?.async('string');
|
|
10652
|
-
if (!tgtXml)
|
|
10653
|
-
continue;
|
|
10654
|
-
const tgtElements = this.findTopLevelFullElements(tgtXml);
|
|
10655
|
-
// Insert after targetAfter element
|
|
10656
|
-
let insertPos;
|
|
10657
|
-
if (copy.targetAfter >= 0 && copy.targetAfter < tgtElements.length) {
|
|
10658
|
-
insertPos = tgtElements[copy.targetAfter].endIndex;
|
|
10659
|
-
}
|
|
10660
|
-
else if (copy.targetAfter < 0) {
|
|
10661
|
-
// Insert at beginning - find first element
|
|
10662
|
-
if (tgtElements.length > 0) {
|
|
10663
|
-
insertPos = tgtElements[0].startIndex;
|
|
10664
|
-
}
|
|
10665
|
-
else {
|
|
10666
|
-
const secMatch = tgtXml.match(/<(?:hs|hp):sec[^>]*>/);
|
|
10667
|
-
insertPos = secMatch ? secMatch.index + secMatch[0].length : 0;
|
|
10668
|
-
}
|
|
10669
|
-
}
|
|
10670
|
-
else {
|
|
10671
|
-
// After last element
|
|
10672
|
-
insertPos = tgtElements.length > 0 ? tgtElements[tgtElements.length - 1].endIndex : tgtXml.lastIndexOf('</');
|
|
10673
|
-
}
|
|
10674
|
-
tgtXml = tgtXml.substring(0, insertPos) + '\n' + clonedXml + tgtXml.substring(insertPos);
|
|
10675
|
-
this._zip.file(tgtPath, tgtXml);
|
|
10676
|
-
}
|
|
10677
|
-
}
|
|
10678
|
-
async applyParagraphMovesToXml() {
|
|
10679
|
-
if (!this._zip)
|
|
10680
|
-
return;
|
|
10681
|
-
for (const move of this._pendingParagraphMoves) {
|
|
10682
|
-
const srcPath = `Contents/section${move.sourceSection}.xml`;
|
|
10683
|
-
let srcXml = await this._zip.file(srcPath)?.async('string');
|
|
10684
|
-
if (!srcXml)
|
|
10685
|
-
continue;
|
|
10686
|
-
const srcElements = this.findTopLevelFullElements(srcXml);
|
|
10687
|
-
if (move.sourceParagraph >= srcElements.length)
|
|
10688
|
-
continue;
|
|
10689
|
-
const srcElem = srcElements[move.sourceParagraph];
|
|
10690
|
-
if (srcElem.type !== 'p')
|
|
10691
|
-
continue;
|
|
10692
|
-
const extractedXml = srcElem.xml;
|
|
10693
|
-
// Remove from source
|
|
10694
|
-
srcXml = srcXml.substring(0, srcElem.startIndex) + srcXml.substring(srcElem.endIndex);
|
|
10695
|
-
this._zip.file(srcPath, srcXml);
|
|
10696
|
-
// Read target section (re-read if same section since we modified it)
|
|
10697
|
-
const tgtPath = `Contents/section${move.targetSection}.xml`;
|
|
10698
|
-
let tgtXml = await this._zip.file(tgtPath)?.async('string');
|
|
10699
|
-
if (!tgtXml)
|
|
10700
|
-
continue;
|
|
10701
|
-
const tgtElements = this.findTopLevelFullElements(tgtXml);
|
|
10702
|
-
// Adjust target index for same-section moves
|
|
10703
|
-
let adjustedTarget = move.targetAfter;
|
|
10704
|
-
if (move.sourceSection === move.targetSection && move.sourceParagraph < move.targetAfter) {
|
|
10705
|
-
adjustedTarget -= 1;
|
|
10706
|
-
}
|
|
10707
|
-
let insertPos;
|
|
10708
|
-
if (adjustedTarget >= 0 && adjustedTarget < tgtElements.length) {
|
|
10709
|
-
insertPos = tgtElements[adjustedTarget].endIndex;
|
|
10710
|
-
}
|
|
10711
|
-
else if (adjustedTarget < 0) {
|
|
10712
|
-
if (tgtElements.length > 0) {
|
|
10713
|
-
insertPos = tgtElements[0].startIndex;
|
|
10714
|
-
}
|
|
10715
|
-
else {
|
|
10716
|
-
const secMatch = tgtXml.match(/<(?:hs|hp):sec[^>]*>/);
|
|
10717
|
-
insertPos = secMatch ? secMatch.index + secMatch[0].length : 0;
|
|
10718
|
-
}
|
|
10719
|
-
}
|
|
10720
|
-
else {
|
|
10721
|
-
insertPos = tgtElements.length > 0 ? tgtElements[tgtElements.length - 1].endIndex : tgtXml.lastIndexOf('</');
|
|
10722
|
-
}
|
|
10723
|
-
tgtXml = tgtXml.substring(0, insertPos) + '\n' + extractedXml + tgtXml.substring(insertPos);
|
|
10724
|
-
this._zip.file(tgtPath, tgtXml);
|
|
10725
|
-
}
|
|
10726
|
-
}
|
|
10727
11687
|
// ============================================================
|
|
10728
11688
|
// Header/Footer XML Persistence
|
|
10729
11689
|
// ============================================================
|
|
@@ -10838,6 +11798,8 @@ exports.HwpxDocument = HwpxDocument;
|
|
|
10838
11798
|
// Constants for magic numbers
|
|
10839
11799
|
HwpxDocument.NESTED_CHECK_LOOKBACK = 500;
|
|
10840
11800
|
HwpxDocument.SEARCH_SKIP_OFFSET = 10;
|
|
11801
|
+
/** Container elements whose content belongs to OTHER paragraphs or objects. */
|
|
11802
|
+
HwpxDocument.NESTED_CONTENT = /<hp:(tbl|subList|equation|pic|rect|ellipse|polygon|curve|arc|line|container|drawText|textart|ole|footNote|endNote|header|footer)\b/;
|
|
10841
11803
|
/**
|
|
10842
11804
|
* Default chunk size for splitting long text (in characters).
|
|
10843
11805
|
* Texts longer than this will be split into multiple <hp:run> elements.
|