@kimdayoun/hwpx-mcp 0.3.0 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,19 +9,47 @@ const pako_1 = __importDefault(require("pako"));
9
9
  const HwpxParser_1 = require("./HwpxParser");
10
10
  const HangingIndentCalculator_1 = require("./HangingIndentCalculator");
11
11
  const MAX_UNDO_STACK_SIZE = 50;
12
+ /**
13
+ * HWPX 텍스트 노드 `<hp:t>` / `<hs:t>` 전용 매처.
14
+ *
15
+ * `<(hp|hs):t([^>]*)>` 처럼 태그명 뒤 경계를 두지 않으면 `<hp:tc>`·`<hp:tr>`·
16
+ * `<hp:tbl>` 같은 형제 태그의 접두사까지 삼킨다. 그 상태로 본문을 지우면
17
+ * 셀 구조가 통째로 사라지고 닫는 태그만 남아 한/글이 파일을 열지 못한다.
18
+ * 뒤에 오는 문자가 공백·`/`·`>` 중 하나임을 강제해 태그명을 정확히 끊는다.
19
+ */
20
+ const T_TAG_WITH_CONTENT = /<(hp|hs):t((?:\s[^>]*)?)>[\s\S]*?<\/\1:t>/g;
21
+ const T_TAG_EMPTY = /<(hp|hs):t((?:\s[^>]*)?)><\/\1:t>/g;
12
22
  class HwpxDocument {
13
23
  constructor(id, path, zip, content, format) {
14
24
  this._isDirty = false;
25
+ /**
26
+ * True once the section element list has changed shape since the XML was
27
+ * parsed (insert/delete/copy/move of paragraphs, tables or images).
28
+ * Parsed XML offsets are unusable from that point until the next save.
29
+ */
30
+ this._structureChanged = false;
15
31
  this._undoStack = [];
16
32
  this._redoStack = [];
17
33
  this._pendingTextReplacements = [];
18
34
  this._pendingDirectTextUpdates = [];
35
+ /**
36
+ * `col` is the cell's position in the memory row; `colAddr` is its grid column.
37
+ * They differ after a merge: memory keeps covered cells, the XML drops them.
38
+ * The XML writer finds the target by colAddr so a write made after a merge
39
+ * lands in the right cell (writes now replay in call order).
40
+ */
19
41
  this._pendingTableCellUpdates = [];
20
42
  this._pendingNestedTableInserts = [];
21
43
  this._pendingImageInserts = [];
22
44
  this._pendingCellImageInserts = [];
23
45
  this._pendingTableInserts = [];
24
- this._tableInsertCounter = 0; // Counter for insertion order
46
+ /**
47
+ * Monotonic counter shared by paragraph and table inserts. Both kinds are
48
+ * replayed into XML in this order so each insert sees exactly the elements
49
+ * that existed when it was made. Replaying all tables before all paragraphs
50
+ * wrote "A, table, A-2, table" to disk as "A, A-2, table, table".
51
+ */
52
+ this._tableInsertCounter = 0;
25
53
  this._pendingImageDeletes = [];
26
54
  this._pendingTableDeletes = [];
27
55
  this._pendingParagraphDeletes = [];
@@ -36,10 +64,25 @@ class HwpxDocument {
36
64
  this._pendingTableRowDeletes = [];
37
65
  this._pendingTableColumnInserts = [];
38
66
  this._pendingTableColumnDeletes = [];
67
+ /**
68
+ * Call order of every pending edit that names a table cell or row/column by
69
+ * index. Each such index is relative to the table as it was at call time,
70
+ * so save must replay these edits in call order (applyTableOpsInCallOrder).
71
+ * A WeakMap keeps the queue element types unchanged and drops entries with
72
+ * their ops (undo, section delete).
73
+ */
74
+ this._tableOpSeq = new WeakMap();
75
+ this._tableOpCounter = 0;
39
76
  this._pendingParagraphCopies = [];
40
77
  this._pendingParagraphMoves = [];
41
78
  this._pendingHeaderUpdates = [];
42
79
  this._pendingFooterUpdates = [];
80
+ /**
81
+ * New sections to materialise as Contents/sectionN.xml on save, in call
82
+ * order. `templateFrom` is the section whose <hp:secPr> (page size, margins)
83
+ * the new section copies — Hancom's own "insert section" does the same.
84
+ */
85
+ this._pendingSectionOps = [];
43
86
  // Cache for character properties (id → font size in pt)
44
87
  this._charPrCache = null;
45
88
  // Private: Pending table move/copy operations
@@ -127,7 +170,10 @@ class HwpxDocument {
127
170
  elements: [{
128
171
  type: 'paragraph',
129
172
  data: {
130
- id: Math.random().toString(36).substring(2, 11),
173
+ // Must match the id written into Contents/section0.xml below. Later
174
+ // inserts anchor on this id; a random value here pointed at a node
175
+ // that does not exist in the XML.
176
+ id: '0',
131
177
  runs: [{ text: '' }],
132
178
  },
133
179
  }],
@@ -227,14 +273,29 @@ class HwpxDocument {
227
273
  zip.file('Contents/section0.xml', `<?xml version="1.0" encoding="UTF-8" standalone="yes" ?><hs:sec xmlns:ha="http://www.hancom.co.kr/hwpml/2011/app" xmlns:hp="http://www.hancom.co.kr/hwpml/2011/paragraph" xmlns:hp10="http://www.hancom.co.kr/hwpml/2016/paragraph" xmlns:hs="http://www.hancom.co.kr/hwpml/2011/section" xmlns:hc="http://www.hancom.co.kr/hwpml/2011/core" xmlns:hh="http://www.hancom.co.kr/hwpml/2011/head" xmlns:hhs="http://www.hancom.co.kr/hwpml/2011/history" xmlns:hm="http://www.hancom.co.kr/hwpml/2011/master-page" xmlns:hpf="http://www.hancom.co.kr/schema/2011/hpf" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:opf="http://www.idpf.org/2007/opf/" xmlns:ooxmlchart="http://www.hancom.co.kr/hwpml/2016/ooxmlchart" xmlns:hwpunitchar="http://www.hancom.co.kr/hwpml/2016/HwpUnitChar" xmlns:epub="http://www.idpf.org/2007/ops" xmlns:config="urn:oasis:names:tc:opendocument:xmlns:config:1.0"><hp:p id="0" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0"><hp:secPr id="" textDirection="HORIZONTAL" spaceColumns="1134" tabStop="8000" tabStopVal="4000" tabStopUnit="HWPUNIT" outlineShapeIDRef="1" memoShapeIDRef="0" textVerticalWidthHead="0" masterPageCnt="0"><hp:grid lineGrid="0" charGrid="0" wongoji="0"/><hp:startNum pageStartsOn="BOTH" page="0" pic="0" tbl="0" equation="0"/><hp:visibility hideFirstHeader="0" hideFirstFooter="0" hideFirstMasterPage="0" border="SHOW_ALL" fill="SHOW_ALL" hideFirstPageNum="0" hideFirstEmptyLine="0" showLineNumber="0"/><hp:pagePr landscape="0" width="59528" height="84188" gutterType="LEFT_ONLY"><hp:pageMar header="4252" footer="4252" left="8504" right="8504" top="5668" bottom="4252" gutter="0"/></hp:pagePr><hp:footNotePr><hp:autoNumFormat type="DIGIT"/><hp:noteLine length="-1" type="SOLID" width="0.12mm" color="#000000"/><hp:noteSpacing aboveLine="850" belowLine="567" betweenNotes="283"/><hp:numbering type="CONTINUOUS" newNum="1"/><hp:placement place="EACH_COLUMN" beneathText="0"/></hp:footNotePr><hp:endNotePr><hp:autoNumFormat type="DIGIT"/><hp:noteLine length="14692" type="SOLID" width="0.12mm" color="#000000"/><hp:noteSpacing aboveLine="850" belowLine="567" betweenNotes="0"/><hp:numbering type="CONTINUOUS" newNum="1"/><hp:placement place="END_OF_DOCUMENT" beneathText="0"/></hp:endNotePr></hp:secPr><hp:t></hp:t></hp:run></hp:p></hs:sec>`);
228
274
  // Create empty BinData folder
229
275
  zip.folder('BinData');
230
- return new HwpxDocument(id, 'new-document.hwpx', zip, content, 'hwpx');
276
+ // A new document has no location on disk yet. Seeding a bare filename here
277
+ // made save_document resolve it against the server process cwd, so callers
278
+ // could not find the file they had just written.
279
+ return new HwpxDocument(id, '', zip, content, 'hwpx');
231
280
  }
232
281
  get id() { return this._id; }
233
282
  get path() { return this._path; }
283
+ /** True once the document has a real location on disk. */
284
+ get hasPath() { return this._path.length > 0; }
285
+ /**
286
+ * Record where the document now lives after a successful write, so the next
287
+ * save without an explicit path targets the same file.
288
+ */
289
+ setPath(newPath) { this._path = newPath; }
234
290
  get format() { return this._format; }
235
291
  get isDirty() { return this._isDirty; }
236
292
  get zip() { return this._zip; }
237
293
  get content() { return this._content; }
294
+ /** Push a table-structure or table-cell edit and remember its call order. */
295
+ queueTableOp(queue, op) {
296
+ this._tableOpSeq.set(op, ++this._tableOpCounter);
297
+ queue.push(op);
298
+ }
238
299
  // ============================================================
239
300
  // Undo/Redo
240
301
  // ============================================================
@@ -256,6 +317,8 @@ class HwpxDocument {
256
317
  const parsed = JSON.parse(state);
257
318
  this._content.sections = parsed.sections;
258
319
  this._content.metadata = parsed.metadata;
320
+ // Undo/redo swaps in a whole element list; parse-time offsets no longer apply.
321
+ this.markStructureChanged();
259
322
  }
260
323
  canUndo() { return this._undoStack.length > 0; }
261
324
  canRedo() { return this._redoStack.length > 0; }
@@ -311,6 +374,7 @@ class HwpxDocument {
311
374
  this._pendingParagraphMoves = [];
312
375
  this._pendingHeaderUpdates = [];
313
376
  this._pendingFooterUpdates = [];
377
+ this._pendingSectionOps = [];
314
378
  if (this._pendingTableMoves)
315
379
  this._pendingTableMoves = [];
316
380
  }
@@ -322,6 +386,60 @@ class HwpxDocument {
322
386
  this._isDirty = true;
323
387
  this.invalidateReadingCache();
324
388
  }
389
+ /**
390
+ * Record that the section element list changed shape. Call from every method
391
+ * that inserts, removes, copies or moves a section-level element. Once set,
392
+ * paragraph edits stop trusting offsets cached at parse time and locate their
393
+ * target in the current XML instead.
394
+ */
395
+ markStructureChanged() {
396
+ this._structureChanged = true;
397
+ }
398
+ /**
399
+ * Resolve "after element N" into an id-based anchor using the memory model
400
+ * as it is right now (before the new element is spliced in).
401
+ *
402
+ * Returns null for "before everything" (N < 0). Elements without an XML
403
+ * paragraph/table of their own (images, shapes) are skipped backwards to the
404
+ * nearest paragraph or table, which is what the XML placement needs.
405
+ *
406
+ * The parser turns a paragraph that is only a line of ─/━/═ into an 'hr'
407
+ * element with a fresh id. That paragraph is still in the XML, so it still
408
+ * takes an occurrence slot there: skipping it here put later anchors one
409
+ * paragraph early (measured on Hancom files with divider lines).
410
+ */
411
+ resolveElementAnchor(sectionIndex, afterElementIndex) {
412
+ const elements = this._content.sections[sectionIndex]?.elements ?? [];
413
+ for (let i = Math.min(afterElementIndex, elements.length - 1); i >= 0; i--) {
414
+ const key = this.anchorKeyOf(elements[i]);
415
+ if (!key)
416
+ continue;
417
+ let occurrence = 0;
418
+ for (let j = 0; j < i; j++) {
419
+ const other = this.anchorKeyOf(elements[j]);
420
+ if (other && other.kind === key.kind && other.id === key.id)
421
+ occurrence++;
422
+ }
423
+ return { ...key, occurrence };
424
+ }
425
+ return null;
426
+ }
427
+ /**
428
+ * The XML node a memory element stands for, or null if it has none of its
429
+ * own. An 'hr' parsed from a divider paragraph stands for that paragraph.
430
+ */
431
+ anchorKeyOf(el) {
432
+ if (!el)
433
+ return null;
434
+ if (el.type === 'hr') {
435
+ const src = el.data.sourceParagraphId;
436
+ return src ? { kind: 'paragraph', id: String(src) } : null;
437
+ }
438
+ if (el.type !== 'paragraph' && el.type !== 'table')
439
+ return null;
440
+ const id = String(el.data.id ?? '');
441
+ return id ? { kind: el.type, id } : null;
442
+ }
325
443
  // ============================================================
326
444
  // Content Access
327
445
  // ============================================================
@@ -389,6 +507,8 @@ class HwpxDocument {
389
507
  index: ei,
390
508
  text: el.data.runs.map(r => r.text).join(''),
391
509
  style: el.data.paraStyle,
510
+ paraPrIDRef: el.data.paraPrId,
511
+ charPrIDRef: el.data.runs.find(r => r.charPrIDRef !== undefined)?.charPrIDRef,
392
512
  });
393
513
  }
394
514
  });
@@ -403,17 +523,32 @@ class HwpxDocument {
403
523
  text: para.runs.map(r => r.text).join(''),
404
524
  runs: para.runs,
405
525
  style: para.paraStyle,
526
+ // Raw header.xml references, so callers can build XML without scraping it.
527
+ paraPrIDRef: para.paraPrId,
528
+ charPrIDRef: para.runs.find(r => r.charPrIDRef !== undefined)?.charPrIDRef,
406
529
  };
407
530
  }
408
531
  updateParagraphText(sectionIndex, elementIndex, runIndex, text) {
409
- const paragraph = this.findParagraphByPath(sectionIndex, elementIndex);
410
- if (!paragraph)
411
- return;
412
- // Auto-delegate to preserve styles method for multi-run paragraphs
413
- if (paragraph.runs.length > 1) {
414
- this.updateParagraphTextPreserveStyles(sectionIndex, elementIndex, text);
415
- return;
416
- }
532
+ const section = this._content.sections[sectionIndex];
533
+ if (!section)
534
+ throw new Error(`Section ${sectionIndex} does not exist.`);
535
+ const element = section.elements[elementIndex];
536
+ if (!element) {
537
+ throw new Error(`Element ${elementIndex} does not exist in section ${sectionIndex} (${section.elements.length} elements).`);
538
+ }
539
+ if (element.type !== 'paragraph') {
540
+ // Reported 2026-09-24: aimed at a table, this answered "Paragraph updated"
541
+ // and changed nothing. Say what is there instead.
542
+ throw new Error(`Element ${elementIndex} in section ${sectionIndex} is a ${element.type}, not a paragraph. ` +
543
+ (element.type === 'table' ? 'Use update_table_cell to change table text.' : 'It has no paragraph text to replace.'));
544
+ }
545
+ const paragraph = element.data;
546
+ // Replacing run 0 means "replace the whole paragraph": the new text goes
547
+ // into the first run and every other run is emptied, so the result takes
548
+ // the first run's character shape. Spreading the text across the old runs
549
+ // (preserve-styles) instead gave the tail of the sentence whatever shape
550
+ // those runs had — reported 2026-09-24: plain + bold paragraph, replaced
551
+ // wholesale, came out bold from the third line on.
417
552
  // Handle case where paragraph has no runs (e.g., run without hp:t tag)
418
553
  // We need to create a run in memory and track the update for XML modification
419
554
  if (!paragraph.runs[runIndex]) {
@@ -436,6 +571,7 @@ class HwpxDocument {
436
571
  elementIndex,
437
572
  paragraphId: paragraph.id || '', // Use stable paragraph ID for reliable identification
438
573
  paragraphOccurrence,
574
+ paragraph,
439
575
  runIndex,
440
576
  oldText,
441
577
  newText: text
@@ -450,6 +586,7 @@ class HwpxDocument {
450
586
  elementIndex,
451
587
  paragraphId: paragraph.id || '',
452
588
  paragraphOccurrence,
589
+ paragraph,
453
590
  runIndex: i,
454
591
  oldText: otherOldText,
455
592
  newText: '' // Clear other runs
@@ -530,6 +667,7 @@ class HwpxDocument {
530
667
  elementIndex,
531
668
  paragraphId: paragraph.id || '',
532
669
  paragraphOccurrence,
670
+ paragraph,
533
671
  runIndex: i,
534
672
  oldText: oldText || '',
535
673
  newText: run.text
@@ -549,12 +687,17 @@ class HwpxDocument {
549
687
  id: paragraphId,
550
688
  runs: [{ text }],
551
689
  };
690
+ // Resolve the XML position before the new paragraph joins the element list.
691
+ const anchor = this.resolveElementAnchor(sectionIndex, afterElementIndex);
552
692
  const newElement = { type: 'paragraph', data: newParagraph };
553
693
  section.elements.splice(afterElementIndex + 1, 0, newElement);
694
+ this.markStructureChanged();
554
695
  // Add to pending list for XML sync
555
696
  this._pendingParagraphInserts.push({
556
697
  sectionIndex,
557
698
  afterElementIndex,
699
+ anchor,
700
+ insertOrder: this._tableInsertCounter++,
558
701
  paragraphId,
559
702
  text,
560
703
  });
@@ -578,6 +721,7 @@ class HwpxDocument {
578
721
  });
579
722
  // Remove from memory
580
723
  section.elements.splice(elementIndex, 1);
724
+ this.markStructureChanged();
581
725
  this.markModified();
582
726
  this.invalidateReadingCache();
583
727
  return true;
@@ -600,6 +744,7 @@ class HwpxDocument {
600
744
  elementIndex,
601
745
  paragraphId: paragraph.id || '', // Use stable paragraph ID
602
746
  paragraphOccurrence,
747
+ paragraph,
603
748
  runIndex: lastRunIndex,
604
749
  oldText,
605
750
  newText
@@ -618,6 +763,7 @@ class HwpxDocument {
618
763
  elementIndex,
619
764
  paragraphId: paragraph.id || '', // Use stable paragraph ID
620
765
  paragraphOccurrence,
766
+ paragraph,
621
767
  runIndex: 0,
622
768
  oldText: '',
623
769
  newText: text
@@ -986,7 +1132,7 @@ class HwpxDocument {
986
1132
  this._pendingTableCellHangingIndents[existingIdx].indentPt = indentPt;
987
1133
  }
988
1134
  else {
989
- this._pendingTableCellHangingIndents.push({
1135
+ this.queueTableOp(this._pendingTableCellHangingIndents, {
990
1136
  sectionIndex,
991
1137
  tableIndex,
992
1138
  row,
@@ -1065,7 +1211,7 @@ class HwpxDocument {
1065
1211
  this._pendingTableCellHangingIndents[existingIdx].indentPt = 0; // 0 means remove
1066
1212
  }
1067
1213
  else {
1068
- this._pendingTableCellHangingIndents.push({
1214
+ this.queueTableOp(this._pendingTableCellHangingIndents, {
1069
1215
  sectionIndex,
1070
1216
  tableIndex,
1071
1217
  row,
@@ -1181,12 +1327,22 @@ class HwpxDocument {
1181
1327
  /**
1182
1328
  * Get table map with headers - maps table indices to their header paragraphs
1183
1329
  * Returns array of table info including the header text from the preceding paragraph
1330
+ *
1331
+ * Two indices are returned because they differ once a document has more than
1332
+ * one section:
1333
+ * - `table_index_in_section` — what every table tool (update_table_cell,
1334
+ * get_table_cell, insert_table_row, …) expects together with
1335
+ * `section_index`. Use this one.
1336
+ * - `table_index` — position across the whole document, kept for callers
1337
+ * that list tables. Passing it to a table tool in section 1+ addresses a
1338
+ * DIFFERENT table (reported 2026-09-24: map said 5, the tool needed 4).
1184
1339
  */
1185
1340
  getTableMap() {
1186
1341
  const result = [];
1187
1342
  let globalTableIndex = 0;
1188
1343
  this._content.sections.forEach((section, sectionIndex) => {
1189
1344
  let lastParagraphText = '';
1345
+ let sectionTableIndex = 0;
1190
1346
  section.elements.forEach((element, _elementIndex) => {
1191
1347
  if (element.type === 'paragraph') {
1192
1348
  // Store the paragraph text as potential header
@@ -1209,6 +1365,7 @@ class HwpxDocument {
1209
1365
  }) || [];
1210
1366
  result.push({
1211
1367
  table_index: globalTableIndex,
1368
+ table_index_in_section: sectionTableIndex,
1212
1369
  section_index: sectionIndex,
1213
1370
  header: lastParagraphText,
1214
1371
  rows,
@@ -1217,6 +1374,7 @@ class HwpxDocument {
1217
1374
  first_row_preview: firstRowPreview,
1218
1375
  });
1219
1376
  globalTableIndex++;
1377
+ sectionTableIndex++;
1220
1378
  // Don't reset lastParagraphText here - next table might reuse same header if consecutive
1221
1379
  }
1222
1380
  });
@@ -1391,8 +1549,15 @@ class HwpxDocument {
1391
1549
  // Get previous value
1392
1550
  const cellData = this.getTableCell(tableInfo.section_index, tableInfo.local_index, position.row, position.col);
1393
1551
  const previousValue = cellData?.text || '';
1394
- // Update the cell
1395
- const updated = this.updateTableCell(tableInfo.section_index, tableInfo.local_index, position.row, position.col, value);
1552
+ // Update the cell. A cell covered by a merge throws; treat it as a failed
1553
+ // path rather than aborting the remaining entries.
1554
+ let updated = false;
1555
+ try {
1556
+ updated = this.updateTableCell(tableInfo.section_index, tableInfo.local_index, position.row, position.col, value);
1557
+ }
1558
+ catch {
1559
+ updated = false;
1560
+ }
1396
1561
  if (updated) {
1397
1562
  result.success++;
1398
1563
  result.details.push({
@@ -1531,6 +1696,7 @@ class HwpxDocument {
1531
1696
  const result = {
1532
1697
  success: 0,
1533
1698
  outOfBounds: [],
1699
+ failed: [],
1534
1700
  updated: [],
1535
1701
  };
1536
1702
  const tableInfo = this.convertGlobalToLocalTableIndex(tableIndex);
@@ -1555,8 +1721,16 @@ class HwpxDocument {
1555
1721
  // Get previous value
1556
1722
  const cellData = this.getTableCell(tableInfo.section_index, tableInfo.local_index, targetRow, targetCol);
1557
1723
  const previousValue = cellData?.text || '';
1558
- // Update cell
1559
- const updated = this.updateTableCell(tableInfo.section_index, tableInfo.local_index, targetRow, targetCol, value);
1724
+ // Update cell. A cell covered by a merge throws; record it and keep
1725
+ // going so one merged position does not discard the whole batch.
1726
+ let updated = false;
1727
+ let failure = '';
1728
+ try {
1729
+ updated = this.updateTableCell(tableInfo.section_index, tableInfo.local_index, targetRow, targetCol, value);
1730
+ }
1731
+ catch (err) {
1732
+ failure = err instanceof Error ? err.message : String(err);
1733
+ }
1560
1734
  if (updated) {
1561
1735
  result.success++;
1562
1736
  result.updated.push({
@@ -1566,6 +1740,14 @@ class HwpxDocument {
1566
1740
  newValue: value,
1567
1741
  });
1568
1742
  }
1743
+ else {
1744
+ result.failed.push({
1745
+ row: targetRow,
1746
+ col: targetCol,
1747
+ value,
1748
+ error: failure || 'Cell update failed',
1749
+ });
1750
+ }
1569
1751
  }
1570
1752
  }
1571
1753
  return result;
@@ -1990,6 +2172,43 @@ class HwpxDocument {
1990
2172
  cell,
1991
2173
  };
1992
2174
  }
2175
+ /**
2176
+ * Find the merged cell that covers (row, col), if that position is not itself
2177
+ * a master cell. A covered cell has no <hp:tc> of its own in the saved XML,
2178
+ * so writing to it succeeds in memory and then silently vanishes on save.
2179
+ */
2180
+ findCoveringMergedCell(table, row, col) {
2181
+ const rows = table.rows;
2182
+ if (!rows)
2183
+ return null;
2184
+ const target = rows[row]?.cells?.[col];
2185
+ if (target && ((target.colSpan ?? 1) > 1 || (target.rowSpan ?? 1) > 1)) {
2186
+ return null; // the position is a master cell
2187
+ }
2188
+ // Merged cells can only originate at or before (row, col), and a table with
2189
+ // no spans at all — the common case — exits on the first row scan.
2190
+ for (let r = 0; r <= row && r < rows.length; r++) {
2191
+ const cells = rows[r]?.cells;
2192
+ if (!cells)
2193
+ continue;
2194
+ const lastCol = Math.min(col, cells.length - 1);
2195
+ for (let c = 0; c <= lastCol; c++) {
2196
+ const cell = cells[c];
2197
+ if (!cell)
2198
+ continue;
2199
+ const rowSpan = cell.rowSpan ?? 1;
2200
+ const colSpan = cell.colSpan ?? 1;
2201
+ if (rowSpan <= 1 && colSpan <= 1)
2202
+ continue;
2203
+ if (r === row && c === col)
2204
+ continue;
2205
+ if (row < r + rowSpan && col < c + colSpan) {
2206
+ return { row: r, col: c };
2207
+ }
2208
+ }
2209
+ }
2210
+ return null;
2211
+ }
1993
2212
  updateTableCell(sectionIndex, tableIndex, row, col, text, charShapeId) {
1994
2213
  const table = this.findTable(sectionIndex, tableIndex);
1995
2214
  if (!table)
@@ -1997,10 +2216,16 @@ class HwpxDocument {
1997
2216
  const cell = table.rows[row]?.cells[col];
1998
2217
  if (!cell)
1999
2218
  return false;
2219
+ // Refuse instead of reporting success and losing the text at save time.
2220
+ const covering = this.findCoveringMergedCell(table, row, col);
2221
+ if (covering) {
2222
+ throw new Error(`Cell (${row}, ${col}) is covered by the merged cell at ` +
2223
+ `(${covering.row}, ${covering.col}); write to the master cell instead`);
2224
+ }
2000
2225
  // Track cell update for XML sync (works for both empty and non-empty cells)
2001
2226
  // Store table ID for reliable XML matching
2002
2227
  // charShapeId is optional - if provided, it will override the existing charPrIDRef
2003
- this._pendingTableCellUpdates.push({ sectionIndex, tableIndex, tableId: table.id, row, col, text, charShapeId });
2228
+ this.queueTableOp(this._pendingTableCellUpdates, { sectionIndex, tableIndex, tableId: table.id, row, col, colAddr: cell.colAddr, text, charShapeId });
2004
2229
  this.saveState();
2005
2230
  if (cell.paragraphs.length > 0 && cell.paragraphs[0].runs.length > 0) {
2006
2231
  cell.paragraphs[0].runs[0].text = text;
@@ -2027,19 +2252,80 @@ class HwpxDocument {
2027
2252
  const table = this.findTable(sectionIndex, tableIndex);
2028
2253
  if (!table || !table.rows[afterRowIndex])
2029
2254
  return false;
2255
+ // A new row between afterRowIndex and afterRowIndex+1 must not cut through
2256
+ // a vertical merge. Cloning a row that holds a rowSpan>1 master, or one that
2257
+ // sits inside such a span, copied the span into the gap and made the merged
2258
+ // area overlap the new row (reported 2026-09-24: rowSpan=2 header, after_row 0).
2259
+ for (const row of table.rows) {
2260
+ for (const cell of row.cells) {
2261
+ const top = cell.rowAddr ?? table.rows.indexOf(row);
2262
+ const span = cell.rowSpan ?? 1;
2263
+ if (span > 1 && top <= afterRowIndex && afterRowIndex < top + span - 1) {
2264
+ throw new Error(`Cannot insert a row after row ${afterRowIndex}: it would split the merged cell at ` +
2265
+ `(${top}, ${cell.colAddr ?? 0}) that spans rows ${top}-${top + span - 1}. ` +
2266
+ `Insert after row ${top + span - 1} instead, or unmerge first.`);
2267
+ }
2268
+ }
2269
+ }
2030
2270
  this.saveState();
2031
- const templateRow = table.rows[afterRowIndex];
2032
- const colCount = templateRow.cells.length;
2271
+ // Same column grid as the XML path (gridCellsForNewRow): one cell per
2272
+ // column position, taking the colAddr/colSpan of the cell that starts
2273
+ // there in the template row or the nearest row above. Sizing the row by
2274
+ // templateRow.cells.length left out a column covered by a vertical merge.
2275
+ // A cell with no colAddr (e.g. added by insertTableColumn, which does not
2276
+ // renumber) is placed by its position in the row, as gridCellsForNewRow
2277
+ // does for the XML. Dropping it made the new row one cell short.
2278
+ const placed = (r) => {
2279
+ let next = 0;
2280
+ return (table.rows[r]?.cells ?? []).map(c => {
2281
+ const span = c.colSpan ?? 1;
2282
+ const col = c.colAddr ?? next;
2283
+ next = col + span;
2284
+ return { col, span };
2285
+ });
2286
+ };
2287
+ const starts = (r) => new Map(placed(r).map(c => [c.col, c.span]));
2288
+ const colCount = Math.max(0, ...table.rows.map((_, r) => Math.max(0, ...placed(r).map(c => c.col + c.span))));
2289
+ const templateStarts = placed(afterRowIndex).map(c => c.col).sort((a, b) => a - b);
2290
+ const grid = [];
2291
+ for (let col = 0; col < colCount;) {
2292
+ let span;
2293
+ for (let r = afterRowIndex; r >= 0 && span === undefined; r--)
2294
+ span = starts(r).get(col);
2295
+ for (let r = afterRowIndex + 1; r < table.rows.length && span === undefined; r++)
2296
+ span = starts(r).get(col);
2297
+ if (span === undefined) {
2298
+ col++;
2299
+ continue;
2300
+ }
2301
+ const next = templateStarts.find(c => c > col);
2302
+ if (next !== undefined && col + span > next)
2303
+ span = next - col;
2304
+ grid.push({ colAddr: col, colSpan: Math.max(1, span) });
2305
+ col += Math.max(1, span);
2306
+ }
2033
2307
  const newRow = {
2034
- cells: Array.from({ length: colCount }, (_, i) => ({
2308
+ cells: grid.map((g, i) => ({
2309
+ rowAddr: afterRowIndex + 1,
2310
+ colAddr: g.colAddr,
2311
+ rowSpan: 1,
2312
+ colSpan: g.colSpan,
2035
2313
  paragraphs: [{
2036
2314
  id: Math.random().toString(36).substring(2, 11),
2037
2315
  runs: [{ text: cellTexts?.[i] || '' }],
2038
2316
  }],
2039
2317
  })),
2040
2318
  };
2319
+ // Keep memory row addresses in step with the XML renumbering, so a later
2320
+ // merge/split/insert on this table reads the right rows.
2321
+ for (const row of table.rows) {
2322
+ for (const cell of row.cells) {
2323
+ if (cell.rowAddr !== undefined && cell.rowAddr > afterRowIndex)
2324
+ cell.rowAddr += 1;
2325
+ }
2326
+ }
2041
2327
  table.rows.splice(afterRowIndex + 1, 0, newRow);
2042
- this._pendingTableRowInserts.push({
2328
+ this.queueTableOp(this._pendingTableRowInserts, {
2043
2329
  sectionIndex,
2044
2330
  tableIndex,
2045
2331
  afterRowIndex,
@@ -2057,8 +2343,27 @@ class HwpxDocument {
2057
2343
  return this.deleteTable(sectionIndex, tableIndex);
2058
2344
  }
2059
2345
  this.saveState();
2346
+ // Mirror applyTableRowDeletesToXml so later edits read the same addresses the
2347
+ // XML has after replay: a vertical merge from an earlier row that reaches the
2348
+ // deleted row loses one row, and cells below move up one row. Stale rowAddr
2349
+ // made the row-insert guard refuse an insert below a merge and allow one
2350
+ // through it (CodeRabbit, 2026-09-24).
2351
+ for (let r = 0; r < rowIndex; r++) {
2352
+ for (const cell of table.rows[r]?.cells ?? []) {
2353
+ const top = cell.rowAddr ?? r;
2354
+ const span = cell.rowSpan ?? 1;
2355
+ if (span > 1 && top + span > rowIndex)
2356
+ cell.rowSpan = span - 1;
2357
+ }
2358
+ }
2060
2359
  table.rows.splice(rowIndex, 1);
2061
- this._pendingTableRowDeletes.push({
2360
+ for (const row of table.rows) {
2361
+ for (const cell of row.cells) {
2362
+ if (cell.rowAddr !== undefined && cell.rowAddr > rowIndex)
2363
+ cell.rowAddr -= 1;
2364
+ }
2365
+ }
2366
+ this.queueTableOp(this._pendingTableRowDeletes, {
2062
2367
  sectionIndex,
2063
2368
  tableIndex,
2064
2369
  rowIndex,
@@ -2099,6 +2404,7 @@ class HwpxDocument {
2099
2404
  });
2100
2405
  // Remove from memory model
2101
2406
  section.elements.splice(elementIndex, 1);
2407
+ this.markStructureChanged();
2102
2408
  this.markModified();
2103
2409
  return true;
2104
2410
  }
@@ -2107,15 +2413,29 @@ class HwpxDocument {
2107
2413
  if (!table)
2108
2414
  return false;
2109
2415
  this.saveState();
2416
+ // Keep memory addresses in step with the XML path (applyTableColumnInsertsToXml
2417
+ // gives the new cell colAddr afterColIndex+1 and shifts the cells after it).
2418
+ // A new cell with no colAddr in the middle of a row made the row read as
2419
+ // [0, (none), 1]: the grid for a later row insert counted column 1 twice and
2420
+ // the new row came out one cell short (CodeRabbit, 2026-09-24).
2110
2421
  for (const row of table.rows) {
2422
+ for (const cell of row.cells) {
2423
+ if (cell.colAddr !== undefined && cell.colAddr > afterColIndex)
2424
+ cell.colAddr += 1;
2425
+ }
2426
+ const rowAddr = row.cells.find(c => c.rowAddr !== undefined)?.rowAddr;
2111
2427
  row.cells.splice(afterColIndex + 1, 0, {
2428
+ colAddr: afterColIndex + 1,
2429
+ ...(rowAddr !== undefined ? { rowAddr } : {}),
2430
+ colSpan: 1,
2431
+ rowSpan: 1,
2112
2432
  paragraphs: [{
2113
2433
  id: Math.random().toString(36).substring(2, 11),
2114
2434
  runs: [{ text: '' }],
2115
2435
  }],
2116
2436
  });
2117
2437
  }
2118
- this._pendingTableColumnInserts.push({
2438
+ this.queueTableOp(this._pendingTableColumnInserts, {
2119
2439
  sectionIndex,
2120
2440
  tableIndex,
2121
2441
  afterColIndex,
@@ -2128,10 +2448,18 @@ class HwpxDocument {
2128
2448
  if (!table || (table.rows[0]?.cells.length || 0) <= 1)
2129
2449
  return false;
2130
2450
  this.saveState();
2451
+ // Mirror applyTableColumnDeletesToXml: cells after the deleted column move one
2452
+ // column left. A write queued after the delete carries the cell's colAddr and
2453
+ // the XML is matched by it, so a stale address sent the text nowhere
2454
+ // (CodeRabbit, 2026-09-24; 0.3.3 dropped these writes too).
2131
2455
  for (const row of table.rows) {
2132
2456
  row.cells.splice(colIndex, 1);
2457
+ for (const cell of row.cells) {
2458
+ if (cell.colAddr !== undefined && cell.colAddr > colIndex)
2459
+ cell.colAddr -= 1;
2460
+ }
2133
2461
  }
2134
- this._pendingTableColumnDeletes.push({
2462
+ this.queueTableOp(this._pendingTableColumnDeletes, {
2135
2463
  sectionIndex,
2136
2464
  tableIndex,
2137
2465
  colIndex,
@@ -2317,12 +2645,13 @@ class HwpxDocument {
2317
2645
  const cellText = cell.paragraphs.map(p => p.runs.map(r => r.text).join('')).join('\n');
2318
2646
  // Use existing pending table cell update mechanism
2319
2647
  this._pendingTableCellUpdates = this._pendingTableCellUpdates || [];
2320
- this._pendingTableCellUpdates.push({
2648
+ this.queueTableOp(this._pendingTableCellUpdates, {
2321
2649
  sectionIndex,
2322
2650
  tableIndex,
2323
2651
  tableId,
2324
2652
  row,
2325
2653
  col,
2654
+ colAddr: cell.colAddr,
2326
2655
  text: cellText,
2327
2656
  });
2328
2657
  this.markModified();
@@ -2389,15 +2718,24 @@ class HwpxDocument {
2389
2718
  const srcElement = srcSection.elements[sourceParagraph];
2390
2719
  if (!srcElement || srcElement.type !== 'paragraph')
2391
2720
  return false;
2721
+ const source = this.resolveElementAnchor(sourceSection, sourceParagraph);
2722
+ if (!source)
2723
+ return false;
2724
+ const anchor = this.resolveElementAnchor(targetSection, targetAfter);
2392
2725
  this.saveState();
2393
2726
  const copy = JSON.parse(JSON.stringify(srcElement));
2394
- copy.data.id = Math.random().toString(36).substring(2, 11);
2727
+ const paragraphId = Math.random().toString(36).substring(2, 11);
2728
+ copy.data.id = paragraphId;
2729
+ delete copy.data._xmlPosition;
2395
2730
  tgtSection.elements.splice(targetAfter + 1, 0, copy);
2731
+ this.markStructureChanged();
2396
2732
  this._pendingParagraphCopies.push({
2397
2733
  sourceSection,
2398
- sourceParagraph,
2399
2734
  targetSection,
2400
- targetAfter,
2735
+ source,
2736
+ anchor,
2737
+ paragraphId,
2738
+ insertOrder: this._tableInsertCounter++,
2401
2739
  });
2402
2740
  this.markModified();
2403
2741
  return true;
@@ -2410,6 +2748,9 @@ class HwpxDocument {
2410
2748
  const srcElement = srcSection.elements[sourceParagraph];
2411
2749
  if (!srcElement || srcElement.type !== 'paragraph')
2412
2750
  return false;
2751
+ const source = this.resolveElementAnchor(sourceSection, sourceParagraph);
2752
+ if (!source)
2753
+ return false;
2413
2754
  this.saveState();
2414
2755
  srcSection.elements.splice(sourceParagraph, 1);
2415
2756
  // Fix same-section index shift: if source was before target, adjust target down
@@ -2417,12 +2758,17 @@ class HwpxDocument {
2417
2758
  if (sourceSection === targetSection && sourceParagraph < targetAfter) {
2418
2759
  adjustedTargetAfter -= 1;
2419
2760
  }
2761
+ // Resolve the destination now that the paragraph has left its old slot,
2762
+ // matching the XML at replay time (source node removed, then re-inserted).
2763
+ const anchor = this.resolveElementAnchor(targetSection, adjustedTargetAfter);
2420
2764
  tgtSection.elements.splice(adjustedTargetAfter + 1, 0, srcElement);
2765
+ this.markStructureChanged();
2421
2766
  this._pendingParagraphMoves.push({
2422
2767
  sourceSection,
2423
- sourceParagraph,
2424
2768
  targetSection,
2425
- targetAfter,
2769
+ source,
2770
+ anchor,
2771
+ insertOrder: this._tableInsertCounter++,
2426
2772
  });
2427
2773
  this.markModified();
2428
2774
  return true;
@@ -2587,8 +2933,11 @@ class HwpxDocument {
2587
2933
  rows: tableRows,
2588
2934
  width: defaultWidth,
2589
2935
  };
2936
+ // Resolve the XML position before the new table joins the element list.
2937
+ const anchor = this.resolveElementAnchor(sectionIndex, afterElementIndex);
2590
2938
  const newElement = { type: 'table', data: newTable };
2591
2939
  section.elements.splice(afterElementIndex + 1, 0, newElement);
2940
+ this.markStructureChanged();
2592
2941
  // Calculate table index
2593
2942
  let tableIndex = 0;
2594
2943
  for (let i = 0; i <= afterElementIndex + 1; i++) {
@@ -2599,10 +2948,10 @@ class HwpxDocument {
2599
2948
  }
2600
2949
  }
2601
2950
  // Add to pending table inserts for XML generation
2602
- // Store the original afterElementIndex and insertOrder for proper sequencing
2603
2951
  this._pendingTableInserts.push({
2604
2952
  sectionIndex,
2605
2953
  afterElementIndex,
2954
+ anchor,
2606
2955
  rows,
2607
2956
  cols,
2608
2957
  width: defaultWidth,
@@ -2661,7 +3010,7 @@ class HwpxDocument {
2661
3010
  if (!this._pendingNestedTableInserts) {
2662
3011
  this._pendingNestedTableInserts = [];
2663
3012
  }
2664
- this._pendingNestedTableInserts.push({
3013
+ this.queueTableOp(this._pendingNestedTableInserts, {
2665
3014
  sectionIndex,
2666
3015
  parentTableIndex,
2667
3016
  row,
@@ -2728,6 +3077,29 @@ class HwpxDocument {
2728
3077
  console.warn(`[HwpxDocument] mergeCells: Single cell selected, no merge needed`);
2729
3078
  return false;
2730
3079
  }
3080
+ // A row whose every own cell falls inside the merge is saved as an <hp:tr>
3081
+ // with no <hp:tc>. 한/글 2024 gave no PDF for such a file (measured: a
3082
+ // full-width two-row merge and a vertical merge in a one-column table; the
3083
+ // same table merged short of full width converted), and a scan of 275 한/글
3084
+ // originals found no row without a cell. 0.3.3 wrote these files too.
3085
+ // Rows built in memory keep covered cells and rows read from a file do not,
3086
+ // so cells are placed by their own address (position only when it has none)
3087
+ // and a cell counts only if no other merged cell covers it.
3088
+ const placedCells = table.rows.flatMap((row, ri) => row.cells.map((cell, ci) => ({ cell, row: cell.rowAddr ?? ri, col: cell.colAddr ?? ci })));
3089
+ const masters = placedCells.filter(p => (p.cell.rowSpan ?? 1) > 1 || (p.cell.colSpan ?? 1) > 1);
3090
+ const coveredByOther = (p) => masters.some(m => m.cell !== p.cell &&
3091
+ p.row >= m.row && p.row < m.row + (m.cell.rowSpan ?? 1) &&
3092
+ p.col >= m.col && p.col < m.col + (m.cell.colSpan ?? 1));
3093
+ for (let r = startRow + 1; r <= endRow; r++) {
3094
+ const keepsCell = placedCells.some(p => p.row === r &&
3095
+ (p.col + (p.cell.colSpan ?? 1) - 1 < startCol || p.col > endCol) &&
3096
+ !coveredByOther(p));
3097
+ if (!keepsCell) {
3098
+ throw new Error(`Cannot merge (${startRow}, ${startCol})-(${endRow}, ${endCol}): row ${r} would have no ` +
3099
+ `cell of its own, and 한/글 does not open a table row without cells. Merge fewer ` +
3100
+ `columns so row ${r} keeps a cell, or delete row ${r} instead.`);
3101
+ }
3102
+ }
2731
3103
  this.saveState();
2732
3104
  // Calculate span values
2733
3105
  const colSpan = endCol - startCol + 1;
@@ -2739,7 +3111,7 @@ class HwpxDocument {
2739
3111
  masterCell.rowSpan = rowSpan;
2740
3112
  }
2741
3113
  // Add to pending merges for XML application during save
2742
- this._pendingCellMerges.push({
3114
+ this.queueTableOp(this._pendingCellMerges, {
2743
3115
  sectionIndex,
2744
3116
  tableIndex,
2745
3117
  startRow,
@@ -2806,7 +3178,7 @@ class HwpxDocument {
2806
3178
  cell.rowSpan = 1;
2807
3179
  }
2808
3180
  // Add to pending splits for XML application during save
2809
- this._pendingCellSplits.push({
3181
+ this.queueTableOp(this._pendingCellSplits, {
2810
3182
  sectionIndex,
2811
3183
  tableIndex,
2812
3184
  row,
@@ -3115,6 +3487,7 @@ class HwpxDocument {
3115
3487
  // Add image element to section
3116
3488
  const newElement = { type: 'image', data: newImage };
3117
3489
  section.elements.splice(afterElementIndex + 1, 0, newElement);
3490
+ this.markStructureChanged();
3118
3491
  // Add to pending inserts for XML sync
3119
3492
  this._pendingImageInserts.push({
3120
3493
  sectionIndex,
@@ -3211,7 +3584,7 @@ class HwpxDocument {
3211
3584
  // Get original image dimensions from binary data
3212
3585
  const orgDimensions = this.getImageDimensions(imageData.data, imageData.mimeType);
3213
3586
  // Add to pending cell image inserts
3214
- this._pendingCellImageInserts.push({
3587
+ this.queueTableOp(this._pendingCellImageInserts, {
3215
3588
  sectionIndex,
3216
3589
  tableIndex,
3217
3590
  row,
@@ -3295,6 +3668,7 @@ class HwpxDocument {
3295
3668
  const index = section.elements.findIndex(el => el.type === 'image' && el.data.id === imageId);
3296
3669
  if (index !== -1) {
3297
3670
  section.elements.splice(index, 1);
3671
+ this.markStructureChanged();
3298
3672
  break;
3299
3673
  }
3300
3674
  }
@@ -3321,6 +3695,7 @@ class HwpxDocument {
3321
3695
  };
3322
3696
  const newElement = { type: 'line', data: newLine };
3323
3697
  section.elements.push(newElement);
3698
+ this.markStructureChanged();
3324
3699
  this.markModified();
3325
3700
  return { id: lineId };
3326
3701
  }
@@ -3341,6 +3716,7 @@ class HwpxDocument {
3341
3716
  };
3342
3717
  const newElement = { type: 'rect', data: newRect };
3343
3718
  section.elements.push(newElement);
3719
+ this.markStructureChanged();
3344
3720
  this.markModified();
3345
3721
  return { id: rectId };
3346
3722
  }
@@ -3361,6 +3737,7 @@ class HwpxDocument {
3361
3737
  };
3362
3738
  const newElement = { type: 'ellipse', data: newEllipse };
3363
3739
  section.elements.push(newElement);
3740
+ this.markStructureChanged();
3364
3741
  this.markModified();
3365
3742
  return { id: ellipseId };
3366
3743
  }
@@ -3381,6 +3758,7 @@ class HwpxDocument {
3381
3758
  };
3382
3759
  const newElement = { type: 'equation', data: newEquation };
3383
3760
  section.elements.splice(afterElementIndex + 1, 0, newElement);
3761
+ this.markStructureChanged();
3384
3762
  this.markModified();
3385
3763
  return { id: equationId };
3386
3764
  }
@@ -3485,13 +3863,18 @@ class HwpxDocument {
3485
3863
  }));
3486
3864
  }
3487
3865
  insertSection(afterSectionIndex) {
3866
+ if (afterSectionIndex < -1 || afterSectionIndex >= this._content.sections.length) {
3867
+ throw new Error(`Cannot insert a section after ${afterSectionIndex}: document has ${this._content.sections.length} section(s).`);
3868
+ }
3488
3869
  this.saveState();
3870
+ // The first paragraph of every section carries <hp:secPr>, so it must have
3871
+ // an XML id the anchors can find. '0' matches the section template below.
3489
3872
  const newSection = {
3490
3873
  id: Math.random().toString(36).substring(2, 11),
3491
3874
  elements: [{
3492
3875
  type: 'paragraph',
3493
3876
  data: {
3494
- id: Math.random().toString(36).substring(2, 11),
3877
+ id: '0',
3495
3878
  runs: [{ text: '' }],
3496
3879
  },
3497
3880
  }],
@@ -3506,9 +3889,44 @@ class HwpxDocument {
3506
3889
  };
3507
3890
  const insertIndex = afterSectionIndex + 1;
3508
3891
  this._content.sections.splice(insertIndex, 0, newSection);
3892
+ this.markStructureChanged();
3893
+ // insertSection used to change only the memory model: save wrote no
3894
+ // sectionN.xml, so a two-section document silently came back with one
3895
+ // section and everything added to the new section was lost (measured on
3896
+ // 0.3.3 with insert_section + insert_table, 2026-09-24).
3897
+ this._pendingSectionOps.push({ op: 'insert', at: insertIndex, templateFrom: Math.max(0, afterSectionIndex) });
3898
+ // Section files are created at the start of save, before every other
3899
+ // pending edit is replayed. Edits recorded earlier still name sections by
3900
+ // their old number; shift those at or after the insertion point so they
3901
+ // land in the same section after the renumbering (measured: an edit to the
3902
+ // old section 0, then insert_section(-1), wrote into the new section 0).
3903
+ this.shiftPendingSectionIndices(insertIndex, +1);
3509
3904
  this.markModified();
3510
3905
  return insertIndex;
3511
3906
  }
3907
+ /**
3908
+ * Add `delta` to every section number held by a pending edit that is >= from.
3909
+ * Covers all pending arrays generically: any numeric field whose name is
3910
+ * sectionIndex or ends in "Section"/"SectionIndex" (source/target pairs).
3911
+ */
3912
+ shiftPendingSectionIndices(from, delta) {
3913
+ const isSectionKey = (k) => k === 'sectionIndex' || /Section(Index)?$/.test(k);
3914
+ for (const key of Object.keys(this)) {
3915
+ if (!String(key).startsWith('_pending') || key === '_pendingSectionOps')
3916
+ continue;
3917
+ const list = this[key];
3918
+ if (!Array.isArray(list))
3919
+ continue;
3920
+ for (const item of list) {
3921
+ if (!item || typeof item !== 'object')
3922
+ continue;
3923
+ for (const [k, v] of Object.entries(item)) {
3924
+ if (isSectionKey(k) && typeof v === 'number' && v >= from)
3925
+ item[k] = v + delta;
3926
+ }
3927
+ }
3928
+ }
3929
+ }
3512
3930
  deleteSection(sectionIndex) {
3513
3931
  if (sectionIndex < 0 || sectionIndex >= this._content.sections.length)
3514
3932
  return false;
@@ -3516,6 +3934,23 @@ class HwpxDocument {
3516
3934
  return false; // Cannot delete the last section
3517
3935
  this.saveState();
3518
3936
  this._content.sections.splice(sectionIndex, 1);
3937
+ this.markStructureChanged();
3938
+ // Same persistence gap as insertSection had: the memory model lost the
3939
+ // section but save kept its file, so the deleted section came back on
3940
+ // reopen. Pending edits aimed at the deleted section are dropped; later
3941
+ // sections move down one number.
3942
+ for (const key of Object.keys(this)) {
3943
+ if (!String(key).startsWith('_pending') || key === '_pendingSectionOps')
3944
+ continue;
3945
+ const list = this[key];
3946
+ if (!Array.isArray(list))
3947
+ continue;
3948
+ const kept = list.filter(item => !(item && typeof item === 'object' &&
3949
+ Object.entries(item).some(([k, v]) => (k === 'sectionIndex' || /Section(Index)?$/.test(k)) && v === sectionIndex)));
3950
+ this[key] = kept;
3951
+ }
3952
+ this.shiftPendingSectionIndices(sectionIndex + 1, -1);
3953
+ this._pendingSectionOps.push({ op: 'delete', at: sectionIndex, templateFrom: 0 });
3519
3954
  this.markModified();
3520
3955
  return true;
3521
3956
  }
@@ -3643,10 +4078,27 @@ class HwpxDocument {
3643
4078
  async syncContentToZip() {
3644
4079
  if (!this._zip)
3645
4080
  return;
3646
- // Apply table inserts FIRST (other operations depend on tables existing in XML)
3647
- if (this._pendingTableInserts && this._pendingTableInserts.length > 0) {
3648
- await this.applyTableInsertsToXml();
4081
+ // New sections first: every later step addresses Contents/sectionN.xml by
4082
+ // the memory section index, so the files must already exist and be numbered
4083
+ // the same way.
4084
+ if (this._pendingSectionOps.length > 0) {
4085
+ await this.applySectionOpsToZip();
4086
+ this._pendingSectionOps = [];
4087
+ }
4088
+ // Replay paragraph/table inserts and paragraph copies/moves together, in
4089
+ // call order, before any text update. Text updates resolve their target in
4090
+ // the current XML, and other operations locate tables by index, so the
4091
+ // structure must already match the memory model.
4092
+ const hasStructuralEdits = this._pendingTableInserts.length > 0 ||
4093
+ this._pendingParagraphInserts.length > 0 ||
4094
+ this._pendingParagraphCopies.length > 0 ||
4095
+ this._pendingParagraphMoves.length > 0;
4096
+ if (hasStructuralEdits) {
4097
+ await this.applyStructuralInsertsToXml();
3649
4098
  this._pendingTableInserts = [];
4099
+ this._pendingParagraphInserts = [];
4100
+ this._pendingParagraphCopies = [];
4101
+ this._pendingParagraphMoves = [];
3650
4102
  }
3651
4103
  // Apply table deletes
3652
4104
  if (this._pendingTableDeletes && this._pendingTableDeletes.length > 0) {
@@ -3663,36 +4115,13 @@ class HwpxDocument {
3663
4115
  await this.applyTableMovesToXml();
3664
4116
  this._pendingTableMoves = [];
3665
4117
  }
3666
- // Apply paragraph inserts
3667
- if (this._pendingParagraphInserts && this._pendingParagraphInserts.length > 0) {
3668
- await this.applyParagraphInsertsToXml();
3669
- this._pendingParagraphInserts = [];
3670
- }
3671
- // Apply table cell updates (preserves original XML structure)
3672
- if (this._pendingTableCellUpdates && this._pendingTableCellUpdates.length > 0) {
3673
- await this.applyTableCellUpdatesToXml();
3674
- this._pendingTableCellUpdates = [];
3675
- }
3676
- // Apply cell merges
3677
- if (this._pendingCellMerges && this._pendingCellMerges.length > 0) {
3678
- await this.applyCellMergesToXml();
3679
- this._pendingCellMerges = [];
3680
- }
3681
- // Apply cell splits
3682
- if (this._pendingCellSplits && this._pendingCellSplits.length > 0) {
3683
- await this.applyCellSplitsToXml();
3684
- this._pendingCellSplits = [];
3685
- }
3686
- // Apply nested table inserts
3687
- if (this._pendingNestedTableInserts && this._pendingNestedTableInserts.length > 0) {
3688
- await this.applyNestedTableInsertsToXml();
3689
- this._pendingNestedTableInserts = [];
3690
- }
3691
- // Apply cell image inserts
3692
- if (this._pendingCellImageInserts && this._pendingCellImageInserts.length > 0) {
3693
- await this.applyCellImageInsertsToXml();
3694
- this._pendingCellImageInserts = [];
3695
- }
4118
+ // Table edits that address cells or rows/columns by index, replayed in
4119
+ // CALL order. Each index is relative to the table as it was when that edit
4120
+ // was made; applying them by kind (all cell writes, then all row inserts,
4121
+ // then column inserts ...) wrote cell text into the pre-insert layout and
4122
+ // dropped text written to a new row or column (CodeRabbit, 2026-09-24; the
4123
+ // same 5 scenarios failed on 0.3.3).
4124
+ await this.applyTableOpsInCallOrder();
3696
4125
  // Apply direct text updates (from updateParagraphText)
3697
4126
  if (this._pendingDirectTextUpdates && this._pendingDirectTextUpdates.length > 0) {
3698
4127
  await this.applyDirectTextUpdatesToXml();
@@ -3718,11 +4147,6 @@ class HwpxDocument {
3718
4147
  await this.applyHangingIndentsToXml();
3719
4148
  this._pendingHangingIndents = [];
3720
4149
  }
3721
- // Apply table cell hanging indent changes
3722
- if (this._pendingTableCellHangingIndents && this._pendingTableCellHangingIndents.length > 0) {
3723
- await this.applyTableCellHangingIndentsToXml();
3724
- this._pendingTableCellHangingIndents = [];
3725
- }
3726
4150
  // Apply paragraph style changes (alignment, etc.)
3727
4151
  if (this._pendingParagraphStyles && this._pendingParagraphStyles.length > 0) {
3728
4152
  await this.applyParagraphStylesToXml();
@@ -3733,36 +4157,6 @@ class HwpxDocument {
3733
4157
  await this.applyCharacterStylesToXml();
3734
4158
  this._pendingCharacterStyles = [];
3735
4159
  }
3736
- // Apply table row inserts
3737
- if (this._pendingTableRowInserts && this._pendingTableRowInserts.length > 0) {
3738
- await this.applyTableRowInsertsToXml();
3739
- this._pendingTableRowInserts = [];
3740
- }
3741
- // Apply table row deletes
3742
- if (this._pendingTableRowDeletes && this._pendingTableRowDeletes.length > 0) {
3743
- await this.applyTableRowDeletesToXml();
3744
- this._pendingTableRowDeletes = [];
3745
- }
3746
- // Apply table column inserts
3747
- if (this._pendingTableColumnInserts && this._pendingTableColumnInserts.length > 0) {
3748
- await this.applyTableColumnInsertsToXml();
3749
- this._pendingTableColumnInserts = [];
3750
- }
3751
- // Apply table column deletes
3752
- if (this._pendingTableColumnDeletes && this._pendingTableColumnDeletes.length > 0) {
3753
- await this.applyTableColumnDeletesToXml();
3754
- this._pendingTableColumnDeletes = [];
3755
- }
3756
- // Apply paragraph copies
3757
- if (this._pendingParagraphCopies && this._pendingParagraphCopies.length > 0) {
3758
- await this.applyParagraphCopiesToXml();
3759
- this._pendingParagraphCopies = [];
3760
- }
3761
- // Apply paragraph moves
3762
- if (this._pendingParagraphMoves && this._pendingParagraphMoves.length > 0) {
3763
- await this.applyParagraphMovesToXml();
3764
- this._pendingParagraphMoves = [];
3765
- }
3766
4160
  // Apply header/footer updates
3767
4161
  if (this._pendingHeaderUpdates && this._pendingHeaderUpdates.length > 0 ||
3768
4162
  this._pendingFooterUpdates && this._pendingFooterUpdates.length > 0) {
@@ -3817,6 +4211,13 @@ class HwpxDocument {
3817
4211
  * The cached positions are populated during parsing in HwpxParser.parseSection().
3818
4212
  */
3819
4213
  getCachedXmlPosition(sectionIndex, elementIndex) {
4214
+ // Cached offsets point into the section XML as it was when parsed. Once any
4215
+ // element has been inserted, removed, copied or moved, the element index no
4216
+ // longer names the same XML node and every earlier offset may have shifted.
4217
+ // Using the cache then rewrites the wrong paragraph — measured: after
4218
+ // copyParagraph on a reopened document, the edit landed on the original.
4219
+ if (this._structureChanged)
4220
+ return undefined;
3820
4221
  const section = this._content?.sections?.[sectionIndex];
3821
4222
  if (!section)
3822
4223
  return undefined;
@@ -3934,10 +4335,10 @@ class HwpxDocument {
3934
4335
  }
3935
4336
  // Clean up empty runs that may be left behind
3936
4337
  // <hp:run charPrIDRef="0"><hp:t/></hp:run> or <hp:run charPrIDRef="0"></hp:run>
3937
- xml = xml.replace(/<hp:run[^>]*>(\s*<hp:t\s*\/>)?\s*<\/hp:run>/g, '');
4338
+ xml = xml.replace(/<hp:run(?:\s[^>]*)?>(\s*<hp:t\s*\/>)?\s*<\/hp:run>/g, '');
3938
4339
  // Clean up empty paragraphs that only contained the image
3939
4340
  // <hp:p ...><hp:linesegarray>...</hp:linesegarray></hp:p>
3940
- xml = xml.replace(/<hp:p[^>]*>\s*(<hp:linesegarray[^>]*>[\s\S]*?<\/hp:linesegarray>)?\s*<\/hp:p>/g, '');
4341
+ xml = xml.replace(/<hp:p(?:\s[^>]*)?>\s*(<hp:linesegarray[^>]*>[\s\S]*?<\/hp:linesegarray>)?\s*<\/hp:p>/g, '');
3941
4342
  if (modified) {
3942
4343
  this._zip.file(sectionPath, xml);
3943
4344
  }
@@ -4054,143 +4455,345 @@ class HwpxDocument {
4054
4455
  }
4055
4456
  }
4056
4457
  /**
4057
- * Apply table inserts to XML.
4058
- * Inserts new tables into the section XML.
4458
+ * Find the end offset of the section-level element an insert anchors to.
4459
+ *
4460
+ * Paragraphs are matched by their own <hp:p id>. Tables are matched by
4461
+ * <hp:tbl id>, either wrapped in a paragraph (the insert goes after that
4462
+ * paragraph) or placed directly in the section (move_table writes them that
4463
+ * way).
4464
+ *
4465
+ * The returned offset is always the end of a TOP-LEVEL element, because that
4466
+ * is the only place a new section-level element may go. But paragraph
4467
+ * occurrences are counted over exactly the paragraphs the parser puts in the
4468
+ * memory model (see parsedParagraphStarts), so they agree with the occurrence
4469
+ * resolveElementAnchor recorded. The parser also lifts paragraphs out of
4470
+ * headers, text boxes and shapes; Hancom reuses id="0" / id="2147483648"
4471
+ * there too. Counting only top-level paragraphs put 47 of 131 sampled Hancom
4472
+ * files' copies and moves on the wrong paragraph.
4473
+ *
4474
+ * Returns -1 if the anchor is not present in the current XML.
4475
+ */
4476
+ findAnchorEnd(xml, anchor) {
4477
+ const topLevel = this.findTopLevelFullElements(xml);
4478
+ const idOf = (fragment) => fragment.slice(0, fragment.indexOf('>') + 1).match(/\bid="([^"]*)"/)?.[1];
4479
+ if (anchor.kind === 'paragraph') {
4480
+ const hit = this.findParsedParagraph(xml, anchor);
4481
+ if (!hit)
4482
+ return -1;
4483
+ // The anchor may sit inside a header/shape; new content goes after the
4484
+ // top-level element that contains it.
4485
+ const owner = topLevel.find(el => el.startIndex <= hit.start && hit.start < el.endIndex);
4486
+ return owner ? owner.endIndex : -1;
4487
+ }
4488
+ let seen = 0;
4489
+ for (const el of topLevel) {
4490
+ if (el.type === 'tbl') {
4491
+ if (idOf(el.xml) !== anchor.id)
4492
+ continue;
4493
+ }
4494
+ else if (!this.wrapsTopLevelTable(el.xml, anchor.id)) {
4495
+ continue;
4496
+ }
4497
+ if (seen === anchor.occurrence)
4498
+ return el.endIndex;
4499
+ seen++;
4500
+ }
4501
+ return -1;
4502
+ }
4503
+ /**
4504
+ * The exact XML range of the memory paragraph a paragraph anchor names,
4505
+ * found by id + occurrence among parsedParagraphStarts. For a paragraph in a
4506
+ * header or text box this is that paragraph alone, not its container.
4059
4507
  */
4060
- async applyTableInsertsToXml() {
4061
- if (!this._zip)
4062
- return;
4063
- // Group inserts by section
4064
- const insertsBySection = new Map();
4065
- for (const insert of this._pendingTableInserts) {
4066
- const sectionInserts = insertsBySection.get(insert.sectionIndex) || [];
4067
- sectionInserts.push({
4068
- afterElementIndex: insert.afterElementIndex,
4069
- rows: insert.rows,
4070
- cols: insert.cols,
4071
- width: insert.width,
4072
- cellWidth: insert.cellWidth,
4073
- insertOrder: insert.insertOrder,
4074
- tableId: insert.tableId,
4075
- });
4076
- insertsBySection.set(insert.sectionIndex, sectionInserts);
4508
+ findParsedParagraph(xml, anchor) {
4509
+ let seen = 0;
4510
+ for (const start of this.parsedParagraphStarts(xml)) {
4511
+ const openEnd = xml.indexOf('>', start) + 1;
4512
+ const id = xml.slice(start, openEnd).match(/\bid="([^"]*)"/)?.[1];
4513
+ if (id !== anchor.id)
4514
+ continue;
4515
+ if (seen === anchor.occurrence) {
4516
+ const end = this.findBalancedParagraphEnd(xml, start);
4517
+ return end === -1 ? null : { start, end };
4518
+ }
4519
+ seen++;
4077
4520
  }
4078
- // Process each section
4079
- for (const [sectionIndex, inserts] of insertsBySection) {
4080
- const sectionPath = `Contents/section${sectionIndex}.xml`;
4081
- const file = this._zip.file(sectionPath);
4082
- if (!file)
4521
+ return null;
4522
+ }
4523
+ /**
4524
+ * Start offsets (in `xml`) of the paragraphs HwpxParser turns into memory
4525
+ * paragraphs, in document order. Mirrors HwpxParser.parseSection:
4526
+ *
4527
+ * - MEMO fields, footnotes and endnotes are ignored;
4528
+ * - paragraphs inside any table are skipped;
4529
+ * - a paragraph that holds a table is kept only if it still has <hp:t>
4530
+ * once its tables are removed.
4531
+ *
4532
+ * Offsets are mapped back to the original XML, so callers can slice it.
4533
+ */
4534
+ parsedParagraphStarts(xml) {
4535
+ // Ranges the parser strips before it looks for paragraphs.
4536
+ const hidden = [];
4537
+ const hide = (re) => {
4538
+ for (const m of xml.matchAll(re))
4539
+ hidden.push([m.index, m.index + m[0].length]);
4540
+ };
4541
+ hide(/<hp:fieldBegin[^>]*type="MEMO"[^>]*>[\s\S]*?<\/hp:fieldBegin>/gi);
4542
+ hide(/<hp:footNote\b[^>]*>[\s\S]*?<\/hp:footNote>/gi);
4543
+ hide(/<hp:endNote\b[^>]*>[\s\S]*?<\/hp:endNote>/gi);
4544
+ const isHidden = (pos) => hidden.some(([a, b]) => pos >= a && pos < b);
4545
+ const tables = this.findAllTablesDeep(xml).filter(t => !isHidden(t.startIndex));
4546
+ const inTable = (pos) => tables.some(t => pos > t.startIndex && pos < t.endIndex);
4547
+ const starts = [];
4548
+ for (const m of xml.matchAll(/<hp:p\b(?=[\s>])[^>]*>/g)) {
4549
+ const start = m.index;
4550
+ if (isHidden(start) || inTable(start))
4083
4551
  continue;
4084
- let xml = await file.async('string');
4085
- // Get maximum id and instid for generating new ones
4086
- const idMatches = xml.matchAll(/id="(\d+)"/g);
4087
- let maxId = 0;
4088
- for (const m of idMatches) {
4089
- maxId = Math.max(maxId, parseInt(m[1], 10));
4090
- }
4091
- // Sort inserts by insertOrder (ascending) - process in the order they were added
4092
- // This ensures tables are inserted sequentially, building on each other
4093
- const sortedInserts = [...inserts].sort((a, b) => a.insertOrder - b.insertOrder);
4094
- for (const insert of sortedInserts) {
4095
- // Use the in-memory table ID for consistency with updateTableCell operations
4096
- const tableId = insert.tableId;
4097
- // Calculate row height based on standard settings
4098
- const rowHeight = 1000; // Default row height in hwpunit
4099
- const tableHeight = rowHeight * insert.rows;
4100
- // Build table XML
4101
- let tableXml = `<hp:tbl id="${tableId}" zOrder="0" numberingType="TABLE" textWrap="TOP_AND_BOTTOM" textFlow="BOTH_SIDES" lock="0" dropcapstyle="None" pageBreak="CELL" repeatHeader="0" rowCnt="${insert.rows}" colCnt="${insert.cols}" cellSpacing="0" borderFillIDRef="2" noAdjust="0">`;
4102
- tableXml += `<hp:sz width="${insert.width}" widthRelTo="ABSOLUTE" height="${tableHeight}" heightRelTo="ABSOLUTE" protect="0"/>`;
4103
- tableXml += `<hp:pos treatAsChar="1" affectLSpacing="0" flowWithText="1" allowOverlap="0" holdAnchorAndSO="0" vertRelTo="PARA" horzRelTo="PARA" vertAlign="TOP" horzAlign="LEFT" vertOffset="0" horzOffset="0"/>`;
4104
- tableXml += `<hp:outMargin left="141" right="141" top="141" bottom="141"/>`;
4105
- tableXml += `<hp:inMargin left="0" right="0" top="0" bottom="0"/>`;
4106
- // Generate rows
4107
- for (let r = 0; r < insert.rows; r++) {
4108
- tableXml += `<hp:tr>`;
4109
- for (let c = 0; c < insert.cols; c++) {
4110
- maxId++;
4111
- const cellParaId = maxId;
4112
- tableXml += `<hp:tc name="" header="0" hasMargin="0" protect="0" editable="0" dirty="0" borderFillIDRef="2">`;
4113
- tableXml += `<hp:subList id="" textDirection="HORIZONTAL" lineWrap="BREAK" vertAlign="CENTER" linkListIDRef="0" linkListNextIDRef="0" textWidth="0" textHeight="0" hasTextRef="0" hasNumRef="0">`;
4114
- tableXml += `<hp:p id="${cellParaId}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0">`;
4115
- tableXml += `<hp:run charPrIDRef="0"><hp:t></hp:t></hp:run>`;
4116
- tableXml += `</hp:p>`;
4117
- tableXml += `</hp:subList>`;
4118
- tableXml += `<hp:cellAddr colAddr="${c}" rowAddr="${r}"/>`;
4119
- tableXml += `<hp:cellSpan colSpan="1" rowSpan="1"/>`;
4120
- tableXml += `<hp:cellSz width="${insert.cellWidth}" height="${rowHeight}"/>`;
4121
- tableXml += `<hp:cellMargin left="141" right="141" top="141" bottom="141"/>`;
4122
- tableXml += `</hp:tc>`;
4123
- }
4124
- tableXml += `</hp:tr>`;
4125
- }
4126
- tableXml += `</hp:tbl>`;
4127
- // Find the position to insert the table
4128
- // We need to insert after a paragraph element
4129
- // Find all <hp:p> elements at the root level (not inside tables)
4130
- const paragraphMatches = [...xml.matchAll(/<hp:p\s[^>]*>.*?<\/hp:p>/gs)];
4131
- // Filter to find only top-level paragraphs (not inside <hp:tbl> or <hp:subList>)
4132
- // For simplicity, insert after the first paragraph if afterElementIndex is 0
4133
- // or find the appropriate position
4134
- let insertPosition = -1;
4135
- let elementCount = -1;
4136
- let searchPos = 0;
4137
- // Find paragraphs and tables at root level using balanced bracket matching
4138
- while (searchPos < xml.length) {
4139
- // Look for next <hp:p or <hp:tbl
4140
- const nextP = xml.indexOf('<hp:p ', searchPos);
4141
- const nextTbl = xml.indexOf('<hp:tbl ', searchPos);
4142
- let nextPos = -1;
4143
- let isTable = false;
4144
- if (nextP !== -1 && (nextTbl === -1 || nextP < nextTbl)) {
4145
- nextPos = nextP;
4146
- isTable = false;
4147
- }
4148
- else if (nextTbl !== -1) {
4149
- nextPos = nextTbl;
4150
- isTable = true;
4151
- }
4152
- if (nextPos === -1)
4153
- break;
4154
- // Check if this is inside a subList (nested)
4155
- const beforeText = xml.substring(Math.max(0, nextPos - HwpxDocument.NESTED_CHECK_LOOKBACK), nextPos);
4156
- const subListOpen = beforeText.lastIndexOf('<hp:subList');
4157
- const subListClose = beforeText.lastIndexOf('</hp:subList>');
4158
- const isNested = subListOpen > subListClose;
4159
- if (!isNested) {
4160
- elementCount++;
4161
- // Find the end of this element using balanced bracket matching
4162
- const endPos = isTable
4163
- ? HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:tbl', '</hp:tbl>')
4164
- : HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:p ', '</hp:p>');
4165
- if (endPos === -1) {
4166
- searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
4167
- continue;
4168
- }
4169
- if (elementCount === insert.afterElementIndex) {
4170
- insertPosition = endPos;
4171
- break;
4172
- }
4173
- searchPos = endPos;
4174
- }
4175
- else {
4176
- searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
4177
- }
4552
+ const end = this.findBalancedParagraphEnd(xml, start);
4553
+ if (end === -1)
4554
+ continue;
4555
+ // Remove only the outermost tables in this paragraph. A nested table
4556
+ // is already inside one of them; cutting it again with its original
4557
+ // offsets would slice the wrong text out of the shortened string.
4558
+ const own = tables.filter(t => t.startIndex >= start && t.endIndex <= end &&
4559
+ !tables.some(o => o !== t && o.startIndex >= start && o.startIndex < t.startIndex && o.endIndex > t.endIndex));
4560
+ if (own.length > 0) {
4561
+ let rest = xml.slice(start, end);
4562
+ for (const t of [...own].sort((a, b) => b.startIndex - a.startIndex)) {
4563
+ rest = rest.slice(0, t.startIndex - start) + rest.slice(t.endIndex - start);
4564
+ }
4565
+ if (!/<hp:t\b[^>]*>/.test(rest))
4566
+ continue;
4567
+ }
4568
+ starts.push(start);
4569
+ }
4570
+ return starts;
4571
+ }
4572
+ /** Every <hp:tbl> range at any depth (outer tables before their nested ones). */
4573
+ findAllTablesDeep(xml) {
4574
+ const out = [];
4575
+ for (const m of xml.matchAll(/<hp:tbl\b/g)) {
4576
+ let depth = 1;
4577
+ let pos = m.index + 7;
4578
+ while (depth > 0 && pos < xml.length) {
4579
+ const nextOpen = xml.indexOf('<hp:tbl', pos);
4580
+ const nextClose = xml.indexOf('</hp:tbl>', pos);
4581
+ if (nextClose === -1)
4582
+ break;
4583
+ if (nextOpen !== -1 && nextOpen < nextClose) {
4584
+ depth++;
4585
+ pos = nextOpen + 7;
4178
4586
  }
4179
- // If position not found, insert at end of section (before </hs:sec>)
4180
- if (insertPosition === -1) {
4181
- const secEnd = xml.lastIndexOf('</hs:sec>');
4182
- if (secEnd !== -1) {
4183
- insertPosition = secEnd;
4184
- }
4587
+ else {
4588
+ depth--;
4589
+ pos = nextClose + 9;
4185
4590
  }
4186
- if (insertPosition !== -1) {
4187
- // Wrap table in a paragraph for proper positioning
4188
- const wrapperXml = `<hp:p id="${maxId + 1}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0">${tableXml}<hp:t></hp:t></hp:run></hp:p>`;
4189
- maxId++;
4190
- xml = xml.substring(0, insertPosition) + wrapperXml + xml.substring(insertPosition);
4591
+ }
4592
+ if (depth === 0)
4593
+ out.push({ startIndex: m.index, endIndex: pos });
4594
+ }
4595
+ return out;
4596
+ }
4597
+ /** End offset of the paragraph opening at `start`, counting nested <hp:p>. */
4598
+ findBalancedParagraphEnd(xml, start) {
4599
+ const openRe = /<hp:p\b(?=[\s>/])[^>]*>/g;
4600
+ let depth = 0;
4601
+ let pos = start;
4602
+ while (pos < xml.length) {
4603
+ openRe.lastIndex = pos;
4604
+ const open = openRe.exec(xml);
4605
+ const close = xml.indexOf('</hp:p>', pos);
4606
+ if (close === -1)
4607
+ return -1;
4608
+ if (open && open.index < close) {
4609
+ if (!open[0].endsWith('/>'))
4610
+ depth++;
4611
+ pos = open.index + open[0].length;
4612
+ }
4613
+ else {
4614
+ depth--;
4615
+ pos = close + 7;
4616
+ if (depth === 0)
4617
+ return pos;
4618
+ }
4619
+ }
4620
+ return -1;
4621
+ }
4622
+ /** True if this paragraph directly (not via a nested table) holds <hp:tbl id>. */
4623
+ wrapsTopLevelTable(paragraphXml, tableId) {
4624
+ const escaped = tableId.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
4625
+ const match = paragraphXml.match(new RegExp(`<(?:hp|hs|hc):tbl\\b[^>]*\\bid="${escaped}"`));
4626
+ if (!match || match.index === undefined)
4627
+ return false;
4628
+ const before = paragraphXml.slice(0, match.index);
4629
+ const opens = (before.match(/<(?:hp|hs|hc):tbl\b/g) || []).length;
4630
+ const closes = (before.match(/<\/(?:hp|hs|hc):tbl>/g) || []).length;
4631
+ return opens === closes;
4632
+ }
4633
+ /**
4634
+ * Offset for an insert with no anchor ("before everything"). The first
4635
+ * paragraph carries <hp:secPr> (page and section settings) and must stay
4636
+ * first, so new content goes right after it.
4637
+ */
4638
+ findSectionHeadEnd(xml) {
4639
+ const topLevel = this.findTopLevelFullElements(xml);
4640
+ const first = topLevel.find(el => el.type === 'p');
4641
+ if (first)
4642
+ return first.endIndex;
4643
+ const secOpen = xml.match(/<(?:hs|hp):sec[^>]*>/);
4644
+ return secOpen && secOpen.index !== undefined ? secOpen.index + secOpen[0].length : -1;
4645
+ }
4646
+ /** Build the XML for a table inserted by insertTable, wrapped in its own paragraph. */
4647
+ buildInsertedTableXml(insert, nextId) {
4648
+ const rowHeight = 1000; // hwpunit
4649
+ const tableHeight = rowHeight * insert.rows;
4650
+ let tableXml = `<hp:tbl id="${insert.tableId}" zOrder="0" numberingType="TABLE" textWrap="TOP_AND_BOTTOM" textFlow="BOTH_SIDES" lock="0" dropcapstyle="None" pageBreak="CELL" repeatHeader="0" rowCnt="${insert.rows}" colCnt="${insert.cols}" cellSpacing="0" borderFillIDRef="2" noAdjust="0">`;
4651
+ tableXml += `<hp:sz width="${insert.width}" widthRelTo="ABSOLUTE" height="${tableHeight}" heightRelTo="ABSOLUTE" protect="0"/>`;
4652
+ tableXml += `<hp:pos treatAsChar="1" affectLSpacing="0" flowWithText="1" allowOverlap="0" holdAnchorAndSO="0" vertRelTo="PARA" horzRelTo="PARA" vertAlign="TOP" horzAlign="LEFT" vertOffset="0" horzOffset="0"/>`;
4653
+ tableXml += `<hp:outMargin left="141" right="141" top="141" bottom="141"/>`;
4654
+ // Cells below use hasMargin="0", which tells Hancom to pad them with this
4655
+ // table-level inMargin and ignore their own cellMargin. A zero inMargin put
4656
+ // text flush against the cell border (measured 0pt). 510/510/141/141 is the
4657
+ // most common value in Hancom-saved tables (1,868 surveyed).
4658
+ tableXml += `<hp:inMargin left="510" right="510" top="141" bottom="141"/>`;
4659
+ for (let r = 0; r < insert.rows; r++) {
4660
+ tableXml += `<hp:tr>`;
4661
+ for (let c = 0; c < insert.cols; c++) {
4662
+ tableXml += `<hp:tc name="" header="0" hasMargin="0" protect="0" editable="0" dirty="0" borderFillIDRef="2">`;
4663
+ tableXml += `<hp:subList id="" textDirection="HORIZONTAL" lineWrap="BREAK" vertAlign="CENTER" linkListIDRef="0" linkListNextIDRef="0" textWidth="0" textHeight="0" hasTextRef="0" hasNumRef="0">`;
4664
+ tableXml += `<hp:p id="${nextId()}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0">`;
4665
+ tableXml += `<hp:run charPrIDRef="0"><hp:t></hp:t></hp:run>`;
4666
+ tableXml += `</hp:p>`;
4667
+ tableXml += `</hp:subList>`;
4668
+ tableXml += `<hp:cellAddr colAddr="${c}" rowAddr="${r}"/>`;
4669
+ tableXml += `<hp:cellSpan colSpan="1" rowSpan="1"/>`;
4670
+ tableXml += `<hp:cellSz width="${insert.cellWidth}" height="${rowHeight}"/>`;
4671
+ tableXml += `<hp:cellMargin left="510" right="510" top="141" bottom="141"/>`;
4672
+ tableXml += `</hp:tc>`;
4673
+ }
4674
+ tableXml += `</hp:tr>`;
4675
+ }
4676
+ tableXml += `</hp:tbl>`;
4677
+ return `<hp:p id="${nextId()}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0">${tableXml}<hp:t></hp:t></hp:run></hp:p>`;
4678
+ }
4679
+ /**
4680
+ * Replay structural edits — paragraph/table inserts and paragraph
4681
+ * copies/moves — into the section XML in the order the calls were made.
4682
+ *
4683
+ * Every edit carries id-based anchors resolved at call time, so it lands
4684
+ * after the same element in the XML that it followed in memory. Replaying in
4685
+ * call order means each anchor already exists (or has already moved) by the
4686
+ * time a later edit needs it. Copies/moves may cross sections, so all
4687
+ * touched sections are held in memory and written once at the end.
4688
+ */
4689
+ async applyStructuralInsertsToXml() {
4690
+ if (!this._zip)
4691
+ return;
4692
+ const edits = [
4693
+ ...this._pendingParagraphInserts.map(i => ({
4694
+ kind: 'paragraph', order: i.insertOrder, sectionIndex: i.sectionIndex,
4695
+ anchor: i.anchor, paragraphId: i.paragraphId, text: i.text,
4696
+ })),
4697
+ ...this._pendingTableInserts.map(i => ({
4698
+ kind: 'table', order: i.insertOrder, sectionIndex: i.sectionIndex,
4699
+ anchor: i.anchor, rows: i.rows, cols: i.cols, width: i.width,
4700
+ cellWidth: i.cellWidth, tableId: i.tableId,
4701
+ })),
4702
+ ...this._pendingParagraphCopies.map(c => ({
4703
+ kind: 'copy', order: c.insertOrder, sectionIndex: c.targetSection,
4704
+ sourceSection: c.sourceSection, source: c.source, anchor: c.anchor, paragraphId: c.paragraphId,
4705
+ })),
4706
+ ...this._pendingParagraphMoves.map(m => ({
4707
+ kind: 'move', order: m.insertOrder, sectionIndex: m.targetSection,
4708
+ sourceSection: m.sourceSection, source: m.source, anchor: m.anchor,
4709
+ })),
4710
+ ].sort((a, b) => a.order - b.order);
4711
+ if (edits.length === 0)
4712
+ return;
4713
+ const sections = new Map();
4714
+ const load = async (index) => {
4715
+ if (sections.has(index))
4716
+ return sections.get(index);
4717
+ const file = this._zip.file(`Contents/section${index}.xml`);
4718
+ if (!file)
4719
+ return undefined;
4720
+ const xml = await file.async('string');
4721
+ sections.set(index, xml);
4722
+ return xml;
4723
+ };
4724
+ // Numeric ids for generated cell/wrapper paragraphs must not collide.
4725
+ const maxIdBySection = new Map();
4726
+ const nextIdFor = (index, xml) => {
4727
+ if (!maxIdBySection.has(index)) {
4728
+ let maxId = 0;
4729
+ for (const m of xml.matchAll(/\bid="(\d+)"/g)) {
4730
+ const n = parseInt(m[1], 10);
4731
+ if (n < 2147483648 && n > maxId)
4732
+ maxId = n;
4733
+ }
4734
+ maxIdBySection.set(index, maxId);
4735
+ }
4736
+ return () => {
4737
+ const next = maxIdBySection.get(index) + 1;
4738
+ maxIdBySection.set(index, next);
4739
+ return next;
4740
+ };
4741
+ };
4742
+ const placeAfter = (xml, anchor) => {
4743
+ const position = anchor ? this.findAnchorEnd(xml, anchor) : this.findSectionHeadEnd(xml);
4744
+ if (position !== -1)
4745
+ return position;
4746
+ // The anchor vanished (e.g. deleted later in the same session) —
4747
+ // append rather than drop the user's content.
4748
+ return Math.max(xml.lastIndexOf('</hs:sec>'), xml.lastIndexOf('</hp:sec>'));
4749
+ };
4750
+ for (const edit of edits) {
4751
+ if (edit.kind === 'copy' || edit.kind === 'move') {
4752
+ const srcXml = await load(edit.sourceSection);
4753
+ if (srcXml === undefined)
4754
+ continue;
4755
+ // Only a section-level paragraph can be copied or moved as a unit; a
4756
+ // paragraph inside a header or text box would drag its container along.
4757
+ const found = this.findParsedParagraph(srcXml, edit.source);
4758
+ if (!found)
4759
+ continue;
4760
+ const srcEl = this.findTopLevelFullElements(srcXml)
4761
+ .find(el => el.type === 'p' && el.startIndex === found.start && el.endIndex === found.end);
4762
+ if (!srcEl)
4763
+ continue;
4764
+ let fragment = srcEl.xml;
4765
+ if (edit.kind === 'copy') {
4766
+ // Same id as the memory copy, so later edits anchored on it find it.
4767
+ fragment = fragment.replace(/^<(hp|hs):p\b([^>]*?)\bid="[^"]*"/, `<$1:p$2id="${edit.paragraphId}"`);
4768
+ // The clone inherits the source's fixed <hp:lineseg> geometry; reset
4769
+ // it so replacement text of a different length does not overlap.
4770
+ fragment = this.resetLinesegInXml(fragment);
4771
+ }
4772
+ else {
4773
+ sections.set(edit.sourceSection, srcXml.slice(0, srcEl.startIndex) + srcXml.slice(srcEl.endIndex));
4191
4774
  }
4775
+ const tgtXml = await load(edit.sectionIndex);
4776
+ if (tgtXml === undefined)
4777
+ continue;
4778
+ const position = placeAfter(tgtXml, edit.anchor);
4779
+ if (position === -1)
4780
+ continue;
4781
+ sections.set(edit.sectionIndex, tgtXml.slice(0, position) + fragment + tgtXml.slice(position));
4782
+ continue;
4192
4783
  }
4193
- this._zip.file(sectionPath, xml);
4784
+ const xml = await load(edit.sectionIndex);
4785
+ if (xml === undefined)
4786
+ continue;
4787
+ const position = placeAfter(xml, edit.anchor);
4788
+ if (position === -1)
4789
+ continue;
4790
+ const newXml = edit.kind === 'paragraph'
4791
+ ? `<hp:p id="${edit.paragraphId}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0"><hp:t>${this.escapeXml(edit.text)}</hp:t></hp:run></hp:p>`
4792
+ : this.buildInsertedTableXml(edit, nextIdFor(edit.sectionIndex, xml));
4793
+ sections.set(edit.sectionIndex, xml.slice(0, position) + newXml + xml.slice(position));
4794
+ }
4795
+ for (const [index, xml] of sections) {
4796
+ this._zip.file(`Contents/section${index}.xml`, xml);
4194
4797
  }
4195
4798
  }
4196
4799
  /**
@@ -4277,12 +4880,13 @@ class HwpxDocument {
4277
4880
  idMap.set(oldId, newId);
4278
4881
  }
4279
4882
  }
4280
- // Second pass: replace all IDs
4281
- let result = xml;
4282
- for (const [oldId, newId] of idMap) {
4283
- result = result.replace(new RegExp(`id="${oldId}"`, 'g'), `id="${newId}"`);
4284
- }
4285
- return result;
4883
+ // Second pass: replace every id in one scan. Building a RegExp per old id
4884
+ // broke on ids with regex metacharacters, and replacing ids one at a time
4885
+ // could rewrite an id that an earlier replacement had just produced.
4886
+ return xml.replace(/id="([^"]+)"/g, (whole, oldId) => {
4887
+ const newId = idMap.get(oldId);
4888
+ return newId === undefined ? whole : `id="${newId}"`;
4889
+ });
4286
4890
  }
4287
4891
  /**
4288
4892
  * Find the position to insert an element after a given element index.
@@ -4300,7 +4904,7 @@ class HwpxDocument {
4300
4904
  // Find all root-level elements (paragraphs, tables)
4301
4905
  const elements = [];
4302
4906
  // Find paragraphs (not inside subList)
4303
- const pRegex = /<hp:p[^>]*>[\s\S]*?<\/hp:p>/g;
4907
+ const pRegex = /<hp:p(?:\s[^>]*)?>[\s\S]*?<\/hp:p>/g;
4304
4908
  let match;
4305
4909
  // Find tables
4306
4910
  const tables = this.findAllTables(xml);
@@ -4314,7 +4918,7 @@ class HwpxDocument {
4314
4918
  const end = start + match[0].length;
4315
4919
  // Check if this paragraph is inside a table (inside subList)
4316
4920
  const beforeMatch = xml.substring(0, start);
4317
- const subListOpen = (beforeMatch.match(/<hp:subList[^>]*>/g) || []).length;
4921
+ const subListOpen = (beforeMatch.match(/<hp:subList(?:\s[^>]*)?>/g) || []).length;
4318
4922
  const subListClose = (beforeMatch.match(/<\/hp:subList>/g) || []).length;
4319
4923
  if (subListOpen === subListClose) {
4320
4924
  // This is a root-level paragraph
@@ -4330,111 +4934,6 @@ class HwpxDocument {
4330
4934
  // Return position after the element at afterIndex
4331
4935
  return elements[afterIndex].end;
4332
4936
  }
4333
- /**
4334
- * Apply paragraph inserts to XML.
4335
- * Inserts new paragraphs at the specified positions.
4336
- */
4337
- async applyParagraphInsertsToXml() {
4338
- if (!this._zip)
4339
- return;
4340
- // Group inserts by section
4341
- const insertsBySection = new Map();
4342
- for (const insert of this._pendingParagraphInserts) {
4343
- const sectionInserts = insertsBySection.get(insert.sectionIndex) || [];
4344
- sectionInserts.push({
4345
- afterElementIndex: insert.afterElementIndex,
4346
- paragraphId: insert.paragraphId,
4347
- text: insert.text,
4348
- });
4349
- insertsBySection.set(insert.sectionIndex, sectionInserts);
4350
- }
4351
- // Process each section
4352
- for (const [sectionIndex, inserts] of insertsBySection) {
4353
- const sectionPath = `Contents/section${sectionIndex}.xml`;
4354
- const file = this._zip.file(sectionPath);
4355
- if (!file)
4356
- continue;
4357
- let xml = await file.async('string');
4358
- // Sort inserts by afterElementIndex in ascending order
4359
- // This ensures each insert happens at the correct position as XML grows
4360
- const sortedInserts = [...inserts].sort((a, b) => a.afterElementIndex - b.afterElementIndex);
4361
- for (const insert of sortedInserts) {
4362
- // Escape text for XML
4363
- const escapedText = this.escapeXml(insert.text);
4364
- // Build paragraph XML
4365
- const paragraphXml = `<hp:p id="${insert.paragraphId}" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0"><hp:run charPrIDRef="0"><hp:t>${escapedText}</hp:t></hp:run></hp:p>`;
4366
- // Find the position to insert
4367
- let insertPosition = -1;
4368
- let elementCount = -1;
4369
- let searchPos = 0;
4370
- // Find paragraphs and tables at root level using balanced bracket matching
4371
- while (searchPos < xml.length) {
4372
- // Look for next <hp:p or <hp:tbl
4373
- const nextP = xml.indexOf('<hp:p ', searchPos);
4374
- const nextTbl = xml.indexOf('<hp:tbl ', searchPos);
4375
- let nextPos = -1;
4376
- let isTable = false;
4377
- if (nextP !== -1 && (nextTbl === -1 || nextP < nextTbl)) {
4378
- nextPos = nextP;
4379
- isTable = false;
4380
- }
4381
- else if (nextTbl !== -1) {
4382
- nextPos = nextTbl;
4383
- isTable = true;
4384
- }
4385
- if (nextPos === -1)
4386
- break;
4387
- // Check if this is inside a subList (nested)
4388
- const beforeText = xml.substring(Math.max(0, nextPos - HwpxDocument.NESTED_CHECK_LOOKBACK), nextPos);
4389
- const subListOpen = beforeText.lastIndexOf('<hp:subList');
4390
- const subListClose = beforeText.lastIndexOf('</hp:subList>');
4391
- const isNested = subListOpen > subListClose;
4392
- if (!isNested) {
4393
- elementCount++;
4394
- // Find the end of this element using balanced bracket matching
4395
- const endPos = isTable
4396
- ? HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:tbl', '</hp:tbl>')
4397
- : HwpxDocument.findClosingTagPosition(xml, nextPos + 1, '<hp:p ', '</hp:p>');
4398
- if (endPos === -1) {
4399
- searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
4400
- continue;
4401
- }
4402
- if (elementCount === insert.afterElementIndex) {
4403
- insertPosition = endPos;
4404
- break;
4405
- }
4406
- searchPos = endPos;
4407
- }
4408
- else {
4409
- searchPos = nextPos + HwpxDocument.SEARCH_SKIP_OFFSET;
4410
- }
4411
- }
4412
- // If afterElementIndex is -1, insert after the first paragraph (which contains secPr)
4413
- // IMPORTANT: <hp:secPr> must remain in the first paragraph for the document to be valid
4414
- if (insert.afterElementIndex === -1) {
4415
- // Find the end of the first <hp:p> element (which contains <hp:secPr>)
4416
- const firstPStart = xml.indexOf('<hp:p');
4417
- if (firstPStart !== -1) {
4418
- const firstPEnd = xml.indexOf('</hp:p>', firstPStart);
4419
- if (firstPEnd !== -1) {
4420
- insertPosition = firstPEnd + '</hp:p>'.length;
4421
- }
4422
- }
4423
- }
4424
- // If position not found, insert at end of section (before </hs:sec>)
4425
- if (insertPosition === -1) {
4426
- const secEnd = xml.lastIndexOf('</hs:sec>');
4427
- if (secEnd !== -1) {
4428
- insertPosition = secEnd;
4429
- }
4430
- }
4431
- if (insertPosition !== -1) {
4432
- xml = xml.substring(0, insertPosition) + paragraphXml + xml.substring(insertPosition);
4433
- }
4434
- }
4435
- this._zip.file(sectionPath, xml);
4436
- }
4437
- }
4438
4937
  /**
4439
4938
  * Apply nested table inserts to XML.
4440
4939
  * Inserts a new table inside a cell of an existing table.
@@ -4506,8 +5005,11 @@ class HwpxDocument {
4506
5005
  if (insert.col >= cells.length)
4507
5006
  continue;
4508
5007
  const cellXml = cells[insert.col].xml;
4509
- // Generate nested table XML
4510
- const nestedTableXml = this.generateNestedTableXml(insert.nestedRows, insert.nestedCols, insert.data);
5008
+ // Size the nested table to the parent cell. A fixed per-cell width
5009
+ // ignored the parent and pushed columns past its border (measured:
5010
+ // 3 × 8000 = 24000 inside a 21260-wide cell).
5011
+ const innerWidth = this.getCellInnerWidth(cellXml, tableXml);
5012
+ const nestedTableXml = this.generateNestedTableXml(insert.nestedRows, insert.nestedCols, insert.data, innerWidth);
4511
5013
  // Insert nested table into cell
4512
5014
  const updatedCellXml = this.insertNestedTableIntoCell(cellXml, nestedTableXml);
4513
5015
  // Update the row with the new cell
@@ -4538,14 +5040,54 @@ class HwpxDocument {
4538
5040
  /**
4539
5041
  * Generate XML for a nested table.
4540
5042
  */
4541
- generateNestedTableXml(rows, cols, data) {
5043
+ /**
5044
+ * Usable width inside a table cell, in hwpunit.
5045
+ *
5046
+ * A cell with hasMargin="0" takes its padding from the table's inMargin, so
5047
+ * the cell's own cellMargin is only authoritative when hasMargin="1".
5048
+ *
5049
+ * Every lookup is scoped to the cell's (or table's) own markup. A nested
5050
+ * table inside the cell carries its own cellSz/cellMargin/inMargin, and a
5051
+ * first-match regex over the whole cell would read those instead — which
5052
+ * sized a second nested table to the first one's column (7086 vs 21260).
5053
+ */
5054
+ getCellInnerWidth(cellXml, tableXml) {
5055
+ // hp:tc children are subList → cellAddr → cellSpan → cellSz → cellMargin,
5056
+ // so the cell's own properties are everything after its last </hp:subList>.
5057
+ const subListEnd = cellXml.lastIndexOf('</hp:subList>');
5058
+ const cellProps = subListEnd === -1 ? cellXml : cellXml.slice(subListEnd);
5059
+ const size = cellProps.match(/<hp:cellSz width="(\d+)"/);
5060
+ if (!size)
5061
+ return null;
5062
+ const width = parseInt(size[1], 10);
5063
+ const openTag = cellXml.slice(0, cellXml.indexOf('>') + 1);
5064
+ const usesOwnMargin = /\bhasMargin="1"/.test(openTag);
5065
+ // Table-level inMargin precedes the first row.
5066
+ const firstRow = tableXml.indexOf('<hp:tr');
5067
+ const tableHead = firstRow === -1 ? tableXml : tableXml.slice(0, firstRow);
5068
+ const margin = usesOwnMargin
5069
+ ? cellProps.match(/<hp:cellMargin left="(\d+)" right="(\d+)"/)
5070
+ : tableHead.match(/<hp:inMargin left="(\d+)" right="(\d+)"/);
5071
+ const padding = margin ? parseInt(margin[1], 10) + parseInt(margin[2], 10) : 0;
5072
+ return Math.max(width - padding, 0);
5073
+ }
5074
+ /**
5075
+ * Generate XML for a nested table.
5076
+ *
5077
+ * @param innerWidth Usable width of the parent cell in hwpunit. The nested
5078
+ * table is sized to fit it exactly; columns share the width evenly.
5079
+ */
5080
+ generateNestedTableXml(rows, cols, data, innerWidth = null) {
4542
5081
  // Generate unique ID
4543
5082
  const id = Math.floor(Math.random() * 2000000000) + 100000000;
4544
5083
  const zOrder = Math.floor(Math.random() * 100);
4545
- // Calculate sizes (in hwpunit, 1 hwpunit = 0.1mm)
4546
- const cellWidth = 8000; // ~80mm per cell
5084
+ // Calculate sizes (hwpunit, 100 = 1pt). Without a parent width fall back to
5085
+ // the previous fixed column width so standalone callers keep working.
5086
+ const tableWidth = innerWidth !== null && innerWidth > 0 ? innerWidth : 8000 * cols;
5087
+ const baseCellWidth = Math.floor(tableWidth / cols);
5088
+ // Give the rounding remainder to the last column so the columns sum to tableWidth.
5089
+ const cellWidthAt = (c) => (c === cols - 1 ? tableWidth - baseCellWidth * (cols - 1) : baseCellWidth);
4547
5090
  const cellHeight = 1400; // ~14mm per cell
4548
- const tableWidth = cellWidth * cols;
4549
5091
  const tableHeight = cellHeight * rows;
4550
5092
  let xml = `<hp:tbl id="${id}" zOrder="${zOrder}" numberingType="TABLE" textWrap="TOP_AND_BOTTOM" textFlow="BOTH_SIDES" lock="0" dropcapstyle="None" pageBreak="NONE" repeatHeader="0" rowCnt="${rows}" colCnt="${cols}" cellSpacing="0" borderFillIDRef="2" noAdjust="0">`;
4551
5093
  // Size element
@@ -4578,7 +5120,7 @@ class HwpxDocument {
4578
5120
  xml += `</hp:subList>`;
4579
5121
  xml += `<hp:cellAddr colAddr="${c}" rowAddr="${r}"/>`;
4580
5122
  xml += `<hp:cellSpan colSpan="1" rowSpan="1"/>`;
4581
- xml += `<hp:cellSz width="${cellWidth}" height="${cellHeight}"/>`;
5123
+ xml += `<hp:cellSz width="${cellWidthAt(c)}" height="${cellHeight}"/>`;
4582
5124
  xml += `<hp:cellMargin left="141" right="141" top="141" bottom="141"/>`;
4583
5125
  xml += `</hp:tc>`;
4584
5126
  }
@@ -4593,13 +5135,13 @@ class HwpxDocument {
4593
5135
  */
4594
5136
  insertNestedTableIntoCell(cellXml, nestedTableXml) {
4595
5137
  // Find the subList in the cell
4596
- const subListMatch = cellXml.match(/<hp:subList[^>]*>/);
5138
+ const subListMatch = cellXml.match(/<hp:subList(?:\s[^>]*)?>/);
4597
5139
  if (!subListMatch) {
4598
5140
  // No subList, try to add to paragraph directly
4599
- const pMatch = cellXml.match(/<hp:p[^>]*>/);
5141
+ const pMatch = cellXml.match(/<hp:p(?:\s[^>]*)?>/);
4600
5142
  if (pMatch) {
4601
5143
  const insertPos = cellXml.indexOf(pMatch[0]) + pMatch[0].length;
4602
- const runXml = `<hp:run charPrIDRef="0"><hp:t> </hp:t>${nestedTableXml}<hp:t/></hp:run>`;
5144
+ const runXml = `<hp:run charPrIDRef="0">${nestedTableXml}<hp:t/></hp:run>`;
4603
5145
  return cellXml.substring(0, insertPos) + runXml + cellXml.substring(insertPos);
4604
5146
  }
4605
5147
  return cellXml;
@@ -4618,8 +5160,10 @@ class HwpxDocument {
4618
5160
  return cellXml;
4619
5161
  // Find the end of the opening <hp:p ...> tag
4620
5162
  const pTagEnd = cellXml.indexOf('>', pStart) + 1;
4621
- // Create new run with nested table
4622
- const runXml = `<hp:run charPrIDRef="0"><hp:t> </hp:t>${nestedTableXml}<hp:t/></hp:run>`;
5163
+ // Create new run with nested table. No leading text: a space before an
5164
+ // inline (treatAsChar) table that fills the cell width forces the table onto
5165
+ // a second line and leaves an empty first line above it (measured in Hancom).
5166
+ const runXml = `<hp:run charPrIDRef="0">${nestedTableXml}<hp:t/></hp:run>`;
4623
5167
  // Insert after the opening <hp:p> tag
4624
5168
  return cellXml.substring(0, pTagEnd) + runXml + cellXml.substring(pTagEnd);
4625
5169
  }
@@ -4904,7 +5448,7 @@ class HwpxDocument {
4904
5448
  }
4905
5449
  // If no cell before masterCol, insert at the beginning of row content
4906
5450
  if (insertPoint === -1) {
4907
- const trMatch = updatedRowXml.match(/<(hp|hs):tr[^>]*>/);
5451
+ const trMatch = updatedRowXml.match(/<(hp|hs):tr(?:\s[^>]*)?>/);
4908
5452
  if (trMatch) {
4909
5453
  insertPoint = trMatch[0].length;
4910
5454
  }
@@ -4941,6 +5485,54 @@ class HwpxDocument {
4941
5485
  </hp:subList>
4942
5486
  </hp:tc>`;
4943
5487
  }
5488
+ /**
5489
+ * Replay every pending table edit (cell text, merge/split, nested table,
5490
+ * cell image, cell hanging indent, row/column insert/delete) in call order.
5491
+ *
5492
+ * Each index an edit carries is relative to the table as it was when the
5493
+ * edit was made. Applying by kind (all cell writes, then all row inserts,
5494
+ * then all column inserts ...) wrote text into the pre-insert layout; and
5495
+ * the row appliers sort their own queue by index, which reorders two
5496
+ * inserts or two deletes on the same table. So edits that change a table's
5497
+ * row/column layout run one at a time. Runs of layout-preserving edits
5498
+ * (cell text, indents, images, nested tables) go to their applier together.
5499
+ */
5500
+ async applyTableOpsInCallOrder() {
5501
+ const kinds = [
5502
+ { layout: false, take: () => this._pendingTableCellUpdates, put: o => { this._pendingTableCellUpdates = o; }, apply: () => this.applyTableCellUpdatesToXml() },
5503
+ { layout: true, take: () => this._pendingCellMerges, put: o => { this._pendingCellMerges = o; }, apply: () => this.applyCellMergesToXml() },
5504
+ { layout: true, take: () => this._pendingCellSplits, put: o => { this._pendingCellSplits = o; }, apply: () => this.applyCellSplitsToXml() },
5505
+ { layout: false, take: () => this._pendingNestedTableInserts, put: o => { this._pendingNestedTableInserts = o; }, apply: () => this.applyNestedTableInsertsToXml() },
5506
+ { layout: false, take: () => this._pendingCellImageInserts, put: o => { this._pendingCellImageInserts = o; }, apply: () => this.applyCellImageInsertsToXml() },
5507
+ { layout: false, take: () => this._pendingTableCellHangingIndents, put: o => { this._pendingTableCellHangingIndents = o; }, apply: () => this.applyTableCellHangingIndentsToXml() },
5508
+ { layout: true, take: () => this._pendingTableRowInserts, put: o => { this._pendingTableRowInserts = o; }, apply: () => this.applyTableRowInsertsToXml() },
5509
+ { layout: true, take: () => this._pendingTableRowDeletes, put: o => { this._pendingTableRowDeletes = o; }, apply: () => this.applyTableRowDeletesToXml() },
5510
+ { layout: true, take: () => this._pendingTableColumnInserts, put: o => { this._pendingTableColumnInserts = o; }, apply: () => this.applyTableColumnInsertsToXml() },
5511
+ { layout: true, take: () => this._pendingTableColumnDeletes, put: o => { this._pendingTableColumnDeletes = o; }, apply: () => this.applyTableColumnDeletesToXml() },
5512
+ ];
5513
+ // Every push goes through queueTableOp, so every op has a sequence number;
5514
+ // a missing one would sort last and keep its queue position.
5515
+ const all = [];
5516
+ for (const kind of kinds) {
5517
+ kind.take().forEach((op, pos) => all.push({ kind, op, seq: this._tableOpSeq.get(op) ?? Number.MAX_SAFE_INTEGER, pos }));
5518
+ kind.put([]);
5519
+ }
5520
+ all.sort((a, b) => a.seq - b.seq || a.pos - b.pos);
5521
+ for (let i = 0; i < all.length;) {
5522
+ const kind = all[i].kind;
5523
+ const batch = [all[i++].op];
5524
+ if (!kind.layout)
5525
+ while (i < all.length && all[i].kind === kind)
5526
+ batch.push(all[i++].op);
5527
+ kind.put(batch);
5528
+ try {
5529
+ await kind.apply();
5530
+ }
5531
+ finally {
5532
+ kind.put([]);
5533
+ }
5534
+ }
5535
+ }
4944
5536
  /**
4945
5537
  * Apply table cell updates to XML while preserving original structure.
4946
5538
  * This function modifies only the text content of specific cells,
@@ -4958,7 +5550,7 @@ class HwpxDocument {
4958
5550
  const updatesBySection = new Map();
4959
5551
  for (const update of this._pendingTableCellUpdates) {
4960
5552
  const sectionUpdates = updatesBySection.get(update.sectionIndex) || [];
4961
- sectionUpdates.push({ tableId: update.tableId, row: update.row, col: update.col, text: update.text, charShapeId: update.charShapeId });
5553
+ sectionUpdates.push({ tableId: update.tableId, row: update.row, col: update.col, colAddr: update.colAddr, text: update.text, charShapeId: update.charShapeId });
4962
5554
  updatesBySection.set(update.sectionIndex, sectionUpdates);
4963
5555
  }
4964
5556
  // Process each section that has updates
@@ -4974,7 +5566,7 @@ class HwpxDocument {
4974
5566
  const updatesByTableId = new Map();
4975
5567
  for (const update of updates) {
4976
5568
  const tableUpdates = updatesByTableId.get(update.tableId) || [];
4977
- tableUpdates.push({ row: update.row, col: update.col, text: update.text, charShapeId: update.charShapeId });
5569
+ tableUpdates.push({ row: update.row, col: update.col, colAddr: update.colAddr, text: update.text, charShapeId: update.charShapeId });
4978
5570
  updatesByTableId.set(update.tableId, tableUpdates);
4979
5571
  }
4980
5572
  // Process each table that has updates (by ID)
@@ -5225,12 +5817,14 @@ class HwpxDocument {
5225
5817
  * Find a table by its ID in XML.
5226
5818
  */
5227
5819
  findTableById(xml, tableId) {
5228
- // Match table with specific ID
5229
- const tableStartRegex = new RegExp(`<(?:hp|hs|hc):tbl[^>]*\\bid="${tableId}"[^>]*>`, 'g');
5820
+ // Match table with specific ID. The id comes from document XML, so it is
5821
+ // escaped: an id holding '.', '(' or '+' matched another table or threw.
5822
+ const id = this.escapeRegex(tableId);
5823
+ const tableStartRegex = new RegExp(`<(?:hp|hs|hc):tbl\\s[^>]*\\bid="${id}"[^>]*>`, 'g');
5230
5824
  const match = tableStartRegex.exec(xml);
5231
5825
  if (!match) {
5232
5826
  // Try alternate ID format (id='...' instead of id="...")
5233
- const altRegex = new RegExp(`<(?:hp|hs|hc):tbl[^>]*\\bid='${tableId}'[^>]*>`, 'g');
5827
+ const altRegex = new RegExp(`<(?:hp|hs|hc):tbl\\s[^>]*\\bid='${id}'[^>]*>`, 'g');
5234
5828
  const altMatch = altRegex.exec(xml);
5235
5829
  if (!altMatch)
5236
5830
  return null;
@@ -5369,7 +5963,7 @@ class HwpxDocument {
5369
5963
  findAllTables(xml) {
5370
5964
  const tables = [];
5371
5965
  // Match both hp:tbl and hs:tbl (different namespace prefixes)
5372
- const tableStartRegex = /<(?:hp|hs|hc):tbl[^>]*>/g;
5966
+ const tableStartRegex = /<(?:hp|hs|hc):tbl(?:\s[^>]*)?>/g;
5373
5967
  let match;
5374
5968
  while ((match = tableStartRegex.exec(xml)) !== null) {
5375
5969
  const startIndex = match.index;
@@ -5474,7 +6068,7 @@ class HwpxDocument {
5474
6068
  if (!updatesByRow.has(update.row)) {
5475
6069
  updatesByRow.set(update.row, []);
5476
6070
  }
5477
- updatesByRow.get(update.row).push({ col: update.col, text: update.text, charShapeId: update.charShapeId });
6071
+ updatesByRow.get(update.row).push({ col: update.col, colAddr: update.colAddr, text: update.text, charShapeId: update.charShapeId });
5478
6072
  }
5479
6073
  // Sort row indices descending to process from end to start (avoid index shifting)
5480
6074
  const sortedRowIndices = Array.from(updatesByRow.keys()).sort((a, b) => b - a);
@@ -5520,17 +6114,28 @@ class HwpxDocument {
5520
6114
  let result = rowXml;
5521
6115
  // Find all cells in this row using depth tracking to handle nested tables correctly
5522
6116
  const cells = this.findAllElementsWithDepth(rowXml, 'tc');
6117
+ // Resolve each update to its <hp:tc> index. The cell's grid column
6118
+ // (colAddr) is authoritative: after a merge the XML row no longer has the
6119
+ // covered cells that memory still lists, so the memory position `col`
6120
+ // points one cell too far. `col` is used only when the write has no
6121
+ // colAddr or the row carries no addresses.
6122
+ const cellCols = cells.map(c => this.cellOwnAttr(c.xml, 'colAddr')?.value);
6123
+ const indexOf = (u) => {
6124
+ if (u.colAddr !== undefined && cellCols.some(a => a !== undefined))
6125
+ return cellCols.indexOf(u.colAddr);
6126
+ return u.col < cells.length ? u.col : -1;
6127
+ };
5523
6128
  // Deduplicate updates for the same cell (keep last value)
5524
6129
  // This prevents stale index issues when the same cell is updated multiple times
5525
6130
  const uniqueUpdates = new Map();
5526
6131
  for (const update of updates) {
5527
- uniqueUpdates.set(update.col, update);
6132
+ const at = indexOf(update);
6133
+ if (at >= 0)
6134
+ uniqueUpdates.set(at, { ...update, col: at });
5528
6135
  }
5529
6136
  // Sort updates by col descending to process from right to left (avoid index shifting)
5530
6137
  const sortedUpdates = Array.from(uniqueUpdates.values()).sort((a, b) => b.col - a.col);
5531
6138
  for (const update of sortedUpdates) {
5532
- if (update.col >= cells.length)
5533
- continue;
5534
6139
  const cellData = cells[update.col];
5535
6140
  // Validate cell before update - capture nested table structure
5536
6141
  const cellTblOpen = (cellData.xml.match(/<(?:hp|hs|hc):tbl[\s>]/g) || []).length;
@@ -5606,7 +6211,7 @@ class HwpxDocument {
5606
6211
  xml = xml.replace(/(<(?:hp|hs|hc):run\s+)charPrIDRef="[^"]*"/, `$1charPrIDRef="${charShapeId}"`);
5607
6212
  }
5608
6213
  // Pattern 1: Cell has existing <hp:t> or <hs:t> or <hc:t> tags with content
5609
- const tTagPattern = /(<(?:hp|hs|hc):t[^>]*>)([^<]*)(<\/(?:hp|hs|hc):t>)/g;
6214
+ const tTagPattern = /(<(?:hp|hs|hc):t(?:\s[^>]*)?>)([^<]*)(<\/(?:hp|hs|hc):t>)/g;
5610
6215
  let foundText = false;
5611
6216
  let result = xml.replace(tTagPattern, (match, openTag, _oldText, closeTag, offset) => {
5612
6217
  // Only replace the first text occurrence
@@ -5619,14 +6224,14 @@ class HwpxDocument {
5619
6224
  if (foundText)
5620
6225
  return this.resetLinesegInXml(result);
5621
6226
  // Pattern 2: Cell has empty <hp:t/> or <hp:t></hp:t> tags
5622
- const emptyTTagPattern = /<((?:hp|hs|hc):t)([^>]*)\s*\/>/;
6227
+ const emptyTTagPattern = /<((?:hp|hs|hc):t)((?:\s[^>]*?)?)\s*\/>/;
5623
6228
  const emptyTMatch = xml.match(emptyTTagPattern);
5624
6229
  if (emptyTMatch) {
5625
6230
  const updated = xml.replace(emptyTTagPattern, `<${emptyTMatch[1]}${emptyTMatch[2]}>${escapedText}</${emptyTMatch[1]}>`);
5626
6231
  return this.resetLinesegInXml(updated);
5627
6232
  }
5628
6233
  // Pattern 3a: Self-closing <hp:run .../> - expand to full run with text
5629
- const selfClosingRunPattern = /<((?:hp|hs|hc):run)([^>]*)\s*\/>/;
6234
+ const selfClosingRunPattern = /<((?:hp|hs|hc):run)((?:\s[^>]*?)?)\s*\/>/;
5630
6235
  const selfClosingRunMatch = xml.match(selfClosingRunPattern);
5631
6236
  if (selfClosingRunMatch) {
5632
6237
  const tagName = selfClosingRunMatch[1]; // e.g., "hp:run"
@@ -5645,7 +6250,7 @@ class HwpxDocument {
5645
6250
  return this.resetLinesegInXml(updated);
5646
6251
  }
5647
6252
  // Pattern 3b: Cell has <hp:run> but no <hp:t> - add text inside run
5648
- const runPattern = /(<(?:hp|hs|hc):run[^>]*>)([\s\S]*?)(<\/(?:hp|hs|hc):run>)/;
6253
+ const runPattern = /(<(?:hp|hs|hc):run(?:\s[^>]*)?>)([\s\S]*?)(<\/(?:hp|hs|hc):run>)/;
5649
6254
  const runMatch = xml.match(runPattern);
5650
6255
  if (runMatch) {
5651
6256
  const prefix = runMatch[1].match(/<(hp|hs|hc):run/)?.[1] || 'hp';
@@ -5654,7 +6259,7 @@ class HwpxDocument {
5654
6259
  return this.resetLinesegInXml(updated);
5655
6260
  }
5656
6261
  // Pattern 4: Cell has <hp:subList><hp:p> structure - find the paragraph and add text
5657
- const subListPattern = /(<(?:hp|hs|hc):subList[^>]*>[\s\S]*?<(?:hp|hs|hc):p[^>]*>)([\s\S]*?)(<\/(?:hp|hs|hc):p>)/;
6262
+ const subListPattern = /(<(?:hp|hs|hc):subList(?:\s[^>]*)?>[\s\S]*?<(?:hp|hs|hc):p(?:\s[^>]*)?>)([\s\S]*?)(<\/(?:hp|hs|hc):p>)/;
5658
6263
  const subListMatch = xml.match(subListPattern);
5659
6264
  if (subListMatch) {
5660
6265
  const prefix = subListMatch[1].match(/<(hp|hs|hc):subList/)?.[1] || 'hp';
@@ -5667,7 +6272,7 @@ class HwpxDocument {
5667
6272
  }
5668
6273
  }
5669
6274
  // Pattern 5: Cell has only <hp:p> without subList
5670
- const pPattern = /(<(?:hp|hs|hc):p[^>]*>)([\s\S]*?)(<\/(?:hp|hs|hc):p>)/;
6275
+ const pPattern = /(<(?:hp|hs|hc):p(?:\s[^>]*)?>)([\s\S]*?)(<\/(?:hp|hs|hc):p>)/;
5671
6276
  const pMatch = xml.match(pPattern);
5672
6277
  if (pMatch) {
5673
6278
  const prefix = pMatch[1].match(/<(hp|hs|hc):p/)?.[1] || 'hp';
@@ -5689,7 +6294,7 @@ class HwpxDocument {
5689
6294
  const charAttr = charShapeId !== undefined ? ` charPrIDRef="${charShapeId}"` : ' charPrIDRef="0"';
5690
6295
  let xml = cellXml;
5691
6296
  // Find the subList element to replace paragraph content
5692
- const subListStartMatch = xml.match(/<(hp|hs|hc):subList[^>]*>/);
6297
+ const subListStartMatch = xml.match(/<(hp|hs|hc):subList(?:\s[^>]*)?>/);
5693
6298
  if (subListStartMatch) {
5694
6299
  const prefix = subListStartMatch[1];
5695
6300
  const startTag = subListStartMatch[0];
@@ -5723,7 +6328,7 @@ class HwpxDocument {
5723
6328
  // Preserve nested tables
5724
6329
  const nestedTables = this.extractNestedTables(subListContent, prefix);
5725
6330
  // Extract paraPrIDRef and styleIDRef from existing paragraph
5726
- const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
6331
+ const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p\s[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
5727
6332
  const paraPrIDRef = existingPMatch?.[1] || '0';
5728
6333
  const styleIDRef = existingPMatch?.[2] || '0';
5729
6334
  const paraId = Math.floor(Math.random() * 2147483647);
@@ -5735,7 +6340,7 @@ class HwpxDocument {
5735
6340
  }
5736
6341
  }
5737
6342
  // Fallback: try to find paragraph directly
5738
- const pStartMatch = xml.match(/<(hp|hs|hc):p[^>]*>/);
6343
+ const pStartMatch = xml.match(/<(hp|hs|hc):p(?:\s[^>]*)?>/);
5739
6344
  if (pStartMatch) {
5740
6345
  const prefix = pStartMatch[1];
5741
6346
  const attrMatch = pStartMatch[0].match(/<(?:hp|hs|hc):p([^>]*)>/);
@@ -5763,7 +6368,7 @@ class HwpxDocument {
5763
6368
  if (depth === 0) {
5764
6369
  lastParagraphEnd = searchIndex;
5765
6370
  const remainingXml = xml.substring(searchIndex);
5766
- const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p[^>]*>/);
6371
+ const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p(?:\s[^>]*)?>/);
5767
6372
  if (!nextPMatch)
5768
6373
  break;
5769
6374
  }
@@ -5794,7 +6399,7 @@ class HwpxDocument {
5794
6399
  const charAttr = charShapeId !== undefined ? ` charPrIDRef="${charShapeId}"` : ' charPrIDRef="0"';
5795
6400
  // Find the OUTER subList element with balanced tag matching
5796
6401
  // This is crucial because cells can contain nested tables with their own subLists
5797
- const subListStartMatch = cellXml.match(/<(hp|hs|hc):subList[^>]*>/);
6402
+ const subListStartMatch = cellXml.match(/<(hp|hs|hc):subList(?:\s[^>]*)?>/);
5798
6403
  if (subListStartMatch) {
5799
6404
  const prefix = subListStartMatch[1];
5800
6405
  const startTag = subListStartMatch[0];
@@ -5832,7 +6437,7 @@ class HwpxDocument {
5832
6437
  // IMPORTANT: Check for nested tables in subList content - preserve them!
5833
6438
  const nestedTables = this.extractNestedTables(subListContent, prefix);
5834
6439
  // Extract paraPrIDRef and styleIDRef from existing paragraph if available
5835
- const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
6440
+ const existingPMatch = subListContent.match(/<(?:hp|hs|hc):p\s[^>]*paraPrIDRef="([^"]*)"[^>]*styleIDRef="([^"]*)"/);
5836
6441
  const paraPrIDRef = existingPMatch?.[1] || '0';
5837
6442
  const styleIDRef = existingPMatch?.[2] || '0';
5838
6443
  // Generate multiple paragraphs with chunked runs for long lines
@@ -5849,7 +6454,7 @@ class HwpxDocument {
5849
6454
  }
5850
6455
  // If no subList found, try to find just paragraphs and replace
5851
6456
  // Use balanced matching for paragraphs too, since they can contain nested tables
5852
- const pStartMatch = cellXml.match(/<(hp|hs|hc):p[^>]*>/);
6457
+ const pStartMatch = cellXml.match(/<(hp|hs|hc):p(?:\s[^>]*)?>/);
5853
6458
  if (pStartMatch) {
5854
6459
  const prefix = pStartMatch[1];
5855
6460
  const firstPStart = cellXml.indexOf(pStartMatch[0]);
@@ -5883,7 +6488,7 @@ class HwpxDocument {
5883
6488
  lastParagraphEnd = searchIndex;
5884
6489
  // Check if there's another paragraph at top level
5885
6490
  const remainingXml = cellXml.substring(searchIndex);
5886
- const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p[^>]*>/);
6491
+ const nextPMatch = remainingXml.match(/^\s*<(hp|hs|hc):p(?:\s[^>]*)?>/);
5887
6492
  if (!nextPMatch) {
5888
6493
  // No more top-level paragraphs
5889
6494
  break;
@@ -6008,7 +6613,7 @@ class HwpxDocument {
6008
6613
  else {
6009
6614
  // Text not found, fall back to first paragraph
6010
6615
  console.warn(`[HwpxDocument] afterText "${insert.afterText}" not found in cell, using first paragraph`);
6011
- const paragraphMatch = targetCell.xml.match(/<hp:p[^>]*>/);
6616
+ const paragraphMatch = targetCell.xml.match(/<hp:p(?:\s[^>]*)?>/);
6012
6617
  if (!paragraphMatch)
6013
6618
  continue;
6014
6619
  insertPosition = targetCell.xml.indexOf(paragraphMatch[0]) + paragraphMatch[0].length;
@@ -6016,7 +6621,7 @@ class HwpxDocument {
6016
6621
  }
6017
6622
  else {
6018
6623
  // Default: find the first <hp:p> in the cell and insert the image inside it
6019
- const paragraphMatch = targetCell.xml.match(/<hp:p[^>]*>/);
6624
+ const paragraphMatch = targetCell.xml.match(/<hp:p(?:\s[^>]*)?>/);
6020
6625
  if (!paragraphMatch)
6021
6626
  continue;
6022
6627
  insertPosition = targetCell.xml.indexOf(paragraphMatch[0]) + paragraphMatch[0].length;
@@ -6106,9 +6711,31 @@ class HwpxDocument {
6106
6711
  async applyDirectTextUpdatesToXml() {
6107
6712
  if (!this._zip)
6108
6713
  return;
6714
+ // Re-anchor every update on the memory paragraph it edits. The element
6715
+ // index and id-occurrence recorded at call time are stale once a later
6716
+ // insert/delete/copy/move reshapes the section: the frozen occurrence then
6717
+ // names another same-id paragraph (measured: [A,B,C] all id="0", edit B,
6718
+ // move C to the front → A was rewritten). At this point the memory model
6719
+ // matches the XML, whose structural edits were already replayed.
6720
+ for (const update of this._pendingDirectTextUpdates) {
6721
+ if (!update.paragraph)
6722
+ continue;
6723
+ const elements = this._content.sections[update.sectionIndex]?.elements ?? [];
6724
+ const now = elements.findIndex(e => e.type === 'paragraph' && e.data === update.paragraph);
6725
+ if (now === -1) {
6726
+ // The paragraph was deleted after the edit; there is nothing to write.
6727
+ update.elementIndex = -1;
6728
+ continue;
6729
+ }
6730
+ update.elementIndex = now;
6731
+ update.paragraphId = update.paragraph.id || '';
6732
+ update.paragraphOccurrence = this.getParagraphOccurrence(update.sectionIndex, now, update.paragraphId);
6733
+ }
6109
6734
  // Group updates by sectionIndex, then by elementIndex
6110
6735
  const updatesBySectionAndElement = new Map();
6111
6736
  for (const update of this._pendingDirectTextUpdates) {
6737
+ if (update.elementIndex < 0)
6738
+ continue;
6112
6739
  let sectionMap = updatesBySectionAndElement.get(update.sectionIndex);
6113
6740
  if (!sectionMap) {
6114
6741
  sectionMap = new Map();
@@ -6127,25 +6754,38 @@ class HwpxDocument {
6127
6754
  if (!file)
6128
6755
  continue;
6129
6756
  let xml = await file.async('string');
6130
- // STEP 1: Pre-compute target paragraph mappings BEFORE any modifications
6131
- // OPTIMIZATION: Use cached XML positions when available (populated during parsing)
6757
+ // STEP 1: Pre-compute target paragraph ranges BEFORE any modifications.
6758
+ //
6759
+ // Each memory paragraph is mapped to its XML paragraph with the parser's
6760
+ // own rule (parsedParagraphStarts), computed once per section. The offsets
6761
+ // the parser cached at load time are not used: they pair memory paragraphs
6762
+ // with a DIFFERENT list (top-level paragraphs of the raw XML), which drifts
6763
+ // wherever the parser lifts paragraphs out of headers, text boxes or
6764
+ // endnotes. Measured on 325 Hancom-saved sections: 16,271 of 70,677 cached
6765
+ // offsets pointed at another paragraph, and an edit then reported success
6766
+ // while its text went to — or vanished into — the wrong paragraph.
6132
6767
  const paragraphTargets = new Map();
6768
+ const starts = this.parsedParagraphStarts(xml);
6769
+ const elements = this._content.sections[sectionIdx]?.elements ?? [];
6770
+ const slotOf = new Map();
6771
+ let slot = 0;
6772
+ elements.forEach((el, i) => {
6773
+ if (this.anchorKeyOf(el)?.kind === 'paragraph')
6774
+ slotOf.set(i, slot++);
6775
+ });
6776
+ const aligned = slot === starts.length;
6133
6777
  for (const [elementIndex, updates] of elementMap) {
6134
- // Try cached position first (from parsing phase)
6135
- const cachedPosition = this.getCachedXmlPosition(sectionIdx, elementIndex);
6136
- if (cachedPosition && cachedPosition.start < xml.length && cachedPosition.end <= xml.length) {
6137
- // Validate cached position by checking if it points to a paragraph element
6138
- const cachedXml = xml.slice(cachedPosition.start, cachedPosition.end);
6139
- if (cachedXml.startsWith('<hp:p') && cachedXml.endsWith('</hp:p>')) {
6140
- paragraphTargets.set(elementIndex, {
6141
- start: cachedPosition.start,
6142
- end: cachedPosition.end,
6143
- xml: cachedXml
6144
- });
6778
+ const k = slotOf.get(elementIndex);
6779
+ if (aligned && k !== undefined) {
6780
+ const start = starts[k];
6781
+ const end = this.findBalancedParagraphEnd(xml, start);
6782
+ if (end !== -1) {
6783
+ paragraphTargets.set(elementIndex, { start, end, xml: xml.slice(start, end) });
6145
6784
  continue;
6146
6785
  }
6147
6786
  }
6148
- // Fallback to full search if no cached position or validation failed
6787
+ // Memory and XML disagree on the paragraph count (should not happen for
6788
+ // parser-produced documents); fall back to id + occurrence search.
6149
6789
  const paragraphId = updates[0]?.paragraphId || '';
6150
6790
  const paragraphOccurrence = updates[0]?.paragraphOccurrence ?? 0;
6151
6791
  const target = this.findTargetParagraphForUpdate(xml, sectionIdx, elementIndex, updates, paragraphId, paragraphOccurrence);
@@ -6170,11 +6810,11 @@ class HwpxDocument {
6170
6810
  // Sort by runIndex to process in order
6171
6811
  updates.sort((a, b) => a.runIndex - b.runIndex);
6172
6812
  // Apply the update directly using pre-computed target location
6173
- if (updates.length > 1) {
6813
+ if (updates.length > 1 || /<hp:t\b/.test(target.xml)) {
6174
6814
  xml = this.replaceRunsInParagraphDirect(xml, target, updates);
6175
6815
  }
6176
6816
  else {
6177
- // For single update, use the existing method with pre-computed target
6817
+ // Empty runs without text tags need a new hp:t element.
6178
6818
  xml = this.replaceTextInElementDirect(xml, target, updates[0].oldText, updates[0].newText);
6179
6819
  }
6180
6820
  }
@@ -6449,10 +7089,10 @@ class HwpxDocument {
6449
7089
  }
6450
7090
  else if (/<hp:t\b[^>]*>/.test(run.xml)) {
6451
7091
  // Has <hp:t>...</hp:t> tags - replace content of FIRST one only (no g flag)
6452
- newRunXml = run.xml.replace(/(<hp:t[^>]*>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
7092
+ newRunXml = run.xml.replace(/(<hp:t(?:\s[^>]*)?>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
6453
7093
  // Remove any additional <hp:t>...</hp:t> tags to prevent duplication
6454
7094
  let firstReplaced = false;
6455
- newRunXml = newRunXml.replace(/<hp:t[^>]*>[^<]*<\/hp:t>/g, (match) => {
7095
+ newRunXml = newRunXml.replace(/<hp:t(?:\s[^>]*)?>[^<]*<\/hp:t>/g, (match) => {
6456
7096
  if (!firstReplaced) {
6457
7097
  firstReplaced = true;
6458
7098
  return match; // Keep the first one
@@ -6483,27 +7123,14 @@ class HwpxDocument {
6483
7123
  return xml.slice(0, targetInOriginal.start) + newParagraphXml + xml.slice(targetInOriginal.end);
6484
7124
  }
6485
7125
  /**
6486
- * Calculate the occurrence index for a paragraph with given ID.
6487
- * Returns how many paragraphs with the same ID appear before this one.
7126
+ * Occurrence index of the paragraph at `elementIndex` among paragraphs with
7127
+ * the same id — counted with the same rule as insert anchors
7128
+ * (resolveElementAnchor), so a text update finds the paragraph that
7129
+ * findParsedParagraph resolves. Divider paragraphs parsed as 'hr' count.
6488
7130
  */
6489
7131
  getParagraphOccurrence(sectionIndex, elementIndex, paragraphId) {
6490
- // Use _content.sections (same as findParagraphByPath) instead of _sections
6491
- if (!this._content || !this._content.sections || !this._content.sections[sectionIndex])
6492
- return 0;
6493
- const section = this._content.sections[sectionIndex];
6494
- if (!section || !section.elements)
6495
- return 0;
6496
- let occurrenceCount = 0;
6497
- for (let i = 0; i < elementIndex; i++) {
6498
- const element = section.elements[i];
6499
- if (element && element.type === 'paragraph') { // Use 'paragraph' not 'p'
6500
- const para = element.data;
6501
- if (para.id === paragraphId) {
6502
- occurrenceCount++;
6503
- }
6504
- }
6505
- }
6506
- return occurrenceCount;
7132
+ const anchor = this.resolveElementAnchor(sectionIndex, elementIndex);
7133
+ return anchor && anchor.kind === 'paragraph' && anchor.id === paragraphId ? anchor.occurrence : 0;
6507
7134
  }
6508
7135
  /**
6509
7136
  * Find paragraph by its ID attribute and occurrence index.
@@ -6799,18 +7426,22 @@ class HwpxDocument {
6799
7426
  // TIER 1: ID-based lookup (most reliable)
6800
7427
  // TIER 2: Index-based lookup with text validation
6801
7428
  // TIER 3: Fuzzy text matching fallback
6802
- // TIER 1: ID-based lookup - DISABLED
6803
- // Problem: XML counting includes nested paragraphs (inside tables),
6804
- // but _content.sections.elements only has top-level elements.
6805
- // This mismatch causes wrong paragraph selection.
6806
- // Solution: Skip ID-based lookup and use index-based (TIER 2) instead.
7429
+ // TIER 1: id + occurrence, counted with the parser's paragraph rule.
6807
7430
  //
6808
- // if (paragraphId) {
6809
- // const idBasedTarget = this.findParagraphById(xml, paragraphId, paragraphOccurrence ?? 0);
6810
- // if (idBasedTarget) {
6811
- // return idBasedTarget;
6812
- // }
6813
- // }
7431
+ // This was disabled because counting every <hp:p> in the XML included cell
7432
+ // paragraphs the memory model does not have. findAnchorEnd counts only
7433
+ // top-level paragraphs the parser keeps, so the occurrence recorded from
7434
+ // the memory model names the same node. Index lookup (TIER 2) is wrong
7435
+ // after a copy: its ±2 text search finds the original first because the
7436
+ // copy carries the same text, and the edit lands on the original.
7437
+ if (paragraphId) {
7438
+ // The paragraph's OWN range. For a paragraph inside a header or text
7439
+ // box, the enclosing top-level paragraph would rewrite the whole body.
7440
+ const hit = this.findParsedParagraph(xml, { kind: 'paragraph', id: paragraphId, occurrence: paragraphOccurrence ?? 0 });
7441
+ if (hit) {
7442
+ return { start: hit.start, end: hit.end, xml: xml.slice(hit.start, hit.end) };
7443
+ }
7444
+ }
6814
7445
  // Calculate paragraph index using _content.sections.elements (same source as elementIndex)
6815
7446
  // This ensures consistency between elementIndex and paragraph counting
6816
7447
  let topLevelParagraphIndex = 0;
@@ -6875,68 +7506,151 @@ class HwpxDocument {
6875
7506
  for (const update of updates) {
6876
7507
  updateMap.set(update.runIndex, update.newText);
6877
7508
  }
6878
- // Find all hp:run elements with their positions
6879
- // Use non-greedy matching and track depth for nested elements
6880
- const runs = [];
6881
- const runOpenRegex = /<hp:run\b[^>]*>/g;
6882
- let match;
6883
- while ((match = runOpenRegex.exec(paragraphXml)) !== null) {
6884
- const runStart = match.index;
6885
- let depth = 1;
6886
- let pos = runStart + match[0].length;
6887
- // Find matching </hp:run> using depth tracking
6888
- while (depth > 0 && pos < paragraphXml.length) {
6889
- const nextOpen = paragraphXml.indexOf('<hp:run', pos);
6890
- const nextClose = paragraphXml.indexOf('</hp:run>', pos);
6891
- if (nextClose === -1)
6892
- break;
6893
- if (nextOpen !== -1 && nextOpen < nextClose) {
6894
- depth++;
6895
- pos = nextOpen + 7;
7509
+ // The paragraph's OWN runs only — its direct children. A paragraph that
7510
+ // holds a table, text box, footnote or endnote also contains the runs of
7511
+ // every paragraph inside those containers. Counting those as its own made
7512
+ // "run N" land in a table cell or endnote: the reported success wrote the
7513
+ // new text into a nested paragraph (or into nothing) and cut the rest.
7514
+ // Measured: 39 of 60 Hancom files lost the text this way (2026-09-24).
7515
+ const runs = this.findDirectChildRuns(paragraphXml);
7516
+ // Filter to only runs that have <hp:t> content (matching memory model behavior)
7517
+ // Memory model only counts runs with text, not runs with only <hp:ctrl> etc.
7518
+ const textRuns = runs.filter(run => /<hp:t\b/.test(this.ownRunText(run.xml)));
7519
+ // The parser creates a model run per non-empty hp:t, not per hp:run.
7520
+ // Merge those updates back into their shared XML run without losing a suffix.
7521
+ const xmlRunUpdates = new Map();
7522
+ let modelRunIndex = 0;
7523
+ for (let i = 0; i < textRuns.length; i++) {
7524
+ const textNodes = [...this.ownRunText(textRuns[i].xml).matchAll(/<hp:t\b[^>]*>([^<]+)<\/hp:t>/g)];
7525
+ const count = Math.max(1, textNodes.length);
7526
+ let changed = false;
7527
+ let escapedText = '';
7528
+ for (let offset = 0; offset < count; offset++) {
7529
+ const index = modelRunIndex + offset;
7530
+ if (updateMap.has(index)) {
7531
+ escapedText += this.escapeXml(updateMap.get(index));
7532
+ changed = true;
6896
7533
  }
6897
7534
  else {
6898
- depth--;
6899
- if (depth === 0) {
6900
- const runEnd = nextClose + '</hp:run>'.length;
6901
- runs.push({
6902
- start: runStart,
6903
- end: runEnd,
6904
- xml: paragraphXml.slice(runStart, runEnd)
6905
- });
6906
- }
6907
- pos = nextClose + 9;
7535
+ escapedText += textNodes[offset]?.[1] || '';
6908
7536
  }
6909
7537
  }
7538
+ if (changed)
7539
+ xmlRunUpdates.set(i, escapedText);
7540
+ modelRunIndex += count;
6910
7541
  }
6911
- // Filter to only runs that have <hp:t> content (matching memory model behavior)
6912
- // Memory model only counts runs with text, not runs with only <hp:ctrl> etc.
6913
- const textRuns = runs.filter(run => /<hp:t\b/.test(run.xml) || /<hp:t\s*\/>/.test(run.xml));
6914
7542
  // Process text runs in reverse order to maintain positions
6915
7543
  for (let i = textRuns.length - 1; i >= 0; i--) {
6916
- if (!updateMap.has(i))
7544
+ if (!xmlRunUpdates.has(i))
6917
7545
  continue;
6918
7546
  const run = textRuns[i];
6919
- const newText = updateMap.get(i);
6920
- const escapedNew = this.escapeXml(newText);
6921
- let newRunXml = run.xml;
6922
- // Find and replace hp:t content within this run
6923
- if (/<hp:t\s*\/>/.test(newRunXml)) {
6924
- // Self-closing tag: <hp:t/> -> <hp:t>newText</hp:t>
6925
- newRunXml = newRunXml.replace(/<hp:t\s*\/>/, `<hp:t>${escapedNew}</hp:t>`);
6926
- }
6927
- else if (/<hp:t\b[^>]*>/.test(newRunXml)) {
6928
- // Has content: replace first hp:t content only
6929
- newRunXml = newRunXml.replace(/(<hp:t\b[^>]*>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
6930
- }
6931
- else {
6932
- // No hp:t tag - add one after the opening hp:run tag
6933
- newRunXml = newRunXml.replace(/(<hp:run\b[^>]*>)/, `$1<hp:t>${escapedNew}</hp:t>`);
6934
- }
7547
+ const escapedNew = xmlRunUpdates.get(i);
7548
+ // Write each XML run's combined text once, preserving text-tag attributes.
7549
+ // Only the run's own <hp:t> are rewritten; text inside a table, equation
7550
+ // or text box that sits in the same run is left untouched.
7551
+ let textWritten = false;
7552
+ const newRunXml = this.mapOwnRunText(run.xml, tXml => tXml.replace(/<hp:t\b([^>]*?)\/>|<hp:t\b([^>]*)>[^<]*<\/hp:t>/g, (_match, selfClosingAttrs, attrs) => {
7553
+ const text = textWritten ? '' : escapedNew;
7554
+ textWritten = true;
7555
+ return `<hp:t${selfClosingAttrs ?? attrs ?? ''}>${text}</hp:t>`;
7556
+ }));
6935
7557
  // Replace in paragraph XML
6936
7558
  paragraphXml = paragraphXml.slice(0, run.start) + newRunXml + paragraphXml.slice(run.end);
6937
7559
  }
6938
7560
  return xml.slice(0, target.start) + paragraphXml + xml.slice(target.end);
6939
7561
  }
7562
+ /** Direct <hp:run> children of a paragraph (runs of nested paragraphs excluded). */
7563
+ findDirectChildRuns(paragraphXml) {
7564
+ const runs = [];
7565
+ const openEnd = paragraphXml.indexOf('>') + 1;
7566
+ let pos = openEnd;
7567
+ let depth = 0; // nesting depth of <hp:p> inside this paragraph
7568
+ const tagRe = /<(\/?)hp:(p|run)\b[^>]*?(\/?)>/g;
7569
+ tagRe.lastIndex = pos;
7570
+ let runStart = -1;
7571
+ let m;
7572
+ while ((m = tagRe.exec(paragraphXml)) !== null) {
7573
+ const [whole, closing, name, selfClosing] = m;
7574
+ if (name === 'p') {
7575
+ if (selfClosing)
7576
+ continue;
7577
+ if (closing) {
7578
+ if (depth === 0)
7579
+ break; // end of this paragraph
7580
+ depth--;
7581
+ }
7582
+ else {
7583
+ depth++;
7584
+ }
7585
+ continue;
7586
+ }
7587
+ if (depth !== 0)
7588
+ continue; // a run of a nested paragraph
7589
+ if (selfClosing) {
7590
+ runs.push({ start: m.index, end: m.index + whole.length, xml: whole });
7591
+ }
7592
+ else if (!closing) {
7593
+ runStart = m.index;
7594
+ }
7595
+ else if (runStart !== -1) {
7596
+ const end = m.index + whole.length;
7597
+ runs.push({ start: runStart, end, xml: paragraphXml.slice(runStart, end) });
7598
+ runStart = -1;
7599
+ }
7600
+ }
7601
+ return runs;
7602
+ }
7603
+ /**
7604
+ * A run's own markup with every nested container (table, equation, text box,
7605
+ * note…) blanked out, so its <hp:t> are the run's own text only.
7606
+ */
7607
+ ownRunText(runXml) {
7608
+ return this.mapOwnRunText(runXml, s => s, true);
7609
+ }
7610
+ /**
7611
+ * Apply `fn` to the parts of a run that are its own text, leaving nested
7612
+ * containers byte-for-byte intact. With `blank`, nested containers are
7613
+ * replaced by an empty marker instead (for reading).
7614
+ */
7615
+ mapOwnRunText(runXml, fn, blank = false) {
7616
+ let out = '';
7617
+ let pos = 0;
7618
+ while (pos < runXml.length) {
7619
+ const rest = runXml.slice(pos);
7620
+ const m = rest.match(HwpxDocument.NESTED_CONTENT);
7621
+ if (!m || m.index === undefined) {
7622
+ out += fn(rest);
7623
+ break;
7624
+ }
7625
+ const openAt = pos + m.index;
7626
+ const name = m[1];
7627
+ out += fn(runXml.slice(pos, openAt));
7628
+ const end = this.findElementEnd(runXml, openAt, name);
7629
+ out += blank ? '<NESTED/>' : runXml.slice(openAt, end);
7630
+ pos = end;
7631
+ }
7632
+ return out;
7633
+ }
7634
+ /** End offset of the <hp:name> element opening at `start` (handles nesting and self-closing). */
7635
+ findElementEnd(xml, start, name) {
7636
+ const tagEnd = xml.indexOf('>', start);
7637
+ if (tagEnd === -1)
7638
+ return xml.length;
7639
+ if (xml[tagEnd - 1] === '/')
7640
+ return tagEnd + 1;
7641
+ const re = new RegExp(`<(/?)hp:${name}\\b[^>]*?(/?)>`, 'g');
7642
+ re.lastIndex = tagEnd + 1;
7643
+ let depth = 1;
7644
+ let m;
7645
+ while ((m = re.exec(xml)) !== null) {
7646
+ if (m[2])
7647
+ continue;
7648
+ depth += m[1] ? -1 : 1;
7649
+ if (depth === 0)
7650
+ return m.index + m[0].length;
7651
+ }
7652
+ return xml.length;
7653
+ }
6940
7654
  /**
6941
7655
  * Replace text in a single run directly using pre-computed target location.
6942
7656
  * Simpler version for single-run updates.
@@ -6950,9 +7664,9 @@ class HwpxDocument {
6950
7664
  // Self-closing: <hp:t/> -> <hp:t>newText</hp:t>
6951
7665
  paragraphXml = paragraphXml.replace(/<hp:t\s*\/>/, `<hp:t>${escapedNew}</hp:t>`);
6952
7666
  }
6953
- else if (/<hp:t[^>]*>/.test(paragraphXml)) {
7667
+ else if (/<hp:t(?:\s[^>]*)?>/.test(paragraphXml)) {
6954
7668
  // Has content or empty: <hp:t>...</hp:t> -> <hp:t>newText</hp:t>
6955
- paragraphXml = paragraphXml.replace(/(<hp:t[^>]*>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
7669
+ paragraphXml = paragraphXml.replace(/(<hp:t(?:\s[^>]*)?>)[^<]*(<\/hp:t>)/, `$1${escapedNew}$2`);
6956
7670
  }
6957
7671
  else if (/<hp:run\b[^>]*>/.test(paragraphXml)) {
6958
7672
  // No <hp:t> tag exists - add one after the <hp:run> opening tag
@@ -7012,7 +7726,7 @@ class HwpxDocument {
7012
7726
  continue;
7013
7727
  // Found the right paragraph! Replace the text
7014
7728
  // Replace within <hp:t> tags
7015
- const pattern1 = new RegExp(`(<hp:t[^>]*>)${this.escapeRegex(escapedOld)}`);
7729
+ const pattern1 = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}`);
7016
7730
  let newParagraphContent = paragraphContent.replace(pattern1, `$1${escapedNew}`);
7017
7731
  // Also try standalone text replacement
7018
7732
  const pattern2 = new RegExp(`>${this.escapeRegex(escapedOld)}<`);
@@ -7275,9 +7989,9 @@ class HwpxDocument {
7275
7989
  // Case 1: Self-closing <hp:t/> - replace with full tag containing new text
7276
7990
  newElementContent = elementContent.replace(/<hp:t\s*\/>/, `<hp:t>${escapedNew}</hp:t>`);
7277
7991
  }
7278
- else if (oldText === '' && /<hp:t[^>]*><\/hp:t>/.test(elementContent)) {
7992
+ else if (oldText === '' && /<hp:t(?:\s[^>]*)?><\/hp:t>/.test(elementContent)) {
7279
7993
  // Case 2: Empty <hp:t></hp:t> - fill with new text
7280
- newElementContent = elementContent.replace(/(<hp:t[^>]*>)<\/hp:t>/, `$1${escapedNew}</hp:t>`);
7994
+ newElementContent = elementContent.replace(/(<hp:t(?:\s[^>]*)?>)<\/hp:t>/, `$1${escapedNew}</hp:t>`);
7281
7995
  }
7282
7996
  else if (oldText === '' && !/<hp:t\b[^>]*>/.test(elementContent)) {
7283
7997
  // Case 3: No hp:t tag at all - add one after the first hp:run opening tag
@@ -7285,7 +7999,7 @@ class HwpxDocument {
7285
7999
  }
7286
8000
  else {
7287
8001
  // Case 4: Normal case - replace text within <hp:t> tags (first match only)
7288
- const pattern1 = new RegExp(`(<hp:t[^>]*>)${this.escapeRegex(escapedOld)}`);
8002
+ const pattern1 = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}`);
7289
8003
  newElementContent = elementContent.replace(pattern1, `$1${escapedNew}`);
7290
8004
  // Also try standalone text replacement if pattern1 didn't match
7291
8005
  if (newElementContent === elementContent) {
@@ -7358,7 +8072,7 @@ class HwpxDocument {
7358
8072
  if (runIndex >= runs.length) {
7359
8073
  // Run index out of bounds, try to replace in any run
7360
8074
  // Replace text within <hp:t> tags (first match only)
7361
- const pattern1 = new RegExp(`(<hp:t[^>]*>)${this.escapeRegex(escapedOld)}`);
8075
+ const pattern1 = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}`);
7362
8076
  let newParagraphContent = paragraphContent.replace(pattern1, `$1${escapedNew}`);
7363
8077
  // Also try standalone text replacement
7364
8078
  if (newParagraphContent === paragraphContent) {
@@ -7371,7 +8085,7 @@ class HwpxDocument {
7371
8085
  const targetRun = runs[runIndex];
7372
8086
  let newRunContent = targetRun.content;
7373
8087
  // Replace within <hp:t> tags in this run
7374
- const tPattern = new RegExp(`(<hp:t[^>]*>)${this.escapeRegex(escapedOld)}(</hp:t>)`);
8088
+ const tPattern = new RegExp(`(<hp:t(?:\\s[^>]*)?>)${this.escapeRegex(escapedOld)}(</hp:t>)`);
7375
8089
  newRunContent = newRunContent.replace(tPattern, `$1${escapedNew}$2`);
7376
8090
  // If no match, try simpler pattern
7377
8091
  if (newRunContent === targetRun.content) {
@@ -7553,7 +8267,7 @@ class HwpxDocument {
7553
8267
  return tblMatch;
7554
8268
  }
7555
8269
  let rowIndex = 0;
7556
- return tblMatch.replace(/<hp:tr[^>]*>([\s\S]*?)<\/hp:tr>/g, (rowMatch) => {
8270
+ return tblMatch.replace(/<hp:tr(?:\s[^>]*)?>([\s\S]*?)<\/hp:tr>/g, (rowMatch) => {
7557
8271
  if (rowIndex >= table.rows.length) {
7558
8272
  rowIndex++;
7559
8273
  return rowMatch;
@@ -8330,6 +9044,84 @@ class HwpxDocument {
8330
9044
  contentHpf = contentHpf.substring(0, insertPos) + newItem + contentHpf.substring(insertPos);
8331
9045
  this._zip.file('Contents/content.hpf', contentHpf);
8332
9046
  }
9047
+ /**
9048
+ * Apply section inserts/deletes to Contents/sectionN.xml, in call order.
9049
+ *
9050
+ * File numbers must keep matching memory section indices, so an insert
9051
+ * renames later files up one (section1 → section2 …) and a delete removes
9052
+ * its file and renames later files down one. content.hpf gets a manifest
9053
+ * item and a spine itemref per section, and header.xml's secCnt follows.
9054
+ */
9055
+ async applySectionOpsToZip() {
9056
+ if (!this._zip)
9057
+ return;
9058
+ const secPath = (i) => `Contents/section${i}.xml`;
9059
+ const countFiles = () => Object.keys(this._zip.files).filter(n => /^Contents\/section\d+\.xml$/.test(n)).length;
9060
+ const move = async (from, to) => {
9061
+ const f = this._zip.file(secPath(from));
9062
+ if (!f)
9063
+ return;
9064
+ this._zip.file(secPath(to), await f.async('string'));
9065
+ this._zip.remove(secPath(from));
9066
+ };
9067
+ for (const op of this._pendingSectionOps) {
9068
+ const fileCount = countFiles();
9069
+ if (op.op === 'delete') {
9070
+ if (op.at >= fileCount || fileCount <= 1)
9071
+ continue;
9072
+ this._zip.remove(secPath(op.at));
9073
+ for (let i = op.at + 1; i < fileCount; i++)
9074
+ await move(i, i - 1);
9075
+ continue;
9076
+ }
9077
+ // Insert: shift later files up, highest first.
9078
+ for (let i = fileCount - 1; i >= op.at; i--)
9079
+ await move(i, i + 1);
9080
+ // Build the new section from the template section's <hs:sec> wrapper and
9081
+ // its first paragraph's <hp:secPr> (page size, margins, numbering).
9082
+ const templateIndex = op.templateFrom >= op.at ? op.templateFrom + 1 : op.templateFrom;
9083
+ const template = await this._zip.file(secPath(templateIndex))?.async('string');
9084
+ this._zip.file(secPath(op.at), this.buildEmptySectionXml(template));
9085
+ }
9086
+ // Manifest + spine: one item per section file, in order.
9087
+ const hpfFile = this._zip.file('Contents/content.hpf');
9088
+ const total = countFiles();
9089
+ if (hpfFile) {
9090
+ let hpf = await hpfFile.async('string');
9091
+ hpf = hpf.replace(/<opf:item\b[^>]*\bid="section\d+"[^>]*\/>\s*/g, '');
9092
+ hpf = hpf.replace(/<opf:itemref\b[^>]*\bidref="section\d+"[^>]*\/>\s*/g, '');
9093
+ const items = Array.from({ length: total }, (_, i) => `<opf:item id="section${i}" href="Contents/section${i}.xml" media-type="application/xml"/>`).join('');
9094
+ const refs = Array.from({ length: total }, (_, i) => `<opf:itemref idref="section${i}" linear="yes"/>`).join('');
9095
+ hpf = hpf.replace('</opf:manifest>', items + '</opf:manifest>');
9096
+ hpf = hpf.replace('</opf:spine>', refs + '</opf:spine>');
9097
+ this._zip.file('Contents/content.hpf', hpf);
9098
+ }
9099
+ const headerFile = this._zip.file('Contents/header.xml');
9100
+ if (headerFile) {
9101
+ const header = await headerFile.async('string');
9102
+ this._zip.file('Contents/header.xml', header.replace(/\bsecCnt="\d+"/, `secCnt="${total}"`));
9103
+ }
9104
+ }
9105
+ /** A section XML holding one empty paragraph with the template's <hp:secPr>. */
9106
+ buildEmptySectionXml(template) {
9107
+ const declaration = '<?xml version="1.0" encoding="UTF-8" standalone="yes" ?>';
9108
+ const secOpen = template?.match(/<hs:sec\b[^>]*>/)?.[0]
9109
+ ?? '<hs:sec xmlns:hp="http://www.hancom.co.kr/hwpml/2011/paragraph" xmlns:hs="http://www.hancom.co.kr/hwpml/2011/section">';
9110
+ let secPr = '';
9111
+ if (template) {
9112
+ const at = template.indexOf('<hp:secPr');
9113
+ if (at !== -1)
9114
+ secPr = template.slice(at, this.findElementEnd(template, at, 'secPr'));
9115
+ }
9116
+ // A fresh column definition follows secPr in Hancom's own first paragraph.
9117
+ const colPr = template?.match(/<hp:ctrl>\s*<hp:colPr\b[^>]*\/>\s*<\/hp:ctrl>/)?.[0] ?? '';
9118
+ return `${declaration}${secOpen}` +
9119
+ `<hp:p id="0" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0">` +
9120
+ `<hp:run charPrIDRef="0">${secPr}${colPr}</hp:run>` +
9121
+ `<hp:run charPrIDRef="0"><hp:t></hp:t></hp:run>` +
9122
+ `<hp:linesegarray><hp:lineseg textpos="0" vertpos="0" vertsize="1000" textheight="1000" baseline="850" spacing="600" horzpos="0" horzsize="0" flags="393216"/></hp:linesegarray>` +
9123
+ `</hp:p></hs:sec>`;
9124
+ }
8333
9125
  /**
8334
9126
  * Add hp:pic tag to section XML
8335
9127
  */
@@ -8432,7 +9224,7 @@ class HwpxDocument {
8432
9224
  for (const table of tables) {
8433
9225
  const tableXml = xml.substring(table.startIndex, table.endIndex);
8434
9226
  // Find cells in this table
8435
- const cellMatches = [...tableXml.matchAll(/<(?:hp|hs):tc[^>]*>([\s\S]*?)<\/(?:hp|hs):tc>/g)];
9227
+ const cellMatches = [...tableXml.matchAll(/<(?:hp|hs):tc(?:\s[^>]*)?>([\s\S]*?)<\/(?:hp|hs):tc>/g)];
8436
9228
  for (const cellMatch of cellMatches) {
8437
9229
  const cellContent = cellMatch[1];
8438
9230
  const textContent = this.extractTextFromCellXml(cellContent);
@@ -8464,7 +9256,7 @@ class HwpxDocument {
8464
9256
  */
8465
9257
  findAllParagraphsInCell(cellXml) {
8466
9258
  const paragraphs = [];
8467
- const paragraphRegex = /<hp:p[^>]*>[\s\S]*?<\/hp:p>/g;
9259
+ const paragraphRegex = /<hp:p(?:\s[^>]*)?>[\s\S]*?<\/hp:p>/g;
8468
9260
  let match;
8469
9261
  while ((match = paragraphRegex.exec(cellXml)) !== null) {
8470
9262
  paragraphs.push({
@@ -8647,7 +9439,7 @@ class HwpxDocument {
8647
9439
  findTblTagIssues(xml) {
8648
9440
  const issues = [];
8649
9441
  // Track table tag positions
8650
- const tblOpenRegex = /<(?:hp|hs|hc):tbl[^>]*>/g;
9442
+ const tblOpenRegex = /<(?:hp|hs|hc):tbl(?:\s[^>]*)?>/g;
8651
9443
  const tblCloseRegex = /<\/(?:hp|hs|hc):tbl>/g;
8652
9444
  const allPositions = [];
8653
9445
  let match;
@@ -8700,11 +9492,11 @@ class HwpxDocument {
8700
9492
  checkNestingErrors(xml) {
8701
9493
  const issues = [];
8702
9494
  // Check for tc outside of tr
8703
- const tcOutsideTr = /<(?:hp|hs|hc):tc[^>]*>(?:(?!<(?:hp|hs|hc):tr[^>]*>).)*?<\/(?:hp|hs|hc):tc>/gs;
9495
+ const tcOutsideTr = /<(?:hp|hs|hc):tc(?:\s[^>]*)?>(?:(?!<(?:hp|hs|hc):tr(?:\s[^>]*)?>).)*?<\/(?:hp|hs|hc):tc>/gs;
8704
9496
  // This is simplified - a full check would need proper nesting validation
8705
9497
  // Check for tr outside of tbl
8706
- const trPattern = /<(?:hp|hs|hc):tr[^>]*>/g;
8707
- const tblPattern = /<(?:hp|hs|hc):tbl[^>]*>/g;
9498
+ const trPattern = /<(?:hp|hs|hc):tr(?:\s[^>]*)?>/g;
9499
+ const tblPattern = /<(?:hp|hs|hc):tbl(?:\s[^>]*)?>/g;
8708
9500
  // Simple check: count if tr appears without preceding tbl
8709
9501
  let match;
8710
9502
  let lastTblPos = -1;
@@ -10247,7 +11039,7 @@ class HwpxDocument {
10247
11039
  return null;
10248
11040
  const targetRowData = rows[targetRow];
10249
11041
  // Extract content inside the row (between <hp:tr...> and </hp:tr>)
10250
- const rowOpenTagMatch = targetRowData.xml.match(/^<(?:hp|hs|hc):tr[^>]*>/);
11042
+ const rowOpenTagMatch = targetRowData.xml.match(/^<(?:hp|hs|hc):tr(?:\s[^>]*)?>/);
10251
11043
  if (!rowOpenTagMatch)
10252
11044
  return null;
10253
11045
  const rowContentStart = rowOpenTagMatch[0].length;
@@ -10261,7 +11053,7 @@ class HwpxDocument {
10261
11053
  return null;
10262
11054
  const targetCellData = cells[targetCol];
10263
11055
  // Extract content inside the cell (between <hp:tc...> and </hp:tc>)
10264
- const cellOpenTagMatch = targetCellData.xml.match(/^<(?:hp|hs|hc):tc[^>]*>/);
11056
+ const cellOpenTagMatch = targetCellData.xml.match(/^<(?:hp|hs|hc):tc(?:\s[^>]*)?>/);
10265
11057
  if (!cellOpenTagMatch)
10266
11058
  return null;
10267
11059
  const cellContentStart = cellOpenTagMatch[0].length;
@@ -10284,6 +11076,219 @@ class HwpxDocument {
10284
11076
  // ============================================================
10285
11077
  // Table Row Insert/Delete XML Persistence
10286
11078
  // ============================================================
11079
+ /**
11080
+ * Scale this table's column widths so they sum to its <hp:sz width>.
11081
+ *
11082
+ * Column widths are read from cells whose colSpan is 1 (the first one seen
11083
+ * per colAddr). Every cell then gets the sum of the scaled widths of the
11084
+ * columns it spans, so merged cells stay aligned. Rounding leftovers go to
11085
+ * the last column so the total is exact. Nested tables are not touched.
11086
+ */
11087
+ fitColumnsToTableWidth(tableXml) {
11088
+ const tableWidth = parseInt(tableXml.match(/^<hp:tbl\b[\s\S]*?<hp:sz width="(\d+)"/)?.[1] ?? '', 10);
11089
+ const colCnt = parseInt(tableXml.match(/^<hp:tbl\b[^>]*\bcolCnt="(\d+)"/)?.[1] ?? '', 10);
11090
+ if (!tableWidth || !colCnt)
11091
+ return tableXml;
11092
+ const rows = this.findAllElementsWithDepth(tableXml, 'tr');
11093
+ const own = [];
11094
+ rows.forEach((row, r) => {
11095
+ for (const cell of this.findAllElementsWithDepth(row.xml, 'tc')) {
11096
+ const tail = cell.xml.lastIndexOf('</hp:subList>');
11097
+ const from = tail === -1 ? 0 : tail;
11098
+ const props = cell.xml.slice(from);
11099
+ const col = parseInt(props.match(/<hp:cellAddr\b[^>]*\bcolAddr="(\d+)"/)?.[1] ?? '-1', 10);
11100
+ const span = parseInt(props.match(/<hp:cellSpan\b[^>]*\bcolSpan="(\d+)"/)?.[1] ?? '1', 10);
11101
+ const sz = props.match(/(<hp:cellSz\b[^>]*\bwidth=")(\d+)(")/);
11102
+ if (col < 0 || !sz || sz.index === undefined)
11103
+ continue;
11104
+ own.push({ row: r, cell, col, span, width: parseInt(sz[2], 10), at: from + sz.index + sz[1].length });
11105
+ }
11106
+ });
11107
+ const widths = new Array(colCnt).fill(0);
11108
+ for (const o of own)
11109
+ if (o.span === 1 && o.col < colCnt && widths[o.col] === 0)
11110
+ widths[o.col] = o.width;
11111
+ if (widths.some(w => w === 0))
11112
+ return tableXml; // cannot derive every column safely
11113
+ const sum = widths.reduce((a, b) => a + b, 0);
11114
+ if (sum === tableWidth)
11115
+ return tableXml;
11116
+ const scaled = widths.map(w => Math.floor((w * tableWidth) / sum));
11117
+ scaled[colCnt - 1] += tableWidth - scaled.reduce((a, b) => a + b, 0);
11118
+ let out = tableXml;
11119
+ for (let r = rows.length - 1; r >= 0; r--) {
11120
+ let rowXml = rows[r].xml;
11121
+ const cellsInRow = own.filter(o => o.row === r).sort((a, b) => b.cell.startIndex - a.cell.startIndex);
11122
+ for (const o of cellsInRow) {
11123
+ const w = scaled.slice(o.col, o.col + o.span).reduce((a, b) => a + b, 0);
11124
+ const newCell = o.cell.xml.slice(0, o.at) + String(w) + o.cell.xml.slice(o.at + String(o.width).length);
11125
+ rowXml = rowXml.slice(0, o.cell.startIndex) + newCell + rowXml.slice(o.cell.endIndex);
11126
+ }
11127
+ out = out.slice(0, rows[r].startIndex) + rowXml + out.slice(rows[r].endIndex);
11128
+ }
11129
+ return out;
11130
+ }
11131
+ /**
11132
+ * Locate one of a cell's OWN address/span attributes (`colAddr`, `rowAddr`,
11133
+ * `colSpan`, `rowSpan`) in `cellXml`, returning the value and the absolute
11134
+ * index of its digits so callers can rewrite it in place.
11135
+ *
11136
+ * Hancom writes them on `<hp:cellAddr>`/`<hp:cellSpan>` after the cell's
11137
+ * sub-list (209/209 corpus files). Hand-made files may put them on the
11138
+ * `<hp:tc>` start tag instead, which the parser also accepts. A nested
11139
+ * table's cells live inside the sub-list, so only the tail is searched for
11140
+ * the child form and only the start tag for the attribute form.
11141
+ */
11142
+ cellOwnAttr(cellXml, name) {
11143
+ const child = name.endsWith('Addr') ? 'cellAddr' : 'cellSpan';
11144
+ const tail = cellXml.lastIndexOf('</hp:subList>');
11145
+ const from = tail === -1 ? 0 : tail;
11146
+ const own = new RegExp(`(<hp:${child}\\b[^>]*\\b${name}=")(\\d+)"`).exec(cellXml.slice(from));
11147
+ if (own) {
11148
+ return { value: parseInt(own[2], 10), at: from + own.index + own[1].length, length: own[2].length };
11149
+ }
11150
+ const startTag = cellXml.slice(0, cellXml.indexOf('>') + 1);
11151
+ const attr = new RegExp(`(\\s${name}=")(\\d+)"`).exec(startTag);
11152
+ if (attr) {
11153
+ return { value: parseInt(attr[2], 10), at: attr.index + attr[1].length, length: attr[2].length };
11154
+ }
11155
+ return null;
11156
+ }
11157
+ /** Rewrite one of a cell's own attributes (see cellOwnAttr); no-op if absent. */
11158
+ setCellOwnAttr(cellXml, name, value) {
11159
+ const a = this.cellOwnAttr(cellXml, name);
11160
+ return a ? cellXml.slice(0, a.at) + String(value) + cellXml.slice(a.at + a.length) : cellXml;
11161
+ }
11162
+ /** A cell's own <hp:cellSz width> (after its sub-list, so never a nested table's). */
11163
+ cellOwnWidth(cellXml) {
11164
+ const tail = cellXml.lastIndexOf('</hp:subList>');
11165
+ const m = cellXml.slice(tail === -1 ? 0 : tail).match(/<hp:cellSz\b[^>]*\bwidth="(\d+)"/);
11166
+ return m ? parseInt(m[1], 10) : null;
11167
+ }
11168
+ /** Rewrite a cell's own <hp:cellSz width>; no-op if the cell has none or width <= 0. */
11169
+ setCellOwnWidth(cellXml, width) {
11170
+ if (width <= 0)
11171
+ return cellXml;
11172
+ const tail = cellXml.lastIndexOf('</hp:subList>');
11173
+ const from = tail === -1 ? 0 : tail;
11174
+ const m = /(<hp:cellSz\b[^>]*\bwidth=")(\d+)"/.exec(cellXml.slice(from));
11175
+ if (!m)
11176
+ return cellXml;
11177
+ const at = from + m.index + m[1].length;
11178
+ return cellXml.slice(0, at) + String(width) + cellXml.slice(at + m[2].length);
11179
+ }
11180
+ /**
11181
+ * Add `delta` to the rowAddr of every cell of THIS table whose rowAddr is
11182
+ * >= fromRow. Nested tables inside cells keep their own addresses.
11183
+ */
11184
+ shiftTableRowAddrs(tableXml, fromRow, delta) {
11185
+ let out = tableXml;
11186
+ const rows = this.findAllElementsWithDepth(out, 'tr');
11187
+ for (let r = rows.length - 1; r >= 0; r--) {
11188
+ const row = rows[r];
11189
+ const cells = this.findAllElementsWithDepth(row.xml, 'tc');
11190
+ let rowXml = row.xml;
11191
+ for (let c = cells.length - 1; c >= 0; c--) {
11192
+ const cell = cells[c];
11193
+ const addr = this.cellOwnAttr(cell.xml, 'rowAddr');
11194
+ if (!addr || addr.value < fromRow)
11195
+ continue;
11196
+ const newCell = this.setCellOwnAttr(cell.xml, 'rowAddr', addr.value + delta);
11197
+ rowXml = rowXml.slice(0, cell.startIndex) + newCell + rowXml.slice(cell.endIndex);
11198
+ }
11199
+ if (rowXml !== row.xml)
11200
+ out = out.slice(0, row.startIndex) + rowXml + out.slice(row.endIndex);
11201
+ }
11202
+ return out;
11203
+ }
11204
+ /**
11205
+ * Clone a table cell for a newly inserted row: same cell attributes, same
11206
+ * first-paragraph formatting, but a single paragraph holding `text`.
11207
+ *
11208
+ * Nested tables and extra paragraphs are dropped. The first run's
11209
+ * charPrIDRef is kept so the new text matches the template cell's font.
11210
+ */
11211
+ cloneCellWithText(cellXml, text) {
11212
+ const subListOpen = cellXml.match(/<(hp|hs):subList\b[^>]*>/);
11213
+ const subListCloseIdx = cellXml.lastIndexOf('</hp:subList>') !== -1
11214
+ ? cellXml.lastIndexOf('</hp:subList>')
11215
+ : cellXml.lastIndexOf('</hs:subList>');
11216
+ if (!subListOpen || subListOpen.index === undefined || subListCloseIdx === -1) {
11217
+ // No sub-list to rebuild — fall back to blanking the text in place.
11218
+ return this.resetLinesegInXml(cellXml.replace(T_TAG_WITH_CONTENT, '<$1:t$2></$1:t>'));
11219
+ }
11220
+ const prefix = subListOpen[1];
11221
+ const inner = cellXml.slice(subListOpen.index + subListOpen[0].length, subListCloseIdx);
11222
+ const firstPara = inner.match(new RegExp(`<${prefix}:p\\b[^>]*>`));
11223
+ const paraOpen = firstPara
11224
+ ? firstPara[0]
11225
+ : `<${prefix}:p id="0" paraPrIDRef="0" styleIDRef="0" pageBreak="0" columnBreak="0" merged="0">`;
11226
+ const firstRun = inner.match(new RegExp(`<${prefix}:run\\b[^>]*charPrIDRef="(\\d+)"`));
11227
+ const charPr = firstRun ? firstRun[1] : '0';
11228
+ const paragraph = `${paraOpen}<${prefix}:run charPrIDRef="${charPr}"><${prefix}:t>${this.escapeXml(text)}</${prefix}:t></${prefix}:run>` +
11229
+ `<${prefix}:linesegarray><${prefix}:lineseg textpos="0" vertpos="0" vertsize="1000" textheight="1000" baseline="850" spacing="600" horzpos="0" horzsize="0" flags="0"/></${prefix}:linesegarray>` +
11230
+ `</${prefix}:p>`;
11231
+ return cellXml.slice(0, subListOpen.index + subListOpen[0].length) + paragraph + cellXml.slice(subListCloseIdx);
11232
+ }
11233
+ /**
11234
+ * Source cells for a new row inserted after `afterRow`, one per column
11235
+ * position, in column order, covering every column 0..colCnt-1 exactly once.
11236
+ *
11237
+ * For each column: the cell that STARTS there in the template row (keeping
11238
+ * its colSpan so horizontal merges carry over), otherwise the nearest row
11239
+ * above whose own cell starts there. A column no row starts is skipped by the
11240
+ * colSpan of the cell covering it. Returned XML still carries the source
11241
+ * addresses; the caller rewrites rowAddr/rowSpan.
11242
+ */
11243
+ gridCellsForNewRow(rows, afterRow) {
11244
+ const ownProps = (cellXml) => ({
11245
+ col: this.cellOwnAttr(cellXml, 'colAddr')?.value ?? -1,
11246
+ span: this.cellOwnAttr(cellXml, 'colSpan')?.value ?? 1,
11247
+ });
11248
+ // Cells with no address anywhere are placed by position in their row.
11249
+ const rowCells = rows.map(r => {
11250
+ let next = 0;
11251
+ return this.findAllElementsWithDepth(r.xml, 'tc').map(c => {
11252
+ const p = ownProps(c.xml);
11253
+ const col = p.col >= 0 ? p.col : next;
11254
+ next = col + p.span;
11255
+ return { xml: c.xml, col, span: p.span };
11256
+ });
11257
+ });
11258
+ const colCount = Math.max(0, ...rowCells.flat().map(c => c.col + c.span));
11259
+ const out = [];
11260
+ for (let col = 0; col < colCount;) {
11261
+ let pick;
11262
+ for (let r = afterRow; r >= 0 && !pick; r--)
11263
+ pick = rowCells[r].find(c => c.col === col);
11264
+ // Nothing above starts here (should not happen in a well-formed table):
11265
+ // fall back to any row below so the grid still has no hole.
11266
+ for (let r = afterRow + 1; r < rowCells.length && !pick; r++)
11267
+ pick = rowCells[r].find(c => c.col === col);
11268
+ if (!pick) {
11269
+ col++;
11270
+ continue;
11271
+ }
11272
+ // A cell borrowed from a row above may span columns the template row
11273
+ // splits; keep the template row's split by clamping to the next column
11274
+ // that the template row starts.
11275
+ let span = Math.max(1, pick.span);
11276
+ const nextTemplateStart = rowCells[afterRow].map(c => c.col).filter(c => c > col).sort((a, b) => a - b)[0];
11277
+ if (nextTemplateStart !== undefined && col + span > nextTemplateStart)
11278
+ span = nextTemplateStart - col;
11279
+ // Narrow through the cell's OWN attributes: a nested table's cells come
11280
+ // first in the XML, so replacing the first <hp:cellSpan> changed the
11281
+ // nested cell and left this one overlapping the next template cell.
11282
+ // Its width shrinks to the columns it still covers, so the row keeps
11283
+ // the table width.
11284
+ const xml = span === pick.span
11285
+ ? pick.xml
11286
+ : this.setCellOwnWidth(this.setCellOwnAttr(pick.xml, 'colSpan', span), Math.round((this.cellOwnWidth(pick.xml) ?? 0) * span / pick.span));
11287
+ out.push(xml);
11288
+ col += span;
11289
+ }
11290
+ return out;
11291
+ }
10287
11292
  async applyTableRowInsertsToXml() {
10288
11293
  if (!this._zip)
10289
11294
  return;
@@ -10310,29 +11315,40 @@ class HwpxDocument {
10310
11315
  if (insert.afterRowIndex >= rows.length)
10311
11316
  continue;
10312
11317
  const templateRow = rows[insert.afterRowIndex];
10313
- // Clone the template row - clear text content but preserve XML structure
10314
- let newRowXml = templateRow.xml;
10315
- // Clear text inside <hp:t> and <hs:t> tags but preserve the tags themselves
10316
- newRowXml = newRowXml.replace(/<(hp|hs):t([^>]*)>[\s\S]*?<\/\1:t>/g, '<$1:t$2></$1:t>');
10317
- // Update rowAddr in each cell
11318
+ // Build the new row from the table's COLUMN GRID, not from the template
11319
+ // row's cells. A row just below a vertical merge has no <hp:tc> for the
11320
+ // merged column (the master above covers it), so cloning its cells gave
11321
+ // the new row a hole there: colCnt=3 but only columns 1-2 present
11322
+ // (CodeRabbit, 2026-09-24). For each column position we take the cell
11323
+ // that starts there in the template row, or — if the template row has
11324
+ // none — the nearest row above that does, cloned as a single-row cell.
11325
+ //
11326
+ // Each new cell keeps its source's formatting but only its FIRST
11327
+ // paragraph, emptied: cloning every paragraph copied multi-line cells
11328
+ // (e.g. "○ a\n○ b\n- c") as three empty lines, so Hancom sized the row
11329
+ // for three lines and the one line of new text sat at the top.
10318
11330
  const newRowAddr = insert.afterRowIndex + 1;
10319
- newRowXml = newRowXml.replace(/rowAddr="(\d+)"/g, `rowAddr="${newRowAddr}"`);
10320
- // Set cell texts if provided
10321
- if (insert.cellTexts) {
10322
- let cellIdx = 0;
10323
- newRowXml = newRowXml.replace(/<(hp|hs):t([^>]*)><\/\1:t>/g, (match, prefix, attrs) => {
10324
- if (cellIdx < insert.cellTexts.length) {
10325
- const text = this.escapeXml(insert.cellTexts[cellIdx]);
10326
- cellIdx++;
10327
- return `<${prefix}:t${attrs}>${text}</${prefix}:t>`;
10328
- }
10329
- cellIdx++;
10330
- return match;
10331
- });
10332
- }
10333
- // Insert after the template row
10334
- const insertPos = templateRow.startIndex + templateRow.xml.length;
10335
- const newTableXml = tableXml.substring(0, insertPos) + '\n' + newRowXml + tableXml.substring(insertPos);
11331
+ const newRowCells = this.gridCellsForNewRow(rows, insert.afterRowIndex);
11332
+ const trOpen = templateRow.xml.slice(0, templateRow.xml.indexOf('>') + 1);
11333
+ let newRowXml = trOpen + newRowCells.map((cellXml, i) => {
11334
+ const text = insert.cellTexts?.[i] ?? '';
11335
+ // New cells sit on row afterRowIndex+1 and span one row each.
11336
+ const cell = this.setCellOwnAttr(this.cloneCellWithText(cellXml, text), 'rowAddr', newRowAddr);
11337
+ return this.setCellOwnAttr(cell, 'rowSpan', 1);
11338
+ }).join('') + '</hp:tr>';
11339
+ // Shift every existing cell below the insertion point down one row.
11340
+ // Without this the next row kept rowAddr=afterRowIndex+1 — the same as
11341
+ // the new row — and Hancom 2020 hung opening the file (reported
11342
+ // 2026-09-24; renumbering rowAddr by <hp:tr> order made it open).
11343
+ // Only the table's OWN cells are touched: a nested table in a cell has
11344
+ // its own row addresses. The delete path does the mirror of this.
11345
+ const shiftedTableXml = this.shiftTableRowAddrs(tableXml, newRowAddr, +1);
11346
+ // Insert after the template row (positions unchanged by the shift above:
11347
+ // it rewrites digits in place only after re-finding rows).
11348
+ const rowsAfterShift = this.findAllElementsWithDepth(shiftedTableXml, 'tr');
11349
+ const anchorRow = rowsAfterShift[insert.afterRowIndex];
11350
+ const insertPos = anchorRow.startIndex + anchorRow.xml.length;
11351
+ const newTableXml = shiftedTableXml.substring(0, insertPos) + '\n' + newRowXml + shiftedTableXml.substring(insertPos);
10336
11352
  // Update rowCnt attribute
10337
11353
  const updatedTableXml = newTableXml.replace(/rowCnt="(\d+)"/, (_m, cnt) => `rowCnt="${parseInt(cnt) + 1}"`);
10338
11354
  xml = xml.substring(0, tables[insert.tableIndex].startIndex) + updatedTableXml + xml.substring(tables[insert.tableIndex].endIndex);
@@ -10480,9 +11496,11 @@ class HwpxDocument {
10480
11496
  }
10481
11497
  if (!templateCell)
10482
11498
  continue;
10483
- // Clone template and clear text
11499
+ // Clone template and clear text.
11500
+ // The tag-name boundary in T_TAG_WITH_CONTENT keeps <hp:tc> structure intact.
10484
11501
  let newCellXml = templateCell.xml;
10485
- newCellXml = newCellXml.replace(/<(hp|hs):t([^>]*)>[\s\S]*?<\/\1:t>/g, '<$1:t$2></$1:t>');
11502
+ newCellXml = newCellXml.replace(T_TAG_WITH_CONTENT, '<$1:t$2></$1:t>');
11503
+ newCellXml = this.resetLinesegInXml(newCellXml);
10486
11504
  // Update colAddr to afterColIndex + 1
10487
11505
  newCellXml = newCellXml.replace(/colAddr="(\d+)"/, `colAddr="${insert.afterColIndex + 1}"`);
10488
11506
  // Also update <hp:cellAddr colAddr="..."> inside the cell
@@ -10508,6 +11526,13 @@ class HwpxDocument {
10508
11526
  }
10509
11527
  // Update colCnt
10510
11528
  tableXml = tableXml.replace(/colCnt="(\d+)"/, (_m, cnt) => `colCnt="${parseInt(cnt) + 1}"`);
11529
+ // Keep the table inside its original width. The new column cloned the
11530
+ // template column's width, so the columns summed to more than the
11531
+ // table: reported 2026-09-24, 4 × 11765 + 11765 = 58825 > body 51024
11532
+ // while <hp:sz width> still said 47060, and the table ran past the
11533
+ // right margin. Scale every column by the same factor so the total is
11534
+ // exactly the table's width again.
11535
+ tableXml = this.fitColumnsToTableWidth(tableXml);
10511
11536
  xml = xml.substring(0, tables[insert.tableIndex].startIndex) + tableXml + xml.substring(tables[insert.tableIndex].endIndex);
10512
11537
  }
10513
11538
  this._zip.file(sectionPath, xml);
@@ -10582,12 +11607,43 @@ class HwpxDocument {
10582
11607
  const prefix = prefixMatch[1];
10583
11608
  const tag = prefixMatch[2];
10584
11609
  const closeTag = `</${prefix}:${tag}>`;
10585
- // For paragraphs, find the close tag accounting for nesting
11610
+ // Paragraphs DO nest: a paragraph that holds a table contains the
11611
+ // paragraphs of every cell. Taking the first </hp:p> cut a table-wrapper
11612
+ // paragraph off inside its first cell, so anything placed "after" it
11613
+ // landed inside that cell (measured: text inserted after a table
11614
+ // appeared in the table's first cell).
10586
11615
  if (tag === 'p') {
10587
- // Paragraphs don't nest, so find the next close tag
10588
- const closeIdx = sectionXml.indexOf(closeTag, elem.start);
10589
- if (closeIdx !== -1) {
10590
- const endIndex = closeIdx + closeTag.length;
11616
+ const openTag = `<${prefix}:p`;
11617
+ let depth = 1;
11618
+ let pos = elem.start + elem.tagLength;
11619
+ let endIndex = -1;
11620
+ while (depth > 0 && pos < sectionXml.length) {
11621
+ const nextClose = sectionXml.indexOf(closeTag, pos);
11622
+ if (nextClose === -1)
11623
+ break;
11624
+ // Count only real <hp:p ...> / <hp:p> opens, not <hp:pic>, <hp:pos>, ...
11625
+ let nextOpen = sectionXml.indexOf(openTag, pos);
11626
+ while (nextOpen !== -1 && nextOpen < nextClose) {
11627
+ const after = sectionXml[nextOpen + openTag.length];
11628
+ if (after === ' ' || after === '>' || after === '/')
11629
+ break;
11630
+ nextOpen = sectionXml.indexOf(openTag, nextOpen + 1);
11631
+ }
11632
+ if (nextOpen !== -1 && nextOpen < nextClose) {
11633
+ const tagEnd = sectionXml.indexOf('>', nextOpen);
11634
+ // A self-closing <hp:p/> does not change depth.
11635
+ if (sectionXml[tagEnd - 1] !== '/')
11636
+ depth++;
11637
+ pos = tagEnd + 1;
11638
+ }
11639
+ else {
11640
+ depth--;
11641
+ pos = nextClose + closeTag.length;
11642
+ if (depth === 0)
11643
+ endIndex = pos;
11644
+ }
11645
+ }
11646
+ if (endIndex !== -1) {
10591
11647
  results.push({
10592
11648
  xml: sectionXml.substring(elem.start, endIndex),
10593
11649
  startIndex: elem.start,
@@ -10628,102 +11684,6 @@ class HwpxDocument {
10628
11684
  }
10629
11685
  return results;
10630
11686
  }
10631
- async applyParagraphCopiesToXml() {
10632
- if (!this._zip)
10633
- return;
10634
- for (const copy of this._pendingParagraphCopies) {
10635
- const srcPath = `Contents/section${copy.sourceSection}.xml`;
10636
- const srcXml = await this._zip.file(srcPath)?.async('string');
10637
- if (!srcXml)
10638
- continue;
10639
- const srcElements = this.findTopLevelFullElements(srcXml);
10640
- if (copy.sourceParagraph >= srcElements.length)
10641
- continue;
10642
- const srcElem = srcElements[copy.sourceParagraph];
10643
- if (srcElem.type !== 'p')
10644
- continue;
10645
- // Clone and regenerate ID
10646
- let clonedXml = srcElem.xml;
10647
- const newId = Math.random().toString(36).substring(2, 11);
10648
- clonedXml = clonedXml.replace(/<(hp|hs):p\s+([^>]*?)id="[^"]*"/, `<$1:p $2id="${newId}"`);
10649
- // Read target section
10650
- const tgtPath = `Contents/section${copy.targetSection}.xml`;
10651
- let tgtXml = await this._zip.file(tgtPath)?.async('string');
10652
- if (!tgtXml)
10653
- continue;
10654
- const tgtElements = this.findTopLevelFullElements(tgtXml);
10655
- // Insert after targetAfter element
10656
- let insertPos;
10657
- if (copy.targetAfter >= 0 && copy.targetAfter < tgtElements.length) {
10658
- insertPos = tgtElements[copy.targetAfter].endIndex;
10659
- }
10660
- else if (copy.targetAfter < 0) {
10661
- // Insert at beginning - find first element
10662
- if (tgtElements.length > 0) {
10663
- insertPos = tgtElements[0].startIndex;
10664
- }
10665
- else {
10666
- const secMatch = tgtXml.match(/<(?:hs|hp):sec[^>]*>/);
10667
- insertPos = secMatch ? secMatch.index + secMatch[0].length : 0;
10668
- }
10669
- }
10670
- else {
10671
- // After last element
10672
- insertPos = tgtElements.length > 0 ? tgtElements[tgtElements.length - 1].endIndex : tgtXml.lastIndexOf('</');
10673
- }
10674
- tgtXml = tgtXml.substring(0, insertPos) + '\n' + clonedXml + tgtXml.substring(insertPos);
10675
- this._zip.file(tgtPath, tgtXml);
10676
- }
10677
- }
10678
- async applyParagraphMovesToXml() {
10679
- if (!this._zip)
10680
- return;
10681
- for (const move of this._pendingParagraphMoves) {
10682
- const srcPath = `Contents/section${move.sourceSection}.xml`;
10683
- let srcXml = await this._zip.file(srcPath)?.async('string');
10684
- if (!srcXml)
10685
- continue;
10686
- const srcElements = this.findTopLevelFullElements(srcXml);
10687
- if (move.sourceParagraph >= srcElements.length)
10688
- continue;
10689
- const srcElem = srcElements[move.sourceParagraph];
10690
- if (srcElem.type !== 'p')
10691
- continue;
10692
- const extractedXml = srcElem.xml;
10693
- // Remove from source
10694
- srcXml = srcXml.substring(0, srcElem.startIndex) + srcXml.substring(srcElem.endIndex);
10695
- this._zip.file(srcPath, srcXml);
10696
- // Read target section (re-read if same section since we modified it)
10697
- const tgtPath = `Contents/section${move.targetSection}.xml`;
10698
- let tgtXml = await this._zip.file(tgtPath)?.async('string');
10699
- if (!tgtXml)
10700
- continue;
10701
- const tgtElements = this.findTopLevelFullElements(tgtXml);
10702
- // Adjust target index for same-section moves
10703
- let adjustedTarget = move.targetAfter;
10704
- if (move.sourceSection === move.targetSection && move.sourceParagraph < move.targetAfter) {
10705
- adjustedTarget -= 1;
10706
- }
10707
- let insertPos;
10708
- if (adjustedTarget >= 0 && adjustedTarget < tgtElements.length) {
10709
- insertPos = tgtElements[adjustedTarget].endIndex;
10710
- }
10711
- else if (adjustedTarget < 0) {
10712
- if (tgtElements.length > 0) {
10713
- insertPos = tgtElements[0].startIndex;
10714
- }
10715
- else {
10716
- const secMatch = tgtXml.match(/<(?:hs|hp):sec[^>]*>/);
10717
- insertPos = secMatch ? secMatch.index + secMatch[0].length : 0;
10718
- }
10719
- }
10720
- else {
10721
- insertPos = tgtElements.length > 0 ? tgtElements[tgtElements.length - 1].endIndex : tgtXml.lastIndexOf('</');
10722
- }
10723
- tgtXml = tgtXml.substring(0, insertPos) + '\n' + extractedXml + tgtXml.substring(insertPos);
10724
- this._zip.file(tgtPath, tgtXml);
10725
- }
10726
- }
10727
11687
  // ============================================================
10728
11688
  // Header/Footer XML Persistence
10729
11689
  // ============================================================
@@ -10838,6 +11798,8 @@ exports.HwpxDocument = HwpxDocument;
10838
11798
  // Constants for magic numbers
10839
11799
  HwpxDocument.NESTED_CHECK_LOOKBACK = 500;
10840
11800
  HwpxDocument.SEARCH_SKIP_OFFSET = 10;
11801
+ /** Container elements whose content belongs to OTHER paragraphs or objects. */
11802
+ HwpxDocument.NESTED_CONTENT = /<hp:(tbl|subList|equation|pic|rect|ellipse|polygon|curve|arc|line|container|drawText|textart|ole|footNote|endNote|header|footer)\b/;
10841
11803
  /**
10842
11804
  * Default chunk size for splitting long text (in characters).
10843
11805
  * Texts longer than this will be split into multiple <hp:run> elements.