@kimdayoun/hwpx-mcp 0.3.0 → 0.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -59,16 +59,34 @@ export declare class HwpxDocument {
59
59
  private _zip;
60
60
  private _content;
61
61
  private _isDirty;
62
+ /**
63
+ * True once the section element list has changed shape since the XML was
64
+ * parsed (insert/delete/copy/move of paragraphs, tables or images).
65
+ * Parsed XML offsets are unusable from that point until the next save.
66
+ */
67
+ private _structureChanged;
62
68
  private _format;
63
69
  private _undoStack;
64
70
  private _redoStack;
65
71
  private _pendingTextReplacements;
66
72
  private _pendingDirectTextUpdates;
73
+ /**
74
+ * `col` is the cell's position in the memory row; `colAddr` is its grid column.
75
+ * They differ after a merge: memory keeps covered cells, the XML drops them.
76
+ * The XML writer finds the target by colAddr so a write made after a merge
77
+ * lands in the right cell (writes now replay in call order).
78
+ */
67
79
  private _pendingTableCellUpdates;
68
80
  private _pendingNestedTableInserts;
69
81
  private _pendingImageInserts;
70
82
  private _pendingCellImageInserts;
71
83
  private _pendingTableInserts;
84
+ /**
85
+ * Monotonic counter shared by paragraph and table inserts. Both kinds are
86
+ * replayed into XML in this order so each insert sees exactly the elements
87
+ * that existed when it was made. Replaying all tables before all paragraphs
88
+ * wrote "A, table, A-2, table" to disk as "A, A-2, table, table".
89
+ */
72
90
  private _tableInsertCounter;
73
91
  private _pendingImageDeletes;
74
92
  private _pendingTableDeletes;
@@ -84,10 +102,25 @@ export declare class HwpxDocument {
84
102
  private _pendingTableRowDeletes;
85
103
  private _pendingTableColumnInserts;
86
104
  private _pendingTableColumnDeletes;
105
+ /**
106
+ * Call order of every pending edit that names a table cell or row/column by
107
+ * index. Each such index is relative to the table as it was at call time,
108
+ * so save must replay these edits in call order (applyTableOpsInCallOrder).
109
+ * A WeakMap keeps the queue element types unchanged and drops entries with
110
+ * their ops (undo, section delete).
111
+ */
112
+ private _tableOpSeq;
113
+ private _tableOpCounter;
87
114
  private _pendingParagraphCopies;
88
115
  private _pendingParagraphMoves;
89
116
  private _pendingHeaderUpdates;
90
117
  private _pendingFooterUpdates;
118
+ /**
119
+ * New sections to materialise as Contents/sectionN.xml on save, in call
120
+ * order. `templateFrom` is the section whose <hp:secPr> (page size, margins)
121
+ * the new section copies — Hancom's own "insert section" does the same.
122
+ */
123
+ private _pendingSectionOps;
91
124
  private _charPrCache;
92
125
  private _originalCharPrCount?;
93
126
  private constructor();
@@ -107,10 +140,19 @@ export declare class HwpxDocument {
107
140
  static createNew(id: string, title?: string, creator?: string): HwpxDocument;
108
141
  get id(): string;
109
142
  get path(): string;
143
+ /** True once the document has a real location on disk. */
144
+ get hasPath(): boolean;
145
+ /**
146
+ * Record where the document now lives after a successful write, so the next
147
+ * save without an explicit path targets the same file.
148
+ */
149
+ setPath(newPath: string): void;
110
150
  get format(): DocumentFormat;
111
151
  get isDirty(): boolean;
112
152
  get zip(): JSZip | null;
113
153
  get content(): HwpxContent;
154
+ /** Push a table-structure or table-cell edit and remember its call order. */
155
+ private queueTableOp;
114
156
  private saveState;
115
157
  private serializeContent;
116
158
  private deserializeContent;
@@ -128,6 +170,32 @@ export declare class HwpxDocument {
128
170
  * Call this after any modification that changes document structure or content.
129
171
  */
130
172
  private markModified;
173
+ /**
174
+ * Record that the section element list changed shape. Call from every method
175
+ * that inserts, removes, copies or moves a section-level element. Once set,
176
+ * paragraph edits stop trusting offsets cached at parse time and locate their
177
+ * target in the current XML instead.
178
+ */
179
+ private markStructureChanged;
180
+ /**
181
+ * Resolve "after element N" into an id-based anchor using the memory model
182
+ * as it is right now (before the new element is spliced in).
183
+ *
184
+ * Returns null for "before everything" (N < 0). Elements without an XML
185
+ * paragraph/table of their own (images, shapes) are skipped backwards to the
186
+ * nearest paragraph or table, which is what the XML placement needs.
187
+ *
188
+ * The parser turns a paragraph that is only a line of ─/━/═ into an 'hr'
189
+ * element with a fresh id. That paragraph is still in the XML, so it still
190
+ * takes an occurrence slot there: skipping it here put later anchors one
191
+ * paragraph early (measured on Hancom files with divider lines).
192
+ */
193
+ private resolveElementAnchor;
194
+ /**
195
+ * The XML node a memory element stands for, or null if it has none of its
196
+ * own. An 'hr' parsed from a divider paragraph stands for that paragraph.
197
+ */
198
+ private anchorKeyOf;
131
199
  getSerializableContent(): object;
132
200
  getAllText(): string;
133
201
  getStructure(): object;
@@ -137,11 +205,15 @@ export declare class HwpxDocument {
137
205
  index: number;
138
206
  text: string;
139
207
  style?: ParagraphStyle;
208
+ paraPrIDRef?: number;
209
+ charPrIDRef?: number;
140
210
  }>;
141
211
  getParagraph(sectionIndex: number, paragraphIndex: number): {
142
212
  text: string;
143
213
  runs: TextRun[];
144
214
  style?: ParagraphStyle;
215
+ paraPrIDRef?: number;
216
+ charPrIDRef?: number;
145
217
  } | null;
146
218
  updateParagraphText(sectionIndex: number, elementIndex: number, runIndex: number, text: string): void;
147
219
  updateParagraphRuns(sectionIndex: number, elementIndex: number, runs: TextRun[]): void;
@@ -304,9 +376,19 @@ export declare class HwpxDocument {
304
376
  /**
305
377
  * Get table map with headers - maps table indices to their header paragraphs
306
378
  * Returns array of table info including the header text from the preceding paragraph
379
+ *
380
+ * Two indices are returned because they differ once a document has more than
381
+ * one section:
382
+ * - `table_index_in_section` — what every table tool (update_table_cell,
383
+ * get_table_cell, insert_table_row, …) expects together with
384
+ * `section_index`. Use this one.
385
+ * - `table_index` — position across the whole document, kept for callers
386
+ * that list tables. Passing it to a table tool in section 1+ addresses a
387
+ * DIFFERENT table (reported 2026-09-24: map said 5, the tool needed 4).
307
388
  */
308
389
  getTableMap(): Array<{
309
390
  table_index: number;
391
+ table_index_in_section: number;
310
392
  section_index: number;
311
393
  header: string;
312
394
  rows: number;
@@ -433,6 +515,12 @@ export declare class HwpxDocument {
433
515
  col: number;
434
516
  value: string;
435
517
  }>;
518
+ failed: Array<{
519
+ row: number;
520
+ col: number;
521
+ value: string;
522
+ error: string;
523
+ }>;
436
524
  updated: Array<{
437
525
  row: number;
438
526
  col: number;
@@ -568,6 +656,12 @@ export declare class HwpxDocument {
568
656
  text: string;
569
657
  cell: TableCell;
570
658
  } | null;
659
+ /**
660
+ * Find the merged cell that covers (row, col), if that position is not itself
661
+ * a master cell. A covered cell has no <hp:tc> of its own in the saved XML,
662
+ * so writing to it succeeds in memory and then silently vanishes on save.
663
+ */
664
+ private findCoveringMergedCell;
571
665
  updateTableCell(sectionIndex: number, tableIndex: number, row: number, col: number, text: string, charShapeId?: number): boolean;
572
666
  setCellProperties(sectionIndex: number, tableIndex: number, row: number, col: number, props: Partial<TableCell>): boolean;
573
667
  insertTableRow(sectionIndex: number, tableIndex: number, afterRowIndex: number, cellTexts?: string[]): boolean;
@@ -832,6 +926,12 @@ export declare class HwpxDocument {
832
926
  pageSettings: PageSettings;
833
927
  }[];
834
928
  insertSection(afterSectionIndex: number): number;
929
+ /**
930
+ * Add `delta` to every section number held by a pending edit that is >= from.
931
+ * Covers all pending arrays generically: any numeric field whose name is
932
+ * sectionIndex or ends in "Section"/"SectionIndex" (source/target pairs).
933
+ */
934
+ private shiftPendingSectionIndices;
835
935
  deleteSection(sectionIndex: number): boolean;
836
936
  getStyles(): {
837
937
  id: number;
@@ -880,10 +980,68 @@ export declare class HwpxDocument {
880
980
  */
881
981
  private applyParagraphDeletesToXml;
882
982
  /**
883
- * Apply table inserts to XML.
884
- * Inserts new tables into the section XML.
983
+ * Find the end offset of the section-level element an insert anchors to.
984
+ *
985
+ * Paragraphs are matched by their own <hp:p id>. Tables are matched by
986
+ * <hp:tbl id>, either wrapped in a paragraph (the insert goes after that
987
+ * paragraph) or placed directly in the section (move_table writes them that
988
+ * way).
989
+ *
990
+ * The returned offset is always the end of a TOP-LEVEL element, because that
991
+ * is the only place a new section-level element may go. But paragraph
992
+ * occurrences are counted over exactly the paragraphs the parser puts in the
993
+ * memory model (see parsedParagraphStarts), so they agree with the occurrence
994
+ * resolveElementAnchor recorded. The parser also lifts paragraphs out of
995
+ * headers, text boxes and shapes; Hancom reuses id="0" / id="2147483648"
996
+ * there too. Counting only top-level paragraphs put 47 of 131 sampled Hancom
997
+ * files' copies and moves on the wrong paragraph.
998
+ *
999
+ * Returns -1 if the anchor is not present in the current XML.
1000
+ */
1001
+ private findAnchorEnd;
1002
+ /**
1003
+ * The exact XML range of the memory paragraph a paragraph anchor names,
1004
+ * found by id + occurrence among parsedParagraphStarts. For a paragraph in a
1005
+ * header or text box this is that paragraph alone, not its container.
885
1006
  */
886
- private applyTableInsertsToXml;
1007
+ private findParsedParagraph;
1008
+ /**
1009
+ * Start offsets (in `xml`) of the paragraphs HwpxParser turns into memory
1010
+ * paragraphs, in document order. Mirrors HwpxParser.parseSection:
1011
+ *
1012
+ * - MEMO fields, footnotes and endnotes are ignored;
1013
+ * - paragraphs inside any table are skipped;
1014
+ * - a paragraph that holds a table is kept only if it still has <hp:t>
1015
+ * once its tables are removed.
1016
+ *
1017
+ * Offsets are mapped back to the original XML, so callers can slice it.
1018
+ */
1019
+ private parsedParagraphStarts;
1020
+ /** Every <hp:tbl> range at any depth (outer tables before their nested ones). */
1021
+ private findAllTablesDeep;
1022
+ /** End offset of the paragraph opening at `start`, counting nested <hp:p>. */
1023
+ private findBalancedParagraphEnd;
1024
+ /** True if this paragraph directly (not via a nested table) holds <hp:tbl id>. */
1025
+ private wrapsTopLevelTable;
1026
+ /**
1027
+ * Offset for an insert with no anchor ("before everything"). The first
1028
+ * paragraph carries <hp:secPr> (page and section settings) and must stay
1029
+ * first, so new content goes right after it.
1030
+ */
1031
+ private findSectionHeadEnd;
1032
+ /** Build the XML for a table inserted by insertTable, wrapped in its own paragraph. */
1033
+ private buildInsertedTableXml;
1034
+ /**
1035
+ * Replay structural edits — paragraph/table inserts and paragraph
1036
+ * copies/moves — into the section XML in the order the calls were made.
1037
+ *
1038
+ * Every edit carries id-based anchors resolved at call time, so it lands
1039
+ * after the same element in the XML that it followed in memory. Replaying in
1040
+ * call order means each anchor already exists (or has already moved) by the
1041
+ * time a later edit needs it. Copies/moves may cross sections, so all
1042
+ * touched sections are held in memory and written once at the end.
1043
+ */
1044
+ private applyStructuralInsertsToXml;
887
1045
  /**
888
1046
  * Apply table move/copy operations to XML.
889
1047
  * Extracts table XML from source and inserts at target position.
@@ -898,11 +1056,6 @@ export declare class HwpxDocument {
898
1056
  * Returns the position after the closing tag of the element.
899
1057
  */
900
1058
  private findInsertPositionForElement;
901
- /**
902
- * Apply paragraph inserts to XML.
903
- * Inserts new paragraphs at the specified positions.
904
- */
905
- private applyParagraphInsertsToXml;
906
1059
  /**
907
1060
  * Apply nested table inserts to XML.
908
1061
  * Inserts a new table inside a cell of an existing table.
@@ -911,6 +1064,24 @@ export declare class HwpxDocument {
911
1064
  /**
912
1065
  * Generate XML for a nested table.
913
1066
  */
1067
+ /**
1068
+ * Usable width inside a table cell, in hwpunit.
1069
+ *
1070
+ * A cell with hasMargin="0" takes its padding from the table's inMargin, so
1071
+ * the cell's own cellMargin is only authoritative when hasMargin="1".
1072
+ *
1073
+ * Every lookup is scoped to the cell's (or table's) own markup. A nested
1074
+ * table inside the cell carries its own cellSz/cellMargin/inMargin, and a
1075
+ * first-match regex over the whole cell would read those instead — which
1076
+ * sized a second nested table to the first one's column (7086 vs 21260).
1077
+ */
1078
+ private getCellInnerWidth;
1079
+ /**
1080
+ * Generate XML for a nested table.
1081
+ *
1082
+ * @param innerWidth Usable width of the parent cell in hwpunit. The nested
1083
+ * table is sized to fit it exactly; columns share the width evenly.
1084
+ */
914
1085
  private generateNestedTableXml;
915
1086
  /**
916
1087
  * Insert a nested table XML into a cell XML.
@@ -942,6 +1113,19 @@ export declare class HwpxDocument {
942
1113
  * Generate an empty cell XML for split operations.
943
1114
  */
944
1115
  private generateEmptyCell;
1116
+ /**
1117
+ * Replay every pending table edit (cell text, merge/split, nested table,
1118
+ * cell image, cell hanging indent, row/column insert/delete) in call order.
1119
+ *
1120
+ * Each index an edit carries is relative to the table as it was when the
1121
+ * edit was made. Applying by kind (all cell writes, then all row inserts,
1122
+ * then all column inserts ...) wrote text into the pre-insert layout; and
1123
+ * the row appliers sort their own queue by index, which reorders two
1124
+ * inserts or two deletes on the same table. So edits that change a table's
1125
+ * row/column layout run one at a time. Runs of layout-preserving edits
1126
+ * (cell text, indents, images, nested tables) go to their applier together.
1127
+ */
1128
+ private applyTableOpsInCallOrder;
945
1129
  /**
946
1130
  * Apply table cell updates to XML while preserving original structure.
947
1131
  * This function modifies only the text content of specific cells,
@@ -1063,8 +1247,10 @@ export declare class HwpxDocument {
1063
1247
  */
1064
1248
  private replaceMultipleRunsInElement;
1065
1249
  /**
1066
- * Calculate the occurrence index for a paragraph with given ID.
1067
- * Returns how many paragraphs with the same ID appear before this one.
1250
+ * Occurrence index of the paragraph at `elementIndex` among paragraphs with
1251
+ * the same id — counted with the same rule as insert anchors
1252
+ * (resolveElementAnchor), so a text update finds the paragraph that
1253
+ * findParsedParagraph resolves. Divider paragraphs parsed as 'hr' count.
1068
1254
  */
1069
1255
  private getParagraphOccurrence;
1070
1256
  /**
@@ -1096,6 +1282,23 @@ export declare class HwpxDocument {
1096
1282
  * Finds hp:run elements and updates their hp:t content.
1097
1283
  */
1098
1284
  private replaceRunsInParagraphDirect;
1285
+ /** Container elements whose content belongs to OTHER paragraphs or objects. */
1286
+ private static readonly NESTED_CONTENT;
1287
+ /** Direct <hp:run> children of a paragraph (runs of nested paragraphs excluded). */
1288
+ private findDirectChildRuns;
1289
+ /**
1290
+ * A run's own markup with every nested container (table, equation, text box,
1291
+ * note…) blanked out, so its <hp:t> are the run's own text only.
1292
+ */
1293
+ private ownRunText;
1294
+ /**
1295
+ * Apply `fn` to the parts of a run that are its own text, leaving nested
1296
+ * containers byte-for-byte intact. With `blank`, nested containers are
1297
+ * replaced by an empty marker instead (for reading).
1298
+ */
1299
+ private mapOwnRunText;
1300
+ /** End offset of the <hp:name> element opening at `start` (handles nesting and self-closing). */
1301
+ private findElementEnd;
1099
1302
  /**
1100
1303
  * Replace text in a single run directly using pre-computed target location.
1101
1304
  * Simpler version for single-run updates.
@@ -1277,6 +1480,17 @@ export declare class HwpxDocument {
1277
1480
  * Add image entry to content.hpf manifest
1278
1481
  */
1279
1482
  private addImageToContentHpf;
1483
+ /**
1484
+ * Apply section inserts/deletes to Contents/sectionN.xml, in call order.
1485
+ *
1486
+ * File numbers must keep matching memory section indices, so an insert
1487
+ * renames later files up one (section1 → section2 …) and a delete removes
1488
+ * its file and renames later files down one. content.hpf gets a manifest
1489
+ * item and a spine itemref per section, and header.xml's secCnt follows.
1490
+ */
1491
+ private applySectionOpsToZip;
1492
+ /** A section XML holding one empty paragraph with the template's <hp:secPr>. */
1493
+ private buildEmptySectionXml;
1280
1494
  /**
1281
1495
  * Add hp:pic tag to section XML
1282
1496
  */
@@ -1591,6 +1805,57 @@ export declare class HwpxDocument {
1591
1805
  * @returns Cell XML content and its position, or null if not found
1592
1806
  */
1593
1807
  private findTableCellInXml;
1808
+ /**
1809
+ * Scale this table's column widths so they sum to its <hp:sz width>.
1810
+ *
1811
+ * Column widths are read from cells whose colSpan is 1 (the first one seen
1812
+ * per colAddr). Every cell then gets the sum of the scaled widths of the
1813
+ * columns it spans, so merged cells stay aligned. Rounding leftovers go to
1814
+ * the last column so the total is exact. Nested tables are not touched.
1815
+ */
1816
+ private fitColumnsToTableWidth;
1817
+ /**
1818
+ * Locate one of a cell's OWN address/span attributes (`colAddr`, `rowAddr`,
1819
+ * `colSpan`, `rowSpan`) in `cellXml`, returning the value and the absolute
1820
+ * index of its digits so callers can rewrite it in place.
1821
+ *
1822
+ * Hancom writes them on `<hp:cellAddr>`/`<hp:cellSpan>` after the cell's
1823
+ * sub-list (209/209 corpus files). Hand-made files may put them on the
1824
+ * `<hp:tc>` start tag instead, which the parser also accepts. A nested
1825
+ * table's cells live inside the sub-list, so only the tail is searched for
1826
+ * the child form and only the start tag for the attribute form.
1827
+ */
1828
+ private cellOwnAttr;
1829
+ /** Rewrite one of a cell's own attributes (see cellOwnAttr); no-op if absent. */
1830
+ private setCellOwnAttr;
1831
+ /** A cell's own <hp:cellSz width> (after its sub-list, so never a nested table's). */
1832
+ private cellOwnWidth;
1833
+ /** Rewrite a cell's own <hp:cellSz width>; no-op if the cell has none or width <= 0. */
1834
+ private setCellOwnWidth;
1835
+ /**
1836
+ * Add `delta` to the rowAddr of every cell of THIS table whose rowAddr is
1837
+ * >= fromRow. Nested tables inside cells keep their own addresses.
1838
+ */
1839
+ private shiftTableRowAddrs;
1840
+ /**
1841
+ * Clone a table cell for a newly inserted row: same cell attributes, same
1842
+ * first-paragraph formatting, but a single paragraph holding `text`.
1843
+ *
1844
+ * Nested tables and extra paragraphs are dropped. The first run's
1845
+ * charPrIDRef is kept so the new text matches the template cell's font.
1846
+ */
1847
+ private cloneCellWithText;
1848
+ /**
1849
+ * Source cells for a new row inserted after `afterRow`, one per column
1850
+ * position, in column order, covering every column 0..colCnt-1 exactly once.
1851
+ *
1852
+ * For each column: the cell that STARTS there in the template row (keeping
1853
+ * its colSpan so horizontal merges carry over), otherwise the nearest row
1854
+ * above whose own cell starts there. A column no row starts is skipped by the
1855
+ * colSpan of the cell covering it. Returned XML still carries the source
1856
+ * addresses; the caller rewrites rowAddr/rowSpan.
1857
+ */
1858
+ private gridCellsForNewRow;
1594
1859
  private applyTableRowInsertsToXml;
1595
1860
  private applyTableRowDeletesToXml;
1596
1861
  private applyTableColumnInsertsToXml;
@@ -1600,8 +1865,6 @@ export declare class HwpxDocument {
1600
1865
  * Returns elements with their complete XML (including closing tags).
1601
1866
  */
1602
1867
  private findTopLevelFullElements;
1603
- private applyParagraphCopiesToXml;
1604
- private applyParagraphMovesToXml;
1605
1868
  private applyHeaderFooterUpdatesToXml;
1606
1869
  }
1607
1870
  export {};