kordoc 4.0.8 → 4.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/README.md +50 -16
  2. package/dist/{-LE4RLXVA.js → -SXQIZHYA.js} +24 -10
  3. package/dist/{parser-TCTCSBYZ.js → chunk-26AJJCFF.js} +1427 -2027
  4. package/dist/chunk-26AJJCFF.js.map +1 -0
  5. package/dist/{parser-KMMUS73Q.js → chunk-4JPPLIIV.js} +1425 -2025
  6. package/dist/chunk-4JPPLIIV.js.map +1 -0
  7. package/dist/chunk-4VNLZ2GZ.js +13 -0
  8. package/dist/chunk-4VNLZ2GZ.js.map +1 -0
  9. package/dist/{chunk-2DZQF6YJ.js → chunk-6SHO3BNV.js} +24 -15
  10. package/dist/chunk-6SHO3BNV.js.map +1 -0
  11. package/dist/chunk-7FCNOKOR.js +144 -0
  12. package/dist/chunk-7FCNOKOR.js.map +1 -0
  13. package/dist/{chunk-6XAAYAWL.cjs → chunk-CMN3SYZ3.js} +62 -304
  14. package/dist/chunk-CMN3SYZ3.js.map +1 -0
  15. package/dist/chunk-DZIXKL7E.js +145 -0
  16. package/dist/chunk-DZIXKL7E.js.map +1 -0
  17. package/dist/chunk-ELOV2GIJ.js +82 -0
  18. package/dist/chunk-ELOV2GIJ.js.map +1 -0
  19. package/dist/chunk-F2HYH62H.cjs +130 -0
  20. package/dist/chunk-F2HYH62H.cjs.map +1 -0
  21. package/dist/{chunk-HMUEU7D7.js → chunk-FSXLSHCW.js} +3 -3
  22. package/dist/{chunk-HMUEU7D7.js.map → chunk-FSXLSHCW.js.map} +1 -1
  23. package/dist/chunk-GEPINOL6.js +83 -0
  24. package/dist/chunk-GEPINOL6.js.map +1 -0
  25. package/dist/chunk-H7OWEYH2.cjs +243 -0
  26. package/dist/chunk-H7OWEYH2.cjs.map +1 -0
  27. package/dist/{chunk-7S3M4N4E.js → chunk-IGGRLWT5.js} +683 -226
  28. package/dist/chunk-IGGRLWT5.js.map +1 -0
  29. package/dist/chunk-J2RZ444H.js +243 -0
  30. package/dist/chunk-J2RZ444H.js.map +1 -0
  31. package/dist/chunk-JP5YH3Z4.js +130 -0
  32. package/dist/chunk-JP5YH3Z4.js.map +1 -0
  33. package/dist/{parser-ITMU2WN5.cjs → chunk-LHROM2BO.cjs} +1409 -2012
  34. package/dist/chunk-LHROM2BO.cjs.map +1 -0
  35. package/dist/chunk-NNDVFNMJ.js +246 -0
  36. package/dist/chunk-NNDVFNMJ.js.map +1 -0
  37. package/dist/{chunk-RROH5WHO.js → chunk-PLIY23LY.js} +2 -2
  38. package/dist/{chunk-BNU5QGIZ.js → chunk-V6U6YS5B.js} +12 -4
  39. package/dist/chunk-V6U6YS5B.js.map +1 -0
  40. package/dist/{chunk-UEIYETCQ.js → chunk-WF2E5P2G.js} +40 -35
  41. package/dist/chunk-WF2E5P2G.js.map +1 -0
  42. package/dist/{chunk-D35VBACN.js → chunk-WQLHPOMJ.js} +7 -4
  43. package/dist/chunk-WQLHPOMJ.js.map +1 -0
  44. package/dist/{chunk-OQPILS7B.js → chunk-XTCT6KQ5.cjs} +79 -183
  45. package/dist/chunk-XTCT6KQ5.cjs.map +1 -0
  46. package/dist/{chunk-MEPHGCPQ.js → chunk-YOO6ET6M.js} +7 -3
  47. package/dist/chunk-YOO6ET6M.js.map +1 -0
  48. package/dist/chunks-ZQEFAZNC.js +10 -0
  49. package/dist/cli.js +228 -36
  50. package/dist/cli.js.map +1 -1
  51. package/dist/{detect-RI2MQ33K.js → detect-YEZX2XCF.js} +2 -2
  52. package/dist/{formula-RXVSQPXI.js → formula-KSNAEBZ2.js} +23 -232
  53. package/dist/formula-KSNAEBZ2.js.map +1 -0
  54. package/dist/{formula-5NKVS2LR.cjs → formula-SB3NHDUM.cjs} +30 -240
  55. package/dist/formula-SB3NHDUM.cjs.map +1 -0
  56. package/dist/{formula-JCNF43NE.js → formula-SXC6LRRC.js} +23 -233
  57. package/dist/formula-SXC6LRRC.js.map +1 -0
  58. package/dist/index.cjs +1280 -589
  59. package/dist/index.cjs.map +1 -1
  60. package/dist/index.d.cts +112 -4
  61. package/dist/index.d.ts +112 -4
  62. package/dist/index.js +929 -238
  63. package/dist/index.js.map +1 -1
  64. package/dist/mcp.js +278 -68
  65. package/dist/mcp.js.map +1 -1
  66. package/dist/models-KA4GLQK5.js +25 -0
  67. package/dist/parser-ND2YGQZR.js +800 -0
  68. package/dist/parser-ND2YGQZR.js.map +1 -0
  69. package/dist/parser-RBZHA6SY.js +804 -0
  70. package/dist/parser-RBZHA6SY.js.map +1 -0
  71. package/dist/parser-WDZOMDG4.cjs +803 -0
  72. package/dist/parser-WDZOMDG4.cjs.map +1 -0
  73. package/dist/pdf-ocr-2Z7VZIFI.cjs +439 -0
  74. package/dist/pdf-ocr-2Z7VZIFI.cjs.map +1 -0
  75. package/dist/pdf-ocr-FJFXFADQ.js +438 -0
  76. package/dist/pdf-ocr-FJFXFADQ.js.map +1 -0
  77. package/dist/pdf-ocr-VYE4DHD4.js +386 -0
  78. package/dist/pdf-ocr-VYE4DHD4.js.map +1 -0
  79. package/dist/{profile-io-SLN4P76T.js → profile-io-TWHDIVPL.js} +3 -3
  80. package/dist/rasterize-W2NEZUSW.js +40 -0
  81. package/dist/rasterize-W2NEZUSW.js.map +1 -0
  82. package/dist/redact-7ILEAUTS.js +12 -0
  83. package/dist/redact-7ILEAUTS.js.map +1 -0
  84. package/dist/render-4MG5QUDG.js +10 -0
  85. package/dist/render-4MG5QUDG.js.map +1 -0
  86. package/dist/seal-D2QNXJVO.js +10 -0
  87. package/dist/seal-D2QNXJVO.js.map +1 -0
  88. package/dist/{setup-57FB3LSP.js → setup-Q2PRE7UA.js} +5 -3
  89. package/dist/setup-Q2PRE7UA.js.map +1 -0
  90. package/dist/{watch-OHQWSVPE.js → watch-VWXJZT2R.js} +78 -23
  91. package/dist/watch-VWXJZT2R.js.map +1 -0
  92. package/package.json +2 -1
  93. package/dist/chunk-2DZQF6YJ.js.map +0 -1
  94. package/dist/chunk-6XAAYAWL.cjs.map +0 -1
  95. package/dist/chunk-7S3M4N4E.js.map +0 -1
  96. package/dist/chunk-BNU5QGIZ.js.map +0 -1
  97. package/dist/chunk-D35VBACN.js.map +0 -1
  98. package/dist/chunk-MEPHGCPQ.js.map +0 -1
  99. package/dist/chunk-OQPILS7B.js.map +0 -1
  100. package/dist/chunk-UEIYETCQ.js.map +0 -1
  101. package/dist/formula-5NKVS2LR.cjs.map +0 -1
  102. package/dist/formula-JCNF43NE.js.map +0 -1
  103. package/dist/formula-RXVSQPXI.js.map +0 -1
  104. package/dist/parser-ITMU2WN5.cjs.map +0 -1
  105. package/dist/parser-KMMUS73Q.js.map +0 -1
  106. package/dist/parser-TCTCSBYZ.js.map +0 -1
  107. package/dist/provider-4ZJKV3DC.js +0 -37
  108. package/dist/provider-4ZJKV3DC.js.map +0 -1
  109. package/dist/provider-AKROB7WQ.js +0 -39
  110. package/dist/provider-AKROB7WQ.js.map +0 -1
  111. package/dist/provider-G4C2V2PD.cjs +0 -39
  112. package/dist/provider-G4C2V2PD.cjs.map +0 -1
  113. package/dist/render-25BT623I.js +0 -10
  114. package/dist/seal-R4TETA4C.js +0 -10
  115. package/dist/setup-57FB3LSP.js.map +0 -1
  116. package/dist/watch-OHQWSVPE.js.map +0 -1
  117. /package/dist/{-LE4RLXVA.js.map → -SXQIZHYA.js.map} +0 -0
  118. /package/dist/{chunk-RROH5WHO.js.map → chunk-PLIY23LY.js.map} +0 -0
  119. /package/dist/{detect-RI2MQ33K.js.map → chunks-ZQEFAZNC.js.map} +0 -0
  120. /package/dist/{profile-io-SLN4P76T.js.map → detect-YEZX2XCF.js.map} +0 -0
  121. /package/dist/{render-25BT623I.js.map → models-KA4GLQK5.js.map} +0 -0
  122. /package/dist/{seal-R4TETA4C.js.map → profile-io-TWHDIVPL.js.map} +0 -0
@@ -1,24 +1,18 @@
1
1
  #!/usr/bin/env node
2
+ import {
3
+ inlineImagesIntoMarkdown
4
+ } from "./chunk-7FCNOKOR.js";
2
5
  import {
3
6
  HEADING_RATIO_H1,
4
7
  HEADING_RATIO_H2,
5
- HEADING_RATIO_H3,
6
- MAX_COLS,
7
- MAX_ROWS,
8
- blocksToMarkdown,
9
- buildTable,
10
- convertTableToText,
11
- dedupeRunningHeaders,
12
- flattenLayoutTables,
13
- inlineImagesIntoMarkdown,
14
- mapPuaText
15
- } from "./chunk-UEIYETCQ.js";
8
+ HEADING_RATIO_H3
9
+ } from "./chunk-4VNLZ2GZ.js";
16
10
  import {
17
11
  detectFormat,
18
12
  detectOle2Format,
19
13
  detectZipFormat,
20
14
  parseLenientCfb
21
- } from "./chunk-MEPHGCPQ.js";
15
+ } from "./chunk-YOO6ET6M.js";
22
16
  import {
23
17
  parsePageRange
24
18
  } from "./chunk-MOL7MDBG.js";
@@ -33,25 +27,36 @@ import {
33
27
  paraTextPureT,
34
28
  patchZipEntries,
35
29
  scanSectionXml
36
- } from "./chunk-D35VBACN.js";
30
+ } from "./chunk-WQLHPOMJ.js";
31
+ import {
32
+ MAX_COLS,
33
+ MAX_ROWS,
34
+ blocksToMarkdown,
35
+ buildTable,
36
+ convertTableToText,
37
+ dedupeRunningHeaders,
38
+ flattenLayoutTables,
39
+ mapPuaText
40
+ } from "./chunk-CMN3SYZ3.js";
37
41
  import {
38
42
  SPACE_EM_FIXED,
39
43
  charWidthEm1000,
40
44
  fitRatioForFewerLines,
41
45
  measureTextWidth,
42
46
  simulateWrap
43
- } from "./chunk-2DZQF6YJ.js";
47
+ } from "./chunk-6SHO3BNV.js";
44
48
  import {
45
49
  MAX_DECOMPRESS_SIZE,
46
50
  MAX_XML_DEPTH,
47
51
  MAX_ZIP_ENTRIES,
52
+ ZipBombError,
48
53
  applyPageText,
49
54
  clampSpan,
50
55
  createSectionShared,
51
56
  createXmlParser,
52
57
  extractTextFromNode,
53
58
  findChildByLocalName
54
- } from "./chunk-BNU5QGIZ.js";
59
+ } from "./chunk-V6U6YS5B.js";
55
60
  import {
56
61
  KordocError,
57
62
  classifyError,
@@ -59,10 +64,11 @@ import {
59
64
  isPathTraversal,
60
65
  normalizeSectionHref,
61
66
  precheckZipSize,
67
+ sanitizeError,
62
68
  sanitizeHref,
63
69
  stripDtd,
64
70
  toArrayBuffer
65
- } from "./chunk-HMUEU7D7.js";
71
+ } from "./chunk-FSXLSHCW.js";
66
72
 
67
73
  // src/index.ts
68
74
  import { readFile } from "fs/promises";
@@ -1277,6 +1283,7 @@ function toRoman(n) {
1277
1283
  return out;
1278
1284
  }
1279
1285
  function formatHeadNumber(n, numFormat) {
1286
+ if (n === 0 && numFormat === "DIGIT") return "0";
1280
1287
  if (n <= 0) n = 1;
1281
1288
  switch (numFormat) {
1282
1289
  case "DIGIT":
@@ -1334,25 +1341,25 @@ function resolveParaHeading(paraEl, ctx) {
1334
1341
  if (!numDef) return headingLevel ? { headingLevel } : null;
1335
1342
  let counters = ctx.shared.numState.get(numId);
1336
1343
  if (!counters) {
1337
- counters = new Array(11).fill(0);
1344
+ counters = new Array(11).fill(-1);
1338
1345
  ctx.shared.numState.set(numId, counters);
1339
1346
  }
1340
1347
  const head = numDef.heads.get(level);
1341
- counters[level] = counters[level] === 0 ? head?.start ?? 1 : counters[level] + 1;
1342
- for (let l = level + 1; l <= 10; l++) counters[l] = 0;
1348
+ counters[level] = counters[level] < 0 ? head?.start ?? 1 : counters[level] + 1;
1349
+ for (let l = level + 1; l <= 10; l++) counters[l] = -1;
1343
1350
  const fmtText = head ? head.text.trim() : `^${level}.`;
1344
1351
  const prefix = fmtText.replace(/\^(10|[1-9])/g, (_, d) => {
1345
1352
  const lv = parseInt(d, 10);
1346
1353
  const refHead = numDef.heads.get(lv);
1347
- const n = counters[lv] || refHead?.start || 1;
1354
+ const n = counters[lv] >= 0 ? counters[lv] : refHead?.start ?? 1;
1348
1355
  return formatHeadNumber(n, refHead?.numFormat || "DIGIT");
1349
1356
  });
1350
1357
  return { prefix: prefix || void 0, headingLevel };
1351
1358
  }
1352
1359
 
1353
1360
  // src/hwpx/table-build.ts
1354
- function buildTableWithCellMeta(state) {
1355
- const table2 = buildTable(state.rows);
1361
+ function buildTableWithCellMeta(state, keepAnchoredEmptyCols) {
1362
+ const table2 = buildTable(state.rows, { keepAnchoredEmptyCols });
1356
1363
  if (state.caption) table2.caption = state.caption;
1357
1364
  const anchors = [];
1358
1365
  {
@@ -1416,7 +1423,7 @@ function completeTable(newTable, tableStack, blocks, ctx) {
1416
1423
  if (newTable.caption) blocks.push({ type: "paragraph", text: newTable.caption, pageNumber: ctx.sectionNum });
1417
1424
  return parentTable;
1418
1425
  }
1419
- const ir = buildTableWithCellMeta(newTable);
1426
+ const ir = buildTableWithCellMeta(newTable, ctx.shared.keepTrailingEmptyCols);
1420
1427
  const block = { type: "table", table: ir, pageNumber: ctx.sectionNum };
1421
1428
  if (parentTable?.cell) {
1422
1429
  const cell2 = parentTable.cell;
@@ -1450,7 +1457,8 @@ function parseSectionXml(xml, styleMap, warnings, sectionNum, shared) {
1450
1457
  walkSection(doc.documentElement, blocks, null, [], ctx);
1451
1458
  return blocks;
1452
1459
  }
1453
- function extractImageRef(el) {
1460
+ function extractImageRef(el, depth = 0) {
1461
+ if (depth > MAX_XML_DEPTH) return null;
1454
1462
  const children = el.childNodes;
1455
1463
  if (!children) return null;
1456
1464
  for (let i = 0; i < children.length; i++) {
@@ -1461,7 +1469,7 @@ function extractImageRef(el) {
1461
1469
  const ref = child.getAttribute("binaryItemIDRef") || child.getAttribute("href") || "";
1462
1470
  if (ref) return ref;
1463
1471
  }
1464
- const nested = extractImageRef(child);
1472
+ const nested = extractImageRef(child, depth + 1);
1465
1473
  if (nested) return nested;
1466
1474
  }
1467
1475
  const directRef = el.getAttribute("binaryItemIDRef") || "";
@@ -1568,7 +1576,7 @@ function walkSection(node, blocks, tableCtx, tableStack, ctx, depth = 0) {
1568
1576
  ;
1569
1577
  (cell2.blocks ??= []).push(cellBlock);
1570
1578
  } else if (!tableCtx) {
1571
- if (/^─{10,}$/.test(text)) {
1579
+ if (ctx.shared.kordocLayout && /^─{10,}$/.test(text)) {
1572
1580
  blocks.push({ type: "separator", pageNumber: ctx.sectionNum });
1573
1581
  tableCtx = walkParagraphChildren(el, blocks, tableCtx, tableStack, ctx, depth + 1);
1574
1582
  break;
@@ -1706,6 +1714,19 @@ function findKordocMarkedCell(el, depth) {
1706
1714
  }
1707
1715
  return null;
1708
1716
  }
1717
+ function hasPageAutoNum(el, depth = 0) {
1718
+ if (depth > 10) return false;
1719
+ const children = el.childNodes;
1720
+ if (!children) return false;
1721
+ for (let i = 0; i < children.length; i++) {
1722
+ const ch = children[i];
1723
+ if (ch.nodeType !== 1) continue;
1724
+ const tag = (ch.tagName || ch.localName || "").replace(/^[^:]+:/, "");
1725
+ if (tag === "autoNum" && ch.getAttribute?.("numType") === "PAGE") return true;
1726
+ if (hasPageAutoNum(ch, depth + 1)) return true;
1727
+ }
1728
+ return false;
1729
+ }
1709
1730
  function collectSubListText(el, ctx, depth = 0) {
1710
1731
  if (depth > 10) return "";
1711
1732
  const parts = [];
@@ -1844,8 +1865,8 @@ function extractDrawTextBlocks(drawTextNode, blocks, ctx) {
1844
1865
  } else {
1845
1866
  const info = extractParagraphInfo(child, ctx.styleMap, ctx);
1846
1867
  let text = info.text.trim();
1868
+ const ph = resolveParaHeading(child, ctx);
1847
1869
  if (text) {
1848
- const ph = resolveParaHeading(child, ctx);
1849
1870
  if (ph?.prefix) text = ph.prefix + " " + text;
1850
1871
  const block = { type: "paragraph", text, style: info.style ?? void 0, pageNumber: ctx.sectionNum };
1851
1872
  if (info.href) block.href = info.href;
@@ -1884,6 +1905,22 @@ function extractParagraphInfo(para2, styleMap, ctx) {
1884
1905
  let href;
1885
1906
  let footnote;
1886
1907
  let charPrId;
1908
+ const linkRanges = [];
1909
+ const openFields = [];
1910
+ const onFieldBegin = (el) => {
1911
+ const url = extractHyperlinkHref(el);
1912
+ if (url) {
1913
+ linkRanges.push({ url, start: text.length });
1914
+ openFields.push({ rangeIdx: linkRanges.length - 1 });
1915
+ if (!href) href = url;
1916
+ } else {
1917
+ openFields.push({});
1918
+ }
1919
+ };
1920
+ const onFieldEnd = () => {
1921
+ const open = openFields.pop();
1922
+ if (open?.rangeIdx !== void 0) linkRanges[open.rangeIdx].end = text.length;
1923
+ };
1887
1924
  const handleCtrl = (ctrlEl) => {
1888
1925
  const kids2 = ctrlEl.childNodes;
1889
1926
  if (!kids2) return;
@@ -1893,10 +1930,13 @@ function extractParagraphInfo(para2, styleMap, ctx) {
1893
1930
  const ktag = (k.tagName || k.localName || "").replace(/^[^:]+:/, "");
1894
1931
  switch (ktag) {
1895
1932
  // 머리말/꼬리말 — 문서당 1회 수집, 본문 앞/뒤 배치
1933
+ // 페이지 번호 크롬(autoNum PAGE + 문자 없는 잔여 텍스트, 예: "- 1 -")은 본문 정보가
1934
+ // 아니므로 방출하지 않는다
1896
1935
  case "header":
1897
1936
  case "footer": {
1898
1937
  if (!ctx) break;
1899
1938
  const t = collectSubListText(k, ctx);
1939
+ if (t && hasPageAutoNum(k) && !/\p{L}/u.test(t)) break;
1900
1940
  if (t) {
1901
1941
  const bucket = ktag === "header" ? ctx.shared.pageText.headers : ctx.shared.pageText.footers;
1902
1942
  if (!bucket.includes(t)) bucket.push(t);
@@ -1910,13 +1950,12 @@ function extractParagraphInfo(para2, styleMap, ctx) {
1910
1950
  if (noteText) footnote = (footnote ? footnote + "; " : "") + noteText;
1911
1951
  break;
1912
1952
  }
1913
- // 하이퍼링크 — fieldBegin type=HYPERLINK의 Path 파라미터
1914
- case "fieldBegin": {
1915
- const url = extractHyperlinkHref(k);
1916
- if (url && !href) href = url;
1953
+ // 하이퍼링크 — fieldBegin type=HYPERLINK의 Path 파라미터 (extent 오프셋 추적)
1954
+ case "fieldBegin":
1955
+ onFieldBegin(k);
1917
1956
  break;
1918
- }
1919
1957
  case "fieldEnd":
1958
+ onFieldEnd();
1920
1959
  break;
1921
1960
  // 변경추적 — 삭제 구간(deleteBegin~End)의 텍스트는 출력 제외 (최종본 상태 재현)
1922
1961
  case "deleteBegin":
@@ -1958,7 +1997,8 @@ function extractParagraphInfo(para2, styleMap, ctx) {
1958
1997
  }
1959
1998
  }
1960
1999
  };
1961
- const walk = (node) => {
2000
+ const walk = (node, depth = 0) => {
2001
+ if (depth > MAX_XML_DEPTH) return;
1962
2002
  const children = node.childNodes;
1963
2003
  if (!children) return;
1964
2004
  for (let i = 0; i < children.length; i++) {
@@ -1979,7 +2019,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
1979
2019
  const tag = (child.tagName || child.localName || "").replace(/^[^:]+:/, "");
1980
2020
  switch (tag) {
1981
2021
  case "t":
1982
- walk(child);
2022
+ walk(child, depth + 1);
1983
2023
  break;
1984
2024
  // 자식 순회 (tab 등 하위 요소 처리)
1985
2025
  case "tab": {
@@ -2012,7 +2052,7 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2012
2052
  const safe = sanitizeHref(url);
2013
2053
  if (safe) href = safe;
2014
2054
  }
2015
- walk(child);
2055
+ walk(child, depth + 1);
2016
2056
  break;
2017
2057
  }
2018
2058
  // 각주/미주
@@ -2028,12 +2068,10 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2028
2068
  case "ctrl":
2029
2069
  handleCtrl(child);
2030
2070
  break;
2031
- // run 직계 fieldBegin (비표준 경로) — 하이퍼링크 URL 추출
2032
- case "fieldBegin": {
2033
- const url = extractHyperlinkHref(child);
2034
- if (url && !href) href = url;
2071
+ // run 직계 fieldBegin (비표준 경로) — 하이퍼링크 URL·extent 추적
2072
+ case "fieldBegin":
2073
+ onFieldBegin(child);
2035
2074
  break;
2036
- }
2037
2075
  // run 직계 변경추적 마커 (비표준 경로)
2038
2076
  case "deleteBegin":
2039
2077
  if (ctx) ctx.shared.track.deleteDepth++;
@@ -2045,6 +2083,8 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2045
2083
  case "insertEnd":
2046
2084
  break;
2047
2085
  case "fieldEnd":
2086
+ onFieldEnd();
2087
+ break;
2048
2088
  case "parameters":
2049
2089
  case "stringParam":
2050
2090
  case "integerParam":
@@ -2083,22 +2123,34 @@ function extractParagraphInfo(para2, styleMap, ctx) {
2083
2123
  case "r": {
2084
2124
  const runCharPr = child.getAttribute("charPrIDRef");
2085
2125
  if (runCharPr && !charPrId) charPrId = runCharPr;
2086
- walk(child);
2126
+ walk(child, depth + 1);
2087
2127
  break;
2088
2128
  }
2089
2129
  default:
2090
- walk(child);
2130
+ walk(child, depth + 1);
2091
2131
  break;
2092
2132
  }
2093
2133
  }
2094
2134
  };
2095
2135
  walk(para2);
2136
+ {
2137
+ const applied = [];
2138
+ const closed = linkRanges.filter((r) => r.end !== void 0 && r.end > r.start).sort((a, b) => b.start - a.start);
2139
+ for (const r of closed) {
2140
+ if (applied.some(([s, e]) => r.start < e && r.end > s)) continue;
2141
+ const anchor = text.slice(r.start, r.end);
2142
+ if (!anchor.trim() || /[\n\x1F\[\]]/.test(anchor)) continue;
2143
+ text = text.slice(0, r.start) + `[${anchor}](${r.url})` + text.slice(r.end);
2144
+ applied.push([r.start, r.end]);
2145
+ }
2146
+ if (applied.length) href = void 0;
2147
+ }
2096
2148
  const leaderIdx = text.indexOf("");
2097
2149
  if (leaderIdx >= 0) text = text.substring(0, leaderIdx);
2098
2150
  let cleanText = text.replace(/[ \t]+/g, " ").trim();
2099
2151
  if (/^그림입니다\.?\s*원본\s*그림의\s*(이름|크기)/.test(cleanText)) cleanText = "";
2100
2152
  cleanText = cleanText.replace(/그림입니다\.?\s*원본\s*그림의\s*(이름|크기)[^\n]*(\n[^\n]*원본\s*그림의\s*(이름|크기)[^\n]*)*/g, "").trim();
2101
- cleanText = cleanText.replace(/(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?/g, "").trim();
2153
+ cleanText = cleanText.replace(/^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|선|직선|곡선|화살표|오각형|육각형|팔각형|별|십자|구름|마름모|도넛|평행사변형|사다리꼴|개체|그리기\s?개체|묶음\s?개체|글상자|표|그림|OLE\s?개체)\s?입니다\.?$/gm, "").trim();
2102
2154
  let style;
2103
2155
  if (styleMap && charPrId) {
2104
2156
  const charProp = styleMap.charProperties.get(charPrId);
@@ -2222,10 +2274,9 @@ var CHAR_LINE = 0;
2222
2274
  var CHAR_SECTION_BREAK = 10;
2223
2275
  var CHAR_PARA = 13;
2224
2276
  var CHAR_TAB = 9;
2225
- var CHAR_HYPHEN = 30;
2226
- var CHAR_NBSP = 31;
2227
- var CHAR_FIXED_NBSP = 24;
2228
- var CHAR_FIXED_WIDTH = 25;
2277
+ var CHAR_HYPHEN = 24;
2278
+ var CHAR_NBSP = 30;
2279
+ var CHAR_FIXED_WIDTH = 31;
2229
2280
  var FLAG_COMPRESSED = 1 << 0;
2230
2281
  var FLAG_ENCRYPTED = 1 << 1;
2231
2282
  var FLAG_DISTRIBUTION = 1 << 2;
@@ -2424,9 +2475,6 @@ function appendParaText(state, data, resolveControl) {
2424
2475
  result += "-";
2425
2476
  break;
2426
2477
  case CHAR_NBSP:
2427
- result += " ";
2428
- break;
2429
- case CHAR_FIXED_NBSP:
2430
2478
  result += "\xA0";
2431
2479
  break;
2432
2480
  // 진짜 NBSP
@@ -2725,8 +2773,8 @@ async function extractImagesFromZip(zip, blocks, decompressed, warnings, sweepUn
2725
2773
  const ext = path.includes(".") ? path.split(".").pop() || "png" : "png";
2726
2774
  const mimeType = imageExtToMime(ext);
2727
2775
  imageIndex++;
2728
- const filename = `image_${String(imageIndex).padStart(3, "0")}.${mimeToExt(mimeType)}`;
2729
- img = { filename, data, mimeType };
2776
+ const filename2 = `image_${String(imageIndex).padStart(3, "0")}.${mimeToExt(mimeType)}`;
2777
+ img = { filename: filename2, data, mimeType };
2730
2778
  images.push(img);
2731
2779
  usedPaths.add(path);
2732
2780
  break;
@@ -2740,12 +2788,13 @@ async function extractImagesFromZip(zip, blocks, decompressed, warnings, sweepUn
2740
2788
  if (!img) {
2741
2789
  block.type = "paragraph";
2742
2790
  block.text = `[\uC774\uBBF8\uC9C0: ${ref}]`;
2743
- if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, `[\uC774\uBBF8\uC9C0: ${ref}]`);
2791
+ if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, () => `[\uC774\uBBF8\uC9C0: ${ref}]`);
2744
2792
  continue;
2745
2793
  }
2746
- block.text = img.filename;
2794
+ const filename = img.filename;
2795
+ block.text = filename;
2747
2796
  block.imageData = { data: img.data, mimeType: img.mimeType, filename: ref };
2748
- if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, `![image](${img.filename})`);
2797
+ if (ownerCell) ownerCell.text = ownerCell.text.replace(`![image](${ref})`, () => `![image](${filename})`);
2749
2798
  }
2750
2799
  if (sweepUnreferenced) {
2751
2800
  const binEntries = zip.file(/(?:^|\/)BinData\//i);
@@ -2991,6 +3040,7 @@ async function parseHwpxDocument(buffer, options) {
2991
3040
  const blocks = [];
2992
3041
  const shared = createSectionShared();
2993
3042
  shared.kordocLayout = await readKordocLayout(zip);
3043
+ shared.keepTrailingEmptyCols = options?.keepTrailingEmptyCols;
2994
3044
  let parsedSections = 0;
2995
3045
  for (let si = 0; si < sectionPaths.length; si++) {
2996
3046
  if (pageFilter && !pageFilter.has(si + 1)) continue;
@@ -2999,12 +3049,12 @@ async function parseHwpxDocument(buffer, options) {
2999
3049
  try {
3000
3050
  const xml = await file.async("text");
3001
3051
  decompressed.total += xml.length * 2;
3002
- if (decompressed.total > MAX_DECOMPRESS_SIZE) throw new KordocError("ZIP \uC555\uCD95 \uD574\uC81C \uD06C\uAE30 \uCD08\uACFC (ZIP bomb \uC758\uC2EC)");
3052
+ if (decompressed.total > MAX_DECOMPRESS_SIZE) throw new ZipBombError("ZIP \uC555\uCD95 \uD574\uC81C \uD06C\uAE30 \uCD08\uACFC (ZIP bomb \uC758\uC2EC)");
3003
3053
  blocks.push(...parseSectionXml(xml, styleMap, warnings, si + 1, shared));
3004
3054
  parsedSections++;
3005
3055
  options?.onProgress?.(parsedSections, totalTarget);
3006
3056
  } catch (secErr) {
3007
- if (secErr instanceof KordocError) throw secErr;
3057
+ if (secErr instanceof ZipBombError) throw secErr;
3008
3058
  warnings.push({ page: si + 1, message: `\uC139\uC158 ${si + 1} \uD30C\uC2F1 \uC2E4\uD328: ${secErr instanceof Error ? secErr.message : "\uC54C \uC218 \uC5C6\uB294 \uC624\uB958"}`, code: "PARTIAL_PARSE" });
3009
3059
  }
3010
3060
  }
@@ -4221,6 +4271,7 @@ function parseHwp5Document(buffer, options) {
4221
4271
  const totalTarget = pageFilter ? pageFilter.size : sections.length;
4222
4272
  const bodyBlocks = [];
4223
4273
  const doc = createHwp5DocState();
4274
+ doc.keepTrailingEmptyCols = options?.keepTrailingEmptyCols;
4224
4275
  let totalDecompressed = 0;
4225
4276
  let parsedSections = 0;
4226
4277
  for (let si = 0; si < sections.length; si++) {
@@ -4836,7 +4887,7 @@ function parseTableControl(ctrl, records, ctx) {
4836
4887
  return table3;
4837
4888
  }
4838
4889
  const cellRows = arrangeCells(rows, cols, cells);
4839
- const table2 = buildTable(cellRows);
4890
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: ctx.doc.keepTrailingEmptyCols });
4840
4891
  if (caption && table2.rows > 0) table2.caption = caption;
4841
4892
  return table2.rows > 0 ? table2 : null;
4842
4893
  }
@@ -16914,8 +16965,50 @@ function decodeJohab(ch) {
16914
16965
  }
16915
16966
  const hit = lookupSymbol(ch);
16916
16967
  if (hit !== null) return hit;
16968
+ return JOHAB_UNMAPPED;
16969
+ }
16970
+ return decodeHwp3Extra(ch);
16971
+ }
16972
+ function decodeHwp3Extra(ch) {
16973
+ if (ch >= 13712 && ch <= 13721) return 8544 + (ch - 13712);
16974
+ if (ch >= 14055 && ch <= 14064) return 9312 + (ch - 14055);
16975
+ switch (ch) {
16976
+ case 129:
16977
+ return 8220;
16978
+ // 왼쪽 큰따옴표
16979
+ case 130:
16980
+ return 8221;
16981
+ // 오른쪽 큰따옴표
16982
+ case 12316:
16983
+ return 9473;
16984
+ // ━ 굵은 가로선 (rhwp: U+F080F, 표시값 직행)
16985
+ case 12349:
16986
+ return 9632;
16987
+ // ■ (rhwp: U+F0827, 표시값 직행)
16988
+ case 13158:
16989
+ return 9633;
16990
+ // □ 글머리 (rhwp: U+F03C5, 한컴 표시값 직행)
16991
+ case 13316:
16992
+ return 8228;
16993
+ // 한 점 리더
16994
+ case 13377:
16995
+ return 9632;
16996
+ // ■
16997
+ case 13382:
16998
+ return 8594;
16999
+ // → 오른쪽 화살표
17000
+ case 13793:
17001
+ return 9472;
17002
+ // ─ 상자 그리기 가로선
17003
+ case 13433:
17004
+ return 9655;
17005
+ // ▷
17006
+ case 13434:
17007
+ return 9654;
17008
+ // ▶
17009
+ default:
17010
+ return JOHAB_UNMAPPED;
16917
17011
  }
16918
- return JOHAB_UNMAPPED;
16919
17012
  }
16920
17013
  function decodeHcharString(bytes) {
16921
17014
  let out = "";
@@ -17030,11 +17123,12 @@ function readHeader(reader) {
17030
17123
  }
17031
17124
 
17032
17125
  // src/hwp3/parser.ts
17126
+ var MAX_DECOMPRESS_SIZE3 = 100 * 1024 * 1024;
17033
17127
  var PARA_SHAPE_SIZE = 187;
17034
17128
  var LINE_INFO_SIZE = 14;
17035
17129
  var INLINE_CHAR_SHAPE_SIZE = 31;
17036
17130
  var SIMPLE_CTRL = /* @__PURE__ */ new Map([
17037
- [9, { extraBytes: 0, extraHchar: 0, emit: " " }],
17131
+ [9, { extraBytes: 6, extraHchar: 3, emit: " " }],
17038
17132
  [7, { extraBytes: 6, extraHchar: 3, emit: "\uFFFC" }],
17039
17133
  [8, { extraBytes: 6, extraHchar: 3, emit: "\uFFFC" }],
17040
17134
  [18, { extraBytes: 6, extraHchar: 3, emit: " " }],
@@ -17065,8 +17159,11 @@ function parseHwp3Document(buffer, _options) {
17065
17159
  const warnings = [];
17066
17160
  if (header.compressed !== 0) {
17067
17161
  try {
17068
- body = inflateRawSync3(tail);
17162
+ body = inflateRawSync3(tail, { maxOutputLength: MAX_DECOMPRESS_SIZE3 });
17069
17163
  } catch (err) {
17164
+ if (err?.code === "ERR_BUFFER_TOO_LARGE") {
17165
+ throw new Error(`HWP3 \uC555\uCD95 \uD574\uC81C \uACB0\uACFC\uAC00 \uCD5C\uB300 \uD5C8\uC6A9 \uD06C\uAE30(${MAX_DECOMPRESS_SIZE3 / 1024 / 1024}MB)\uB97C \uCD08\uACFC\uD588\uC2B5\uB2C8\uB2E4`);
17166
+ }
17070
17167
  const msg2 = err instanceof Error ? err.message : String(err);
17071
17168
  throw new Error(`HWP3 \uC555\uCD95 \uD574\uC81C \uC2E4\uD328: ${msg2}`);
17072
17169
  }
@@ -17207,6 +17304,12 @@ function parseCharStream(reader, charCount, ctx) {
17207
17304
  parseParagraphList2(reader, ctx);
17208
17305
  break;
17209
17306
  }
17307
+ case 5:
17308
+ if (headerVal1 > 0 && headerVal1 < 1e6) reader.skip(headerVal1);
17309
+ break;
17310
+ case 6:
17311
+ reader.skip(34);
17312
+ break;
17210
17313
  case 29:
17211
17314
  if (headerVal1 < 1e6) reader.skip(headerVal1);
17212
17315
  break;
@@ -17263,7 +17366,7 @@ function isDistributionSentinel(markdown) {
17263
17366
  import JSZip3 from "jszip";
17264
17367
  import { DOMParser } from "@xmldom/xmldom";
17265
17368
  var MAX_SHEETS = 100;
17266
- var MAX_DECOMPRESS_SIZE3 = 100 * 1024 * 1024;
17369
+ var MAX_DECOMPRESS_SIZE4 = 100 * 1024 * 1024;
17267
17370
  var MAX_ROWS2 = 1e4;
17268
17371
  var MAX_COLS2 = 200;
17269
17372
  function cleanNumericValue(raw) {
@@ -17303,16 +17406,87 @@ function getTextContent(el) {
17303
17406
  function parseXml(text) {
17304
17407
  return new DOMParser().parseFromString(stripDtd(text), "text/xml");
17305
17408
  }
17409
+ function collectRichText(root) {
17410
+ let out = "";
17411
+ const walk = (node) => {
17412
+ const children = node.childNodes;
17413
+ for (let i = 0; i < children.length; i++) {
17414
+ if (children[i].nodeType !== 1) continue;
17415
+ const el = children[i];
17416
+ const local = el.localName || el.tagName?.replace(/^[^:]+:/, "") || "";
17417
+ if (local === "rPh") continue;
17418
+ if (local === "t") out += el.textContent ?? "";
17419
+ else walk(el);
17420
+ }
17421
+ };
17422
+ walk(root);
17423
+ return out;
17424
+ }
17306
17425
  function parseSharedStrings(xml) {
17307
17426
  const doc = parseXml(xml);
17308
17427
  const strings = [];
17309
17428
  const siList = getElements(doc.documentElement, "si");
17310
17429
  for (const si of siList) {
17311
- const tElements = getElements(si, "t");
17312
- strings.push(tElements.map((t) => t.textContent ?? "").join(""));
17430
+ strings.push(collectRichText(si));
17313
17431
  }
17314
17432
  return strings;
17315
17433
  }
17434
+ var BUILTIN_DATE_FMT = /* @__PURE__ */ new Map([
17435
+ [14, "date"],
17436
+ [15, "date"],
17437
+ [16, "date"],
17438
+ [17, "date"],
17439
+ [18, "datetime"],
17440
+ [19, "datetime"],
17441
+ [20, "datetime"],
17442
+ [21, "datetime"],
17443
+ [22, "datetime"],
17444
+ [45, "datetime"],
17445
+ [46, "datetime"],
17446
+ [47, "datetime"]
17447
+ ]);
17448
+ function classifyDateFormat(code) {
17449
+ const stripped = code.replace(/"[^"]*"/g, "").replace(/\[[^\]]*\]/g, "").replace(/\\./g, "");
17450
+ if (!/[ymdh]/i.test(stripped)) return null;
17451
+ return /[hs]/i.test(stripped) ? "datetime" : "date";
17452
+ }
17453
+ function dateKindOfFmt(fmtId, customFormats) {
17454
+ const builtin = BUILTIN_DATE_FMT.get(fmtId);
17455
+ if (builtin) return builtin;
17456
+ const code = customFormats.get(fmtId);
17457
+ return code !== void 0 ? classifyDateFormat(code) : null;
17458
+ }
17459
+ function dateSerialToIso(serial, date1904, kind) {
17460
+ if (!isFinite(serial) || serial < 0) return null;
17461
+ let days = serial;
17462
+ if (date1904) days += 1462;
17463
+ else if (days < 60) days += 1;
17464
+ const ms = Math.round((days - 25569) * 864e5);
17465
+ const d = new Date(ms);
17466
+ if (isNaN(d.getTime()) || d.getUTCFullYear() > 9999) return null;
17467
+ const iso = d.toISOString();
17468
+ return kind === "datetime" ? iso.slice(0, 19) : iso.slice(0, 10);
17469
+ }
17470
+ function parseStyleDateXfs(xml) {
17471
+ const doc = parseXml(xml);
17472
+ const customFormats = /* @__PURE__ */ new Map();
17473
+ for (const el of getElements(doc.documentElement, "numFmt")) {
17474
+ const id = parseInt(el.getAttribute("numFmtId") ?? "", 10);
17475
+ if (isNaN(id)) continue;
17476
+ customFormats.set(id, el.getAttribute("formatCode") ?? "");
17477
+ }
17478
+ const dateXfs = /* @__PURE__ */ new Map();
17479
+ const cellXfsEls = getElements(doc.documentElement, "cellXfs");
17480
+ if (cellXfsEls.length === 0) return dateXfs;
17481
+ const xfs = getElements(cellXfsEls[0], "xf");
17482
+ for (let i = 0; i < xfs.length; i++) {
17483
+ const fmtId = parseInt(xfs[i].getAttribute("numFmtId") ?? "", 10);
17484
+ if (isNaN(fmtId)) continue;
17485
+ const kind = dateKindOfFmt(fmtId, customFormats);
17486
+ if (kind) dateXfs.set(i, kind);
17487
+ }
17488
+ return dateXfs;
17489
+ }
17316
17490
  function parseWorkbook(xml) {
17317
17491
  const doc = parseXml(xml);
17318
17492
  const sheets = [];
@@ -17324,7 +17498,9 @@ function parseWorkbook(xml) {
17324
17498
  rId: el.getAttribute("r:id") ?? ""
17325
17499
  });
17326
17500
  }
17327
- return sheets;
17501
+ const prEls = getElements(doc.documentElement, "workbookPr");
17502
+ const d1904 = prEls.length > 0 ? prEls[0].getAttribute("date1904") : null;
17503
+ return { sheets, date1904: d1904 === "1" || d1904 === "true" };
17328
17504
  }
17329
17505
  function parseRels(xml) {
17330
17506
  const doc = parseXml(xml);
@@ -17337,21 +17513,25 @@ function parseRels(xml) {
17337
17513
  }
17338
17514
  return map;
17339
17515
  }
17340
- function parseWorksheet(xml, sharedStrings) {
17516
+ function parseWorksheet(xml, sharedStrings, dateXfs, date1904) {
17341
17517
  const doc = parseXml(xml);
17342
17518
  const grid = [];
17343
17519
  let maxRow = 0;
17344
17520
  let maxCol = 0;
17345
17521
  const rows = getElements(doc.documentElement, "row");
17522
+ let prevRow = -1;
17346
17523
  for (const rowEl of rows) {
17347
- const rowNum = parseInt(rowEl.getAttribute("r") ?? "0", 10) - 1;
17524
+ const rAttr = rowEl.getAttribute("r");
17525
+ const rowNum = rAttr !== null ? parseInt(rAttr, 10) - 1 : prevRow + 1;
17348
17526
  if (rowNum < 0 || rowNum >= MAX_ROWS2) continue;
17527
+ if (Number.isFinite(rowNum)) prevRow = rowNum;
17349
17528
  const cells = getElements(rowEl, "c");
17529
+ let prevCol = -1;
17350
17530
  for (const cellEl of cells) {
17351
17531
  const ref = cellEl.getAttribute("r");
17352
- if (!ref) continue;
17353
- const pos = parseCellRef(ref);
17354
- if (!pos || pos.col >= MAX_COLS2) continue;
17532
+ const pos = ref !== null ? parseCellRef(ref) : { col: prevCol + 1, row: rowNum };
17533
+ if (!pos || !Number.isFinite(pos.row) || pos.row < 0 || pos.row >= MAX_ROWS2 || pos.col >= MAX_COLS2) continue;
17534
+ prevCol = pos.col;
17355
17535
  const type = cellEl.getAttribute("t");
17356
17536
  const vElements = getElements(cellEl, "v");
17357
17537
  const fElements = getElements(cellEl, "f");
@@ -17365,12 +17545,19 @@ function parseWorksheet(xml, sharedStrings) {
17365
17545
  value = raw === "1" ? "TRUE" : "FALSE";
17366
17546
  } else {
17367
17547
  value = cleanNumericValue(raw);
17548
+ if (type === null || type === "n") {
17549
+ const sAttr = cellEl.getAttribute("s");
17550
+ const kind = sAttr !== null ? dateXfs.get(parseInt(sAttr, 10)) : void 0;
17551
+ if (kind) {
17552
+ const iso = dateSerialToIso(parseFloat(raw), date1904, kind);
17553
+ if (iso) value = iso;
17554
+ }
17555
+ }
17368
17556
  }
17369
17557
  } else if (type === "inlineStr") {
17370
17558
  const isEl = getElements(cellEl, "is");
17371
17559
  if (isEl.length > 0) {
17372
- const tElements = getElements(isEl[0], "t");
17373
- value = tElements.map((t) => t.textContent ?? "").join("");
17560
+ value = collectRichText(isEl[0]);
17374
17561
  }
17375
17562
  }
17376
17563
  if (!value && fElements.length > 0) {
@@ -17389,11 +17576,18 @@ function parseWorksheet(xml, sharedStrings) {
17389
17576
  const ref = el.getAttribute("ref");
17390
17577
  if (!ref) continue;
17391
17578
  const m = parseMergeRef(ref);
17392
- if (m) merges.push(m);
17579
+ if (m) {
17580
+ merges.push({
17581
+ startCol: Math.min(m.startCol, MAX_COLS2 - 1),
17582
+ startRow: Math.min(m.startRow, MAX_ROWS2 - 1),
17583
+ endCol: Math.min(m.endCol, MAX_COLS2 - 1),
17584
+ endRow: Math.min(m.endRow, MAX_ROWS2 - 1)
17585
+ });
17586
+ }
17393
17587
  }
17394
17588
  return { grid, merges, maxRow, maxCol };
17395
17589
  }
17396
- function sheetToBlocks(sheetName, grid, merges, maxRow, maxCol, sheetIndex) {
17590
+ function sheetToBlocks(sheetName, grid, merges, maxRow, maxCol, sheetIndex, keepAnchoredEmptyCols) {
17397
17591
  const blocks = [];
17398
17592
  if (sheetName) {
17399
17593
  blocks.push({
@@ -17445,7 +17639,7 @@ function sheetToBlocks(sheetName, grid, merges, maxRow, maxCol, sheetIndex) {
17445
17639
  cellRows.push(row);
17446
17640
  }
17447
17641
  if (cellRows.length > 0) {
17448
- const table2 = buildTable(cellRows);
17642
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols });
17449
17643
  if (table2.rows > 0) {
17450
17644
  blocks.push({ type: "table", table: table2, pageNumber: sheetIndex + 1 });
17451
17645
  }
@@ -17453,7 +17647,7 @@ function sheetToBlocks(sheetName, grid, merges, maxRow, maxCol, sheetIndex) {
17453
17647
  return blocks;
17454
17648
  }
17455
17649
  async function parseXlsxDocument(buffer, options) {
17456
- precheckZipSize(buffer, MAX_DECOMPRESS_SIZE3);
17650
+ precheckZipSize(buffer, MAX_DECOMPRESS_SIZE4);
17457
17651
  const zip = await JSZip3.loadAsync(buffer);
17458
17652
  const warnings = [];
17459
17653
  const workbookFile = zip.file("xl/workbook.xml");
@@ -17465,10 +17659,18 @@ async function parseXlsxDocument(buffer, options) {
17465
17659
  if (ssFile) {
17466
17660
  sharedStrings = parseSharedStrings(await ssFile.async("text"));
17467
17661
  }
17468
- const sheets = parseWorkbook(await workbookFile.async("text"));
17662
+ const { sheets, date1904 } = parseWorkbook(await workbookFile.async("text"));
17469
17663
  if (sheets.length === 0) {
17470
17664
  throw new KordocError("XLSX \uD30C\uC77C\uC5D0 \uC2DC\uD2B8\uAC00 \uC5C6\uC2B5\uB2C8\uB2E4");
17471
17665
  }
17666
+ let dateXfs = /* @__PURE__ */ new Map();
17667
+ const stylesFile = zip.file("xl/styles.xml");
17668
+ if (stylesFile) {
17669
+ try {
17670
+ dateXfs = parseStyleDateXfs(await stylesFile.async("text"));
17671
+ } catch {
17672
+ }
17673
+ }
17472
17674
  let relsMap = /* @__PURE__ */ new Map();
17473
17675
  const relsFile = zip.file("xl/_rels/workbook.xml.rels");
17474
17676
  if (relsFile) {
@@ -17506,8 +17708,8 @@ async function parseXlsxDocument(buffer, options) {
17506
17708
  }
17507
17709
  try {
17508
17710
  const sheetXml = await sheetFile.async("text");
17509
- const { grid, merges, maxRow, maxCol } = parseWorksheet(sheetXml, sharedStrings);
17510
- const sheetBlocks = sheetToBlocks(sheet.name, grid, merges, maxRow, maxCol, i);
17711
+ const { grid, merges, maxRow, maxCol } = parseWorksheet(sheetXml, sharedStrings, dateXfs, date1904);
17712
+ const sheetBlocks = sheetToBlocks(sheet.name, grid, merges, maxRow, maxCol, i, options?.keepTrailingEmptyCols);
17511
17713
  blocks.push(...sheetBlocks);
17512
17714
  } catch (err) {
17513
17715
  warnings.push({
@@ -17550,7 +17752,10 @@ var OP_CONTINUE = 60;
17550
17752
  var OP_BOUNDSHEET8 = 133;
17551
17753
  var OP_SST = 252;
17552
17754
  var OP_CODEPAGE = 66;
17755
+ var OP_DATE1904 = 34;
17553
17756
  var OP_FILEPASS = 47;
17757
+ var OP_FORMAT = 1054;
17758
+ var OP_XF = 224;
17554
17759
  var OP_NUMBER = 515;
17555
17760
  var OP_RK = 638;
17556
17761
  var OP_MULRK = 189;
@@ -17562,6 +17767,8 @@ var OP_BOOLERR = 517;
17562
17767
  var OP_BLANK = 513;
17563
17768
  var OP_MULBLANK = 190;
17564
17769
  var OP_MERGECELLS = 229;
17770
+ var OP_SHRFMLA = 1212;
17771
+ var OP_ARRAY = 545;
17565
17772
  var DT_GLOBALS = 5;
17566
17773
  var DT_WORKSHEET = 16;
17567
17774
  var MAX_RECORDS2 = 1e6;
@@ -17657,7 +17864,7 @@ function decodeUtf16Le(buf) {
17657
17864
  }
17658
17865
 
17659
17866
  // src/xls/sst.ts
17660
- function parseString(buf, offset, segments) {
17867
+ function parseString(buf, offset, segments, segCursor) {
17661
17868
  if (offset + 3 > buf.length) return null;
17662
17869
  const cch = buf.readUInt16LE(offset);
17663
17870
  let flags = buf.readUInt8(offset + 2);
@@ -17680,7 +17887,15 @@ function parseString(buf, offset, segments) {
17680
17887
  const charBytes = [];
17681
17888
  let charsRead = 0;
17682
17889
  while (charsRead < cch) {
17683
- const nextBoundary = segments.find((s) => s > off) ?? buf.length;
17890
+ while (segCursor.idx < segments.length && segments[segCursor.idx] < off) segCursor.idx++;
17891
+ if (segCursor.idx < segments.length && segments[segCursor.idx] === off) {
17892
+ if (off >= buf.length) return null;
17893
+ flags = buf.readUInt8(off);
17894
+ highByte = (flags & 1) !== 0;
17895
+ off += 1;
17896
+ segCursor.idx++;
17897
+ }
17898
+ const nextBoundary = segCursor.idx < segments.length ? segments[segCursor.idx] : buf.length;
17684
17899
  const remainChars = cch - charsRead;
17685
17900
  const bytesPerChar = highByte ? 2 : 1;
17686
17901
  const bytesAvail = nextBoundary - off;
@@ -17691,12 +17906,10 @@ function parseString(buf, offset, segments) {
17691
17906
  charBytes.push(highByte ? slice : padToUtf16(slice));
17692
17907
  off += bytesToRead;
17693
17908
  charsRead += charsInThisRun;
17694
- }
17695
- if (charsRead < cch) {
17696
- if (off >= buf.length) return null;
17697
- flags = buf.readUInt8(off);
17698
- highByte = (flags & 1) !== 0;
17699
- off += 1;
17909
+ } else if (nextBoundary < buf.length) {
17910
+ off = nextBoundary;
17911
+ } else {
17912
+ return null;
17700
17913
  }
17701
17914
  }
17702
17915
  const text = decodeUtf16Le(Buffer.concat(charBytes));
@@ -17721,8 +17934,9 @@ function decodeSST(records) {
17721
17934
  const cstUnique = combined.readUInt32LE(4);
17722
17935
  const strings = [];
17723
17936
  let off = 8;
17937
+ const segCursor = { idx: 0 };
17724
17938
  for (let i = 0; i < cstUnique && off < combined.length; i++) {
17725
- const r = parseString(combined, off, segments);
17939
+ const r = parseString(combined, off, segments, segCursor);
17726
17940
  if (!r) break;
17727
17941
  strings.push(r.text);
17728
17942
  off += r.consumed;
@@ -17797,7 +18011,7 @@ function decodeFormulaStringRecord(data) {
17797
18011
  return decodeUtf16Le(padded);
17798
18012
  }
17799
18013
  }
17800
- function extractSheetCells(records, bofIndex, sst) {
18014
+ function extractSheetCells(records, bofIndex, sst, convertNum) {
17801
18015
  const cells = [];
17802
18016
  const merges = [];
17803
18017
  const bofOffset = records[bofIndex].offset;
@@ -17815,14 +18029,16 @@ function extractSheetCells(records, bofIndex, sst) {
17815
18029
  case OP_NUMBER: {
17816
18030
  const h = readCellHeader(rec.data);
17817
18031
  if (h && rec.data.length >= 14) {
17818
- cells.push({ row: h.row, col: h.col, value: rec.data.readDoubleLE(6) });
18032
+ const n = rec.data.readDoubleLE(6);
18033
+ cells.push({ row: h.row, col: h.col, value: convertNum ? convertNum(n, h.ixfe) : n });
17819
18034
  }
17820
18035
  break;
17821
18036
  }
17822
18037
  case OP_RK: {
17823
18038
  const h = readCellHeader(rec.data);
17824
18039
  if (h && rec.data.length >= 10) {
17825
- cells.push({ row: h.row, col: h.col, value: decodeRk(rec.data.readInt32LE(6)) });
18040
+ const n = decodeRk(rec.data.readInt32LE(6));
18041
+ cells.push({ row: h.row, col: h.col, value: convertNum ? convertNum(n, h.ixfe) : n });
17826
18042
  }
17827
18043
  break;
17828
18044
  }
@@ -17830,7 +18046,7 @@ function extractSheetCells(records, bofIndex, sst) {
17830
18046
  const m = decodeMulRk(rec.data);
17831
18047
  if (m) {
17832
18048
  for (const c of m.cells) {
17833
- cells.push({ row: m.row, col: c.col, value: c.value });
18049
+ cells.push({ row: m.row, col: c.col, value: convertNum ? convertNum(c.value, c.ixfe) : c.value });
17834
18050
  }
17835
18051
  }
17836
18052
  break;
@@ -17855,19 +18071,26 @@ function extractSheetCells(records, bofIndex, sst) {
17855
18071
  if (h && rec.data.length >= 14) {
17856
18072
  const result = decodeFormulaResult(rec.data.subarray(6, 14));
17857
18073
  if (result.kind === "stringRef") {
17858
- const next = records[i + 1];
18074
+ let j = i + 1;
18075
+ while (j < records.length && (records[j].opcode === OP_SHRFMLA || records[j].opcode === OP_ARRAY)) j++;
18076
+ const next = records[j];
17859
18077
  if (next && next.opcode === OP_STRING) {
17860
18078
  cells.push({
17861
18079
  row: h.row,
17862
18080
  col: h.col,
17863
18081
  value: decodeFormulaStringRecord(next.data)
17864
18082
  });
17865
- i++;
18083
+ i = j;
17866
18084
  } else {
17867
18085
  cells.push({ row: h.row, col: h.col, value: "" });
17868
18086
  }
17869
18087
  } else {
17870
- cells.push({ row: h.row, col: h.col, value: result.value });
18088
+ const v = result.value;
18089
+ cells.push({
18090
+ row: h.row,
18091
+ col: h.col,
18092
+ value: convertNum && typeof v === "number" ? convertNum(v, h.ixfe) : v
18093
+ });
17871
18094
  }
17872
18095
  }
17873
18096
  break;
@@ -17917,7 +18140,7 @@ function extractSheetCells(records, bofIndex, sst) {
17917
18140
 
17918
18141
  // src/xls/parser.ts
17919
18142
  var MAX_SHEETS2 = 100;
17920
- var MAX_ROWS3 = 1e5;
18143
+ var MAX_ROWS3 = 65536;
17921
18144
  var MAX_COLS3 = 1e3;
17922
18145
  function decodeBoundSheet(data) {
17923
18146
  if (data.length < 8) return null;
@@ -17940,10 +18163,33 @@ function decodeBoundSheet(data) {
17940
18163
  }
17941
18164
  return { name, lbPlyPos, dt };
17942
18165
  }
18166
+ function decodeFormatRecord(data) {
18167
+ if (data.length < 5) return null;
18168
+ const ifmt = data.readUInt16LE(0);
18169
+ const cch = data.readUInt16LE(2);
18170
+ const flags = data.readUInt8(4);
18171
+ const highByte = (flags & 1) !== 0;
18172
+ const start = 5;
18173
+ let code;
18174
+ if (highByte) {
18175
+ const end = Math.min(start + cch * 2, data.length);
18176
+ code = decodeUtf16Le(data.subarray(start, end));
18177
+ } else {
18178
+ const end = Math.min(start + cch, data.length);
18179
+ const slice = data.subarray(start, end);
18180
+ const padded = Buffer.alloc(slice.length * 2);
18181
+ for (let i = 0; i < slice.length; i++) padded[i * 2] = slice[i];
18182
+ code = decodeUtf16Le(padded);
18183
+ }
18184
+ return { ifmt, code };
18185
+ }
17943
18186
  function processGlobals(records) {
17944
18187
  const sheets = [];
17945
18188
  let codePage = 1200;
17946
18189
  let encrypted = false;
18190
+ let date1904 = false;
18191
+ const customFormats = /* @__PURE__ */ new Map();
18192
+ const xfFmtIds = [];
17947
18193
  const firstBof = records[0];
17948
18194
  if (!firstBof || firstBof.opcode !== OP_BOF) {
17949
18195
  throw new KordocError("XLS: \uCCAB \uB808\uCF54\uB4DC\uAC00 BOF\uAC00 \uC544\uB2D8");
@@ -17966,21 +18212,41 @@ function processGlobals(records) {
17966
18212
  codePage = r.data.readUInt16LE(0);
17967
18213
  } else if (r.opcode === OP_FILEPASS) {
17968
18214
  encrypted = true;
18215
+ } else if (r.opcode === OP_DATE1904 && r.data.length >= 2) {
18216
+ date1904 = r.data.readUInt16LE(0) === 1;
18217
+ } else if (r.opcode === OP_FORMAT) {
18218
+ const f = decodeFormatRecord(r.data);
18219
+ if (f) customFormats.set(f.ifmt, f.code);
18220
+ } else if (r.opcode === OP_XF && r.data.length >= 4) {
18221
+ xfFmtIds.push(r.data.readUInt16LE(2));
17969
18222
  }
17970
18223
  i++;
17971
18224
  }
18225
+ const dateXfs = /* @__PURE__ */ new Map();
18226
+ for (let k = 0; k < xfFmtIds.length; k++) {
18227
+ const kind = dateKindOfFmt(xfFmtIds[k], customFormats);
18228
+ if (kind) dateXfs.set(k, kind);
18229
+ }
17972
18230
  const globalsRecords = records.slice(0, i);
17973
18231
  const sst = decodeSST(globalsRecords);
17974
- return { sheets, sst, codePage, encrypted, endIndex: i };
18232
+ return { sheets, sst, codePage, encrypted, dateXfs, date1904, endIndex: i };
17975
18233
  }
17976
18234
  function findSheetBofIndex(records, lbPlyPos) {
17977
18235
  const exact = records.findIndex(
17978
18236
  (r) => r.opcode === OP_BOF && r.offset === lbPlyPos
17979
18237
  );
17980
18238
  if (exact >= 0) return exact;
17981
- const bofIndices = records.map((r, idx) => r.opcode === OP_BOF ? idx : -1).filter((idx) => idx >= 0);
17982
- if (bofIndices.length === 0) return -1;
17983
- return bofIndices.length > 1 ? bofIndices[1] : -1;
18239
+ let best = -1;
18240
+ let bestOffset = Infinity;
18241
+ for (let idx = 1; idx < records.length; idx++) {
18242
+ const r = records[idx];
18243
+ if (r.opcode !== OP_BOF) continue;
18244
+ if (r.offset >= lbPlyPos && r.offset < bestOffset) {
18245
+ best = idx;
18246
+ bestOffset = r.offset;
18247
+ }
18248
+ }
18249
+ return best;
17984
18250
  }
17985
18251
  function cellValueToText(v) {
17986
18252
  if (v === null || v === void 0) return "";
@@ -17992,7 +18258,7 @@ function cellValueToText(v) {
17992
18258
  if (typeof v === "boolean") return v ? "TRUE" : "FALSE";
17993
18259
  return v;
17994
18260
  }
17995
- function sheetToBlocks2(sheetName, sheet, sheetIndex) {
18261
+ function sheetToBlocks2(sheetName, sheet, sheetIndex, keepAnchoredEmptyCols) {
17996
18262
  const blocks = [];
17997
18263
  if (sheetName) {
17998
18264
  blocks.push({
@@ -18065,7 +18331,7 @@ function sheetToBlocks2(sheetName, sheet, sheetIndex) {
18065
18331
  cellRows.push(row);
18066
18332
  }
18067
18333
  if (cellRows.length > 0) {
18068
- const table2 = buildTable(cellRows);
18334
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols });
18069
18335
  if (table2.rows > 0) {
18070
18336
  blocks.push({ type: "table", table: table2, pageNumber: sheetIndex + 1 });
18071
18337
  }
@@ -18111,6 +18377,14 @@ async function parseXlsDocument(buffer, options) {
18111
18377
  ]
18112
18378
  };
18113
18379
  }
18380
+ const convertNum = globals.dateXfs.size > 0 ? (n, ixfe) => {
18381
+ const kind = globals.dateXfs.get(ixfe);
18382
+ if (kind) {
18383
+ const iso = dateSerialToIso(n, globals.date1904, kind);
18384
+ if (iso) return iso;
18385
+ }
18386
+ return n;
18387
+ } : void 0;
18114
18388
  const totalSheets = Math.min(globals.sheets.length, MAX_SHEETS2);
18115
18389
  let pageFilter = null;
18116
18390
  if (options?.pages) {
@@ -18137,8 +18411,8 @@ async function parseXlsDocument(buffer, options) {
18137
18411
  continue;
18138
18412
  }
18139
18413
  try {
18140
- const { sheet } = extractSheetCells(records, bofIdx, globals.sst);
18141
- const blocks = sheetToBlocks2(meta.name, sheet, i);
18414
+ const { sheet } = extractSheetCells(records, bofIdx, globals.sst, convertNum);
18415
+ const blocks = sheetToBlocks2(meta.name, sheet, i, options?.keepTrailingEmptyCols);
18142
18416
  allBlocks.push(...blocks);
18143
18417
  } catch (e) {
18144
18418
  warnings.push({
@@ -18265,6 +18539,9 @@ var NARY_MAP = {
18265
18539
  "\u2A02": "\\bigotimes",
18266
18540
  "\u2A00": "\\bigodot"
18267
18541
  };
18542
+ function onOffVal(v) {
18543
+ return v !== "0" && v !== "false" && v !== "off";
18544
+ }
18268
18545
  function mapDelim(ch, isLeft) {
18269
18546
  const l = {
18270
18547
  "(": "(",
@@ -18291,10 +18568,23 @@ function mapDelim(ch, isLeft) {
18291
18568
  const map = isLeft ? l : r;
18292
18569
  return map[ch] ?? ch;
18293
18570
  }
18571
+ function isBalancedWrap(s) {
18572
+ let depth = 0;
18573
+ for (let i = 0; i < s.length; i++) {
18574
+ const ch = s[i];
18575
+ if (ch === "{" && s[i - 1] !== "\\") depth++;
18576
+ else if (ch === "}" && s[i - 1] !== "\\") {
18577
+ depth--;
18578
+ if (depth === 0) return i === s.length - 1;
18579
+ if (depth < 0) return false;
18580
+ }
18581
+ }
18582
+ return false;
18583
+ }
18294
18584
  function grp(body) {
18295
18585
  const s = body.trim();
18296
18586
  if (s.length === 0) return "{}";
18297
- if (s.startsWith("{") && s.endsWith("}")) return s;
18587
+ if (s.startsWith("{") && s.endsWith("}") && isBalancedWrap(s)) return s;
18298
18588
  return "{" + s + "}";
18299
18589
  }
18300
18590
  function childrenToLatex(parent) {
@@ -18388,8 +18678,8 @@ function nodeToLatex(el) {
18388
18678
  }
18389
18679
  const sh = firstKid(naryPr, "subHide");
18390
18680
  const ph = firstKid(naryPr, "supHide");
18391
- if (sh) subHide = (sh.getAttribute("m:val") ?? sh.getAttribute("val")) !== "0";
18392
- if (ph) supHide = (ph.getAttribute("m:val") ?? ph.getAttribute("val")) !== "0";
18681
+ if (sh) subHide = onOffVal(sh.getAttribute("m:val") ?? sh.getAttribute("val"));
18682
+ if (ph) supHide = onOffVal(ph.getAttribute("m:val") ?? ph.getAttribute("val"));
18393
18683
  const ll = firstKid(naryPr, "limLoc");
18394
18684
  if (ll) limLoc = ll.getAttribute("m:val") ?? ll.getAttribute("val") ?? "";
18395
18685
  }
@@ -18536,7 +18826,7 @@ function isDisplayMath(el) {
18536
18826
  }
18537
18827
 
18538
18828
  // src/docx/parser.ts
18539
- var MAX_DECOMPRESS_SIZE4 = 100 * 1024 * 1024;
18829
+ var MAX_DECOMPRESS_SIZE5 = 100 * 1024 * 1024;
18540
18830
  function matchesLocal(el, localName2) {
18541
18831
  return el.localName === localName2 || (el.tagName?.endsWith(`:${localName2}`) ?? false);
18542
18832
  }
@@ -18554,6 +18844,8 @@ function effectiveChildElements(parent) {
18554
18844
  result.push(...effectiveChildElements(c));
18555
18845
  }
18556
18846
  }
18847
+ } else if (matchesLocal(el, "ins") || matchesLocal(el, "smartTag")) {
18848
+ result.push(...effectiveChildElements(el));
18557
18849
  } else {
18558
18850
  result.push(el);
18559
18851
  }
@@ -18618,6 +18910,21 @@ function parseStyles(xml) {
18618
18910
  }
18619
18911
  styles.set(styleId, { name, basedOn, outlineLevel });
18620
18912
  }
18913
+ for (const [styleId, info] of styles) {
18914
+ if (info.outlineLevel !== void 0) continue;
18915
+ const seen = /* @__PURE__ */ new Set([styleId]);
18916
+ let cur = info.basedOn;
18917
+ while (cur && !seen.has(cur)) {
18918
+ seen.add(cur);
18919
+ const parent = styles.get(cur);
18920
+ if (!parent) break;
18921
+ if (parent.outlineLevel !== void 0) {
18922
+ info.outlineLevel = parent.outlineLevel;
18923
+ break;
18924
+ }
18925
+ cur = parent.basedOn;
18926
+ }
18927
+ }
18621
18928
  return styles;
18622
18929
  }
18623
18930
  function parseNumbering(xml) {
@@ -18705,8 +19012,12 @@ function collectOmmlRoots(p) {
18705
19012
  return out;
18706
19013
  }
18707
19014
  function extractRun(r) {
18708
- const tElements = getChildElements(r, "t");
18709
- const text = tElements.map((t) => t.textContent ?? "").join("");
19015
+ let text = "";
19016
+ for (const el of effectiveChildElements(r)) {
19017
+ if (matchesLocal(el, "t")) text += el.textContent ?? "";
19018
+ else if (matchesLocal(el, "br") || matchesLocal(el, "cr")) text += "\n";
19019
+ else if (matchesLocal(el, "tab")) text += " ";
19020
+ }
18710
19021
  let bold = false;
18711
19022
  let italic = false;
18712
19023
  const rPrEls = getChildElements(r, "rPr");
@@ -18861,7 +19172,7 @@ function collectTextboxParagraphs(node, inTxbx = false, out = [], depth = 0) {
18861
19172
  }
18862
19173
  return out;
18863
19174
  }
18864
- function parseTable(tbl2, styles, numbering, footnotes, rels) {
19175
+ function parseTable(tbl2, styles, numbering, footnotes, rels, keepEmptyCols) {
18865
19176
  const trElements = getChildElements(tbl2, "tr");
18866
19177
  if (trElements.length === 0) return null;
18867
19178
  const rawRows = [];
@@ -18923,7 +19234,7 @@ ${cell2.text}` : cell2.text;
18923
19234
  return { text: cell2.text, colSpan: cell2.colSpan, rowSpan, colAddr: cell2.col, rowAddr: r };
18924
19235
  })
18925
19236
  );
18926
- const table2 = buildTable(cellRows);
19237
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: keepEmptyCols });
18927
19238
  if (table2.rows === 0 || table2.cols === 0) return null;
18928
19239
  return { type: "table", table: table2 };
18929
19240
  }
@@ -19025,7 +19336,7 @@ function emitParagraphImages(p, imageMap, linked, out) {
19025
19336
  }
19026
19337
  }
19027
19338
  async function parseDocxDocument(buffer, options) {
19028
- precheckZipSize(buffer, MAX_DECOMPRESS_SIZE4);
19339
+ precheckZipSize(buffer, MAX_DECOMPRESS_SIZE5);
19029
19340
  const zip = await JSZip4.loadAsync(buffer);
19030
19341
  const warnings = [];
19031
19342
  const docFile = zip.file("word/document.xml");
@@ -19096,7 +19407,7 @@ async function parseDocxDocument(buffer, options) {
19096
19407
  if (imageMap.size > 0) emitParagraphImages(tp, imageMap, linkedImages, blocks);
19097
19408
  }
19098
19409
  } else if (localName2 === "tbl") {
19099
- const block = parseTable(el, styles, numbering, footnotes, rels);
19410
+ const block = parseTable(el, styles, numbering, footnotes, rels, options?.keepTrailingEmptyCols);
19100
19411
  if (block) blocks.push(block);
19101
19412
  }
19102
19413
  }
@@ -19188,7 +19499,7 @@ function parseHwpmlDocument(buffer, options) {
19188
19499
  if (localName(el) !== "SECTION") continue;
19189
19500
  sectionIdx++;
19190
19501
  if (pageFilter && !pageFilter.has(sectionIdx)) continue;
19191
- parseSection2(el, blocks, paraShapeMap, sectionIdx, warnings);
19502
+ parseSection2(el, blocks, paraShapeMap, sectionIdx, warnings, options?.keepTrailingEmptyCols ?? false);
19192
19503
  }
19193
19504
  const outline = blocks.filter((b) => b.type === "heading" && b.text).map((b) => ({ level: b.level ?? 1, text: b.text, pageNumber: b.pageNumber }));
19194
19505
  const markdown = blocksToMarkdown(blocks);
@@ -19224,10 +19535,10 @@ function buildParaShapeMap(root) {
19224
19535
  }
19225
19536
  return map;
19226
19537
  }
19227
- function parseSection2(section, blocks, paraShapeMap, sectionNum, warnings) {
19228
- walkContent(section, blocks, paraShapeMap, sectionNum, warnings, false);
19538
+ function parseSection2(section, blocks, paraShapeMap, sectionNum, warnings, keep) {
19539
+ walkContent(section, blocks, paraShapeMap, sectionNum, warnings, false, keep);
19229
19540
  }
19230
- function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth = 0) {
19541
+ function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth = 0) {
19231
19542
  if (depth > MAX_XML_DEPTH2) return;
19232
19543
  const children = node.childNodes;
19233
19544
  for (let i = 0; i < children.length; i++) {
@@ -19240,24 +19551,24 @@ function walkContent(node, blocks, paraShapeMap, sectionNum, warnings, inHeaderF
19240
19551
  if (tag === "P") {
19241
19552
  if (!inHeaderFooter) {
19242
19553
  parseParagraph3(el, blocks, paraShapeMap, sectionNum);
19243
- walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings);
19554
+ walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19244
19555
  }
19245
19556
  continue;
19246
19557
  }
19247
19558
  if (tag === "TABLE") {
19248
19559
  if (!inHeaderFooter) {
19249
- parseTable2(el, blocks, paraShapeMap, sectionNum, warnings);
19560
+ parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19250
19561
  }
19251
19562
  continue;
19252
19563
  }
19253
19564
  if (tag === "PARALIST" || tag === "SECTION" || tag === "COLDEF") {
19254
- walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth + 1);
19565
+ walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth + 1);
19255
19566
  continue;
19256
19567
  }
19257
- walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, depth + 1);
19568
+ walkContent(el, blocks, paraShapeMap, sectionNum, warnings, inHeaderFooter, keep, depth + 1);
19258
19569
  }
19259
19570
  }
19260
- function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, depth = 0) {
19571
+ function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, keep, depth = 0) {
19261
19572
  if (depth > MAX_XML_DEPTH2) return;
19262
19573
  const children = node.childNodes;
19263
19574
  for (let i = 0; i < children.length; i++) {
@@ -19265,11 +19576,11 @@ function walkTablesInP(node, blocks, paraShapeMap, sectionNum, warnings, depth =
19265
19576
  if (el.nodeType !== 1) continue;
19266
19577
  const tag = localName(el);
19267
19578
  if (tag === "TABLE") {
19268
- parseTable2(el, blocks, paraShapeMap, sectionNum, warnings);
19579
+ parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep);
19269
19580
  continue;
19270
19581
  }
19271
19582
  if (tag === "FOOTNOTE" || tag === "ENDNOTE" || tag === "HEADER" || tag === "FOOTER") continue;
19272
- walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, depth + 1);
19583
+ walkTablesInP(el, blocks, paraShapeMap, sectionNum, warnings, keep, depth + 1);
19273
19584
  }
19274
19585
  }
19275
19586
  function parseParagraph3(el, blocks, paraShapeMap, sectionNum) {
@@ -19305,7 +19616,7 @@ function collectCharText(node, parts, depth = 0) {
19305
19616
  }
19306
19617
  }
19307
19618
  }
19308
- function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings) {
19619
+ function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings, keep) {
19309
19620
  const cells = [];
19310
19621
  const rowCount = parseInt(el.getAttribute("RowCount") ?? "0", 10);
19311
19622
  const colCount = parseInt(el.getAttribute("ColCount") ?? "0", 10);
@@ -19322,8 +19633,8 @@ function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings) {
19322
19633
  for (let j = 0; j < rowCells.length; j++) {
19323
19634
  const cellEl = rowCells[j];
19324
19635
  if (cellEl.nodeType !== 1 || localName(cellEl) !== "CELL") continue;
19325
- const colAddr = parseInt(cellEl.getAttribute("ColAddr") ?? "0", 10);
19326
- const rowAddr = parseInt(cellEl.getAttribute("RowAddr") ?? "0", 10);
19636
+ const colAddr = parseInt(cellEl.getAttribute("ColAddr") ?? "0", 10) || 0;
19637
+ const rowAddr = parseInt(cellEl.getAttribute("RowAddr") ?? "0", 10) || 0;
19327
19638
  const colSpan = Math.min(Math.max(1, parseInt(cellEl.getAttribute("ColSpan") ?? "1", 10) || 1), MAX_TABLE_COLS);
19328
19639
  const rowSpan = Math.min(Math.max(1, parseInt(cellEl.getAttribute("RowSpan") ?? "1", 10) || 1), MAX_TABLE_ROWS);
19329
19640
  const cellText = extractCellText(cellEl);
@@ -19331,25 +19642,12 @@ function parseTable2(el, blocks, paraShapeMap, sectionNum, warnings) {
19331
19642
  }
19332
19643
  }
19333
19644
  if (cells.length === 0) return;
19334
- const grid = Array.from({ length: rowCount }, () => Array(colCount).fill(null));
19645
+ const cellRows = Array.from({ length: rowCount }, () => []);
19335
19646
  for (const cell2 of cells) {
19336
- const r = cell2.rowAddr ?? 0;
19337
- const c = cell2.colAddr ?? 0;
19338
- if (isNaN(r) || isNaN(c) || r >= rowCount || c >= colCount) continue;
19339
- grid[r][c] = cell2;
19340
- for (let dr = 0; dr < cell2.rowSpan; dr++) {
19341
- for (let dc = 0; dc < cell2.colSpan; dc++) {
19342
- if (dr === 0 && dc === 0) continue;
19343
- if (r + dr < rowCount && c + dc < colCount) {
19344
- grid[r + dr][c + dc] = { text: "", colSpan: 1, rowSpan: 1 };
19345
- }
19346
- }
19347
- }
19647
+ const r = Math.min(Math.max(cell2.rowAddr ?? 0, 0), rowCount - 1);
19648
+ cellRows[r].push(cell2);
19348
19649
  }
19349
- const cellRows = grid.map(
19350
- (row) => row.map((cell2) => cell2 ?? { text: "", colSpan: 1, rowSpan: 1 })
19351
- );
19352
- const table2 = buildTable(cellRows);
19650
+ const table2 = buildTable(cellRows, { keepAnchoredEmptyCols: keep });
19353
19651
  const caption = extractShapeCaption(el);
19354
19652
  if (caption.text && caption.before) {
19355
19653
  blocks.push({ type: "paragraph", text: caption.text, pageNumber: sectionNum });
@@ -19585,7 +19883,7 @@ function findMatchingKey(cellLabel, values) {
19585
19883
  bestKey = key;
19586
19884
  }
19587
19885
  } else if (key.startsWith(cellLabel)) {
19588
- if (cellLabel.length >= key.length * 0.6 && cellLabel.length > bestLen) {
19886
+ if (cellLabel.length >= key.length * 0.75 && cellLabel.length > bestLen) {
19589
19887
  bestLen = cellLabel.length;
19590
19888
  bestKey = key;
19591
19889
  }
@@ -19625,7 +19923,7 @@ function fillInCellPatterns(cellText, values, matchedLabels, blockedLabels) {
19625
19923
  const matchKey = values.available(normalizedKw) ? normalizedKw : void 0;
19626
19924
  if (matchKey === void 0) return match;
19627
19925
  const val = values.peek(matchKey);
19628
- const isTruthy = ["\u2611", "\u2713", "\u2714", "v", "V", "true", "1", "yes", "o", "O"].includes(val.trim()) || val.trim() === "";
19926
+ const isTruthy = ["\u2611", "\u2713", "\u2714", "v", "V", "true", "1", "yes", "o", "O"].includes(val.trim());
19629
19927
  if (!isTruthy) return match;
19630
19928
  values.consume(matchKey);
19631
19929
  matchedLabels.add(matchKey);
@@ -19647,14 +19945,20 @@ function fillInCellPatterns(cellText, values, matchedLabels, blockedLabels) {
19647
19945
  );
19648
19946
  return matches.length > 0 ? { text, matches } : null;
19649
19947
  }
19650
- var INLINE_LABEL_RE = /([가-힣A-Za-z]{2,10})\s*[::]/g;
19948
+ var INLINE_LABEL_RE = /((?:[가-힣A-Za-z]{1,10} )?)([가-힣A-Za-z]{2,10})\s*[::]/g;
19651
19949
  function scanInlineSegments(text) {
19652
19950
  const labels = [];
19653
19951
  INLINE_LABEL_RE.lastIndex = 0;
19654
19952
  let m;
19655
19953
  while ((m = INLINE_LABEL_RE.exec(text)) !== null) {
19656
19954
  if (text[INLINE_LABEL_RE.lastIndex] === "/") continue;
19657
- labels.push({ label: m[1], start: m.index, end: INLINE_LABEL_RE.lastIndex });
19955
+ labels.push({
19956
+ label: m[2],
19957
+ ext: m[1] ? m[1] + m[2] : void 0,
19958
+ extStart: m[1] ? m.index : void 0,
19959
+ start: m.index + m[1].length,
19960
+ end: INLINE_LABEL_RE.lastIndex
19961
+ });
19658
19962
  }
19659
19963
  const segments = [];
19660
19964
  for (let i = 0; i < labels.length; i++) {
@@ -19672,21 +19976,44 @@ function scanInlineSegments(text) {
19672
19976
  labelStart: cur.start,
19673
19977
  valueStart: vs,
19674
19978
  valueEnd: ve,
19675
- value: text.slice(vs, ve)
19979
+ value: text.slice(vs, ve),
19980
+ ...cur.ext !== void 0 ? { extLabel: cur.ext, extStart: cur.extStart } : {}
19676
19981
  });
19677
19982
  }
19678
19983
  return segments;
19679
19984
  }
19985
+ function matchInlineSegment(seg, values, blockedLabels) {
19986
+ const nlabel = normalizeLabel(seg.label);
19987
+ if (seg.extLabel !== void 0) {
19988
+ const nExt = normalizeLabel(seg.extLabel);
19989
+ if (nExt !== nlabel && !blockedLabels?.has(nExt) && values.has(nExt)) {
19990
+ return { key: nExt, label: seg.extLabel, viaExt: true };
19991
+ }
19992
+ }
19993
+ if (blockedLabels?.has(nlabel)) return void 0;
19994
+ const key = findMatchingKey(nlabel, values);
19995
+ return key !== void 0 ? { key, label: seg.label, viaExt: false } : void 0;
19996
+ }
19997
+ function clampSegmentEnd(text, seg, next, nextViaExt) {
19998
+ let ve = seg.valueEnd;
19999
+ if (nextViaExt && next?.extStart !== void 0 && next.extStart < ve) ve = next.extStart;
20000
+ while (ve > seg.valueStart && /\s/.test(text[ve - 1])) ve--;
20001
+ return ve;
20002
+ }
19680
20003
  function padInsertion(text, pos, value) {
19681
20004
  const lead = pos > 0 && !/\s/.test(text[pos - 1]) ? " " : "";
19682
20005
  const trail = pos < text.length && !/\s/.test(text[pos]) ? " " : "";
19683
20006
  return lead + value + trail;
19684
20007
  }
19685
- function normalizeValues(values) {
20008
+ function normalizeValues(values, warnings) {
19686
20009
  const map = /* @__PURE__ */ new Map();
19687
20010
  for (const [label, raw] of Object.entries(values)) {
19688
20011
  const { value, format } = typeof raw === "object" && !Array.isArray(raw) ? raw : { value: raw, format: void 0 };
19689
- map.set(normalizeLabel(label), Array.isArray(value) ? value.map((v) => formatFillValue(v, format)) : formatFillValue(value, format));
20012
+ const key = normalizeLabel(label);
20013
+ if (map.has(key)) {
20014
+ warnings?.push(`\uC785\uB825 \uB77C\uBCA8 "${label}"\uC774 \uC815\uADDC\uD654 \uD0A4 "${key}"\uC5D0\uC11C \uB2E4\uB978 \uB77C\uBCA8\uACFC \uCDA9\uB3CC \u2014 \uB4A4 \uAC12\uC73C\uB85C \uB36E\uC5B4\uC500`);
20015
+ }
20016
+ map.set(key, Array.isArray(value) ? value.map((v) => formatFillValue(v, format)) : formatFillValue(value, format));
19690
20017
  }
19691
20018
  return map;
19692
20019
  }
@@ -19875,10 +20202,10 @@ function extractFromTable(table2) {
19875
20202
  }
19876
20203
  if (fields.length === 0 && table2.rows >= 2 && table2.cols >= 2) {
19877
20204
  const headerRow = table2.cells[0];
19878
- const allLabels = headerRow.every((cell2) => {
19879
- const t = cell2.text.trim();
20205
+ const allLabels = headerRow?.every((cell2) => {
20206
+ const t = cell2?.text.trim() ?? "";
19880
20207
  return t.length > 0 && t.length <= 20;
19881
- });
20208
+ }) ?? false;
19882
20209
  if (allLabels) {
19883
20210
  for (let r = 1; r < table2.rows; r++) {
19884
20211
  for (let c = 0; c < table2.cols; c++) {
@@ -19969,7 +20296,8 @@ function fillFormFields(blocks, values, blockedLabels) {
19969
20296
  const cloned = structuredClone(blocks);
19970
20297
  const filled = [];
19971
20298
  const matchedLabels = /* @__PURE__ */ new Set();
19972
- const normalizedValues = normalizeValues(values);
20299
+ const warnings = [];
20300
+ const normalizedValues = normalizeValues(values, warnings);
19973
20301
  const cursor = new ValueCursor(normalizedValues);
19974
20302
  const allTables = collectIRTables(cloned, 0);
19975
20303
  const patternFilledCells = /* @__PURE__ */ new Set();
@@ -19998,7 +20326,7 @@ function fillFormFields(blocks, values, blockedLabels) {
19998
20326
  if (newText !== block.text) block.text = newText;
19999
20327
  }
20000
20328
  const unmatched = resolveUnmatched(normalizedValues, matchedLabels, values);
20001
- return { blocks: cloned, filled, unmatched };
20329
+ return { blocks: cloned, filled, unmatched, ...warnings.length > 0 ? { warnings } : {} };
20002
20330
  }
20003
20331
  function collectIRTables(blocks, depth) {
20004
20332
  if (depth > 16) return [];
@@ -20032,13 +20360,31 @@ function coveredPositions(table2) {
20032
20360
  }
20033
20361
  return covered;
20034
20362
  }
20363
+ function isHeaderDataTable(table2, covered) {
20364
+ if (table2.rows < 2) return false;
20365
+ const headerRow = table2.cells[0];
20366
+ if (!headerRow?.length) return false;
20367
+ const allLabels = headerRow.every((cell2) => {
20368
+ const t = cell2?.text.trim() ?? "";
20369
+ return t.length > 0 && t.length <= 20 && isLabelCell(t);
20370
+ });
20371
+ if (!allLabels) return false;
20372
+ for (let c = 0; c < table2.cols; c++) {
20373
+ if (covered.has(`1,${c}`)) continue;
20374
+ const cell2 = table2.cells[1]?.[c];
20375
+ if (!cell2) break;
20376
+ return !isLabelCell(cell2.text);
20377
+ }
20378
+ return true;
20379
+ }
20035
20380
  function fillTable(table2, values, filled, matchedLabels, patternFilledCells, blockedLabels) {
20036
20381
  if (table2.cols < 2) return;
20037
20382
  const covered = coveredPositions(table2);
20038
- for (let r = 0; r < table2.rows; r++) {
20383
+ const skipHeaderRow = isHeaderDataTable(table2, covered);
20384
+ for (let r = skipHeaderRow ? 1 : 0; r < table2.rows; r++) {
20039
20385
  for (let c = 0; c < table2.cols; c++) {
20040
20386
  if (covered.has(`${r},${c}`)) continue;
20041
- const labelCell = table2.cells[r][c];
20387
+ const labelCell = table2.cells[r]?.[c];
20042
20388
  if (!labelCell) continue;
20043
20389
  if (!isLabelCell(labelCell.text)) continue;
20044
20390
  let vc = c + labelCell.colSpan;
@@ -20071,10 +20417,10 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20071
20417
  }
20072
20418
  if (table2.rows >= 2 && table2.cols >= 2) {
20073
20419
  const headerRow = table2.cells[0];
20074
- const allLabels = headerRow.every((cell2) => {
20075
- const t = cell2.text.trim();
20420
+ const allLabels = headerRow?.every((cell2) => {
20421
+ const t = cell2?.text.trim() ?? "";
20076
20422
  return t.length > 0 && t.length <= 20 && isLabelCell(t);
20077
- });
20423
+ }) ?? false;
20078
20424
  if (!allLabels) return;
20079
20425
  for (let r = 1; r < table2.rows; r++) {
20080
20426
  for (let c = 0; c < table2.cols; c++) {
@@ -20089,7 +20435,11 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20089
20435
  if (!values.isArray(matchKey) && matchedLabels.has(matchKey)) continue;
20090
20436
  const newValue = values.consume(matchKey);
20091
20437
  if (newValue === void 0) continue;
20092
- valueCell.text = newValue;
20438
+ if (patternFilledCells?.has(valueCell)) {
20439
+ valueCell.text = newValue + " " + valueCell.text;
20440
+ } else {
20441
+ valueCell.text = newValue;
20442
+ }
20093
20443
  matchedLabels.add(matchKey);
20094
20444
  filled.push({
20095
20445
  label: headerCell.text.trim(),
@@ -20105,20 +20455,22 @@ function fillTable(table2, values, filled, matchedLabels, patternFilledCells, bl
20105
20455
  function fillInlineFields(text, values, filled, matchedLabels, blockedLabels) {
20106
20456
  const segments = scanInlineSegments(text);
20107
20457
  if (segments.length === 0) return text;
20458
+ const matches = segments.map((seg) => matchInlineSegment(seg, values, blockedLabels));
20108
20459
  let out = "";
20109
20460
  let pos = 0;
20110
- for (const seg of segments) {
20111
- const nlabel = normalizeLabel(seg.label);
20112
- if (blockedLabels?.has(nlabel)) continue;
20113
- const matchKey = findMatchingKey(nlabel, values);
20114
- if (matchKey === void 0) continue;
20461
+ for (let i = 0; i < segments.length; i++) {
20462
+ const seg = segments[i];
20463
+ const matched = matches[i];
20464
+ if (matched === void 0) continue;
20465
+ const matchKey = matched.key;
20115
20466
  const newValue = values.consume(matchKey);
20116
20467
  if (newValue === void 0) continue;
20117
20468
  matchedLabels.add(matchKey);
20118
- filled.push({ label: seg.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20469
+ filled.push({ label: matched.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20470
+ const ve = clampSegmentEnd(text, seg, segments[i + 1], matches[i + 1]?.viaExt ?? false);
20119
20471
  out += text.slice(pos, seg.valueStart);
20120
- out += seg.valueStart === seg.valueEnd ? padInsertion(text, seg.valueStart, newValue) : newValue;
20121
- pos = seg.valueEnd;
20472
+ out += seg.valueStart === ve ? padInsertion(text, seg.valueStart, newValue) : newValue;
20473
+ pos = ve;
20122
20474
  }
20123
20475
  out += text.slice(pos);
20124
20476
  return out;
@@ -20133,7 +20485,8 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20133
20485
  if (sectionPaths.length === 0) {
20134
20486
  throw new KordocError("HWPX\uC5D0\uC11C \uC139\uC158 \uD30C\uC77C\uC744 \uCC3E\uC744 \uC218 \uC5C6\uC2B5\uB2C8\uB2E4");
20135
20487
  }
20136
- const normalizedValues = normalizeValues(values);
20488
+ const warnings = [];
20489
+ const normalizedValues = normalizeValues(values, warnings);
20137
20490
  const cursor = new ValueCursor(normalizedValues);
20138
20491
  const matchedLabels = /* @__PURE__ */ new Set();
20139
20492
  const filled = [];
@@ -20195,7 +20548,17 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20195
20548
  }
20196
20549
  }
20197
20550
  for (const table2 of allTables) {
20198
- for (let rowIdx = 0; rowIdx < table2.rows.length; rowIdx++) {
20551
+ const skipHeaderRow = table2.rows.length >= 2 && (() => {
20552
+ const first = table2.rows[0];
20553
+ const allLabels = first.length > 0 && first.every((cell2) => {
20554
+ const t = cellLabelText(cell2).trim();
20555
+ return t.length > 0 && t.length <= 20 && isLabelCell(t);
20556
+ });
20557
+ if (!allLabels) return false;
20558
+ const d0 = table2.rows[1][0];
20559
+ return d0 === void 0 || !isLabelCell(cellLabelText(d0));
20560
+ })();
20561
+ for (let rowIdx = skipHeaderRow ? 1 : 0; rowIdx < table2.rows.length; rowIdx++) {
20199
20562
  const cells = table2.rows[rowIdx];
20200
20563
  for (let colIdx = 0; colIdx < cells.length - 1; colIdx++) {
20201
20564
  const labelText = cellLabelText(cells[colIdx]);
@@ -20266,9 +20629,30 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20266
20629
  const matchKey = findMatchingKey(headerLabel, cursor);
20267
20630
  if (matchKey === void 0) continue;
20268
20631
  if (!cursor.isArray(matchKey) && matchedLabels.has(matchKey)) continue;
20632
+ const dataCell = dataCells[colIdx];
20633
+ if (patternApplied.has(dataCell)) {
20634
+ const target = dataCell.paragraphs.find((p) => p.tRanges.length > 0) ?? dataCell.paragraphs[0];
20635
+ if (!target) continue;
20636
+ const l = led(target);
20637
+ if (l.fullText !== void 0) continue;
20638
+ const newValue2 = cursor.consume(matchKey);
20639
+ if (newValue2 === void 0) continue;
20640
+ l.ranges.push({ start: 0, end: 0, replacement: newValue2 + " " });
20641
+ l.filledIdx.push(filled.length);
20642
+ l.matchKeys.push(matchKey);
20643
+ matchedLabels.add(matchKey);
20644
+ filled.push({
20645
+ label: cellLabelText(headerCells[colIdx]).trim(),
20646
+ value: newValue2,
20647
+ row: rowIdx,
20648
+ col: colIdx,
20649
+ key: matchKey
20650
+ });
20651
+ continue;
20652
+ }
20269
20653
  const newValue = cursor.consume(matchKey);
20270
20654
  if (newValue === void 0) continue;
20271
- const paras = dataCells[colIdx].paragraphs;
20655
+ const paras = dataCell.paragraphs;
20272
20656
  if (paras.length === 0) continue;
20273
20657
  const l0 = led(paras[0]);
20274
20658
  l0.fullText = newValue;
@@ -20297,20 +20681,23 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20297
20681
  const existing = ledger.get(para2);
20298
20682
  if (existing?.fullText !== void 0) continue;
20299
20683
  const text = matchText(para2);
20300
- for (const seg of scanInlineSegments(text)) {
20301
- const nlabel = normalizeLabel(seg.label);
20302
- if (blockedLabels?.has(nlabel)) continue;
20303
- const matchKey = findMatchingKey(nlabel, cursor);
20304
- if (matchKey === void 0) continue;
20684
+ const segments = scanInlineSegments(text);
20685
+ const matches = segments.map((seg) => matchInlineSegment(seg, cursor, blockedLabels));
20686
+ for (let i = 0; i < segments.length; i++) {
20687
+ const seg = segments[i];
20688
+ const matched = matches[i];
20689
+ if (matched === void 0) continue;
20690
+ const matchKey = matched.key;
20305
20691
  const newValue = cursor.consume(matchKey);
20306
20692
  if (newValue === void 0) continue;
20307
- const replacement = seg.valueStart === seg.valueEnd ? padInsertion(text, seg.valueStart, newValue) : newValue;
20693
+ const ve = clampSegmentEnd(text, seg, segments[i + 1], matches[i + 1]?.viaExt ?? false);
20694
+ const replacement = seg.valueStart === ve ? padInsertion(text, seg.valueStart, newValue) : newValue;
20308
20695
  const l = led(para2);
20309
- l.ranges.push({ start: seg.valueStart, end: seg.valueEnd, replacement });
20696
+ l.ranges.push({ start: seg.valueStart, end: ve, replacement });
20310
20697
  matchedLabels.add(matchKey);
20311
20698
  l.filledIdx.push(filled.length);
20312
20699
  l.matchKeys.push(matchKey);
20313
- filled.push({ label: seg.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20700
+ filled.push({ label: matched.label.trim(), value: newValue, row: -1, col: -1, key: matchKey });
20314
20701
  }
20315
20702
  }
20316
20703
  const splices = [];
@@ -20363,7 +20750,7 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20363
20750
  }
20364
20751
  }
20365
20752
  for (const k of failedKeys) {
20366
- if (!succeededKeys.has(k)) matchedLabels.delete(k);
20753
+ if (!succeededKeys.has(k) || cursor.isArray(k)) matchedLabels.delete(k);
20367
20754
  }
20368
20755
  const cleanFilled = filled.filter((f) => f !== null);
20369
20756
  const unmatched = resolveUnmatched(normalizedValues, matchedLabels, values);
@@ -20371,7 +20758,8 @@ async function fillHwpx(hwpxBuffer, values, blockedLabels) {
20371
20758
  return {
20372
20759
  buffer: out.buffer.slice(out.byteOffset, out.byteOffset + out.byteLength),
20373
20760
  filled: cleanFilled,
20374
- unmatched
20761
+ unmatched,
20762
+ ...warnings.length > 0 ? { warnings } : {}
20375
20763
  };
20376
20764
  }
20377
20765
 
@@ -20648,27 +21036,39 @@ function parseMarkdownToBlocks(md2) {
20648
21036
  if (/^<table[\s>]/i.test(line.trimStart())) {
20649
21037
  const htmlLines = [];
20650
21038
  let depth = 0;
20651
- while (i < lines.length) {
20652
- const l = lines[i];
21039
+ let closed = false;
21040
+ let j = i;
21041
+ while (j < lines.length) {
21042
+ const l = lines[j];
20653
21043
  htmlLines.push(l);
20654
21044
  depth += (l.match(/<table[\s>]/gi) ?? []).length;
20655
21045
  depth -= (l.match(/<\/table>/gi) ?? []).length;
20656
- i++;
20657
- if (depth <= 0) break;
21046
+ j++;
21047
+ if (depth <= 0) {
21048
+ closed = true;
21049
+ break;
21050
+ }
21051
+ }
21052
+ if (closed) {
21053
+ blocks.push({ type: "html_table", text: htmlLines.join("\n") });
21054
+ i = j;
21055
+ continue;
20658
21056
  }
20659
- blocks.push({ type: "html_table", text: htmlLines.join("\n") });
20660
- continue;
20661
21057
  }
20662
21058
  if (line.trimStart().startsWith("|")) {
20663
21059
  const tableRows = [];
21060
+ let sepSeen = false;
20664
21061
  while (i < lines.length && lines[i].trimStart().startsWith("|")) {
20665
21062
  const row = lines[i];
20666
- const sepCells = row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|");
20667
- if (sepCells.every((c) => /^\s*:?-+:?\s*$/.test(c))) {
20668
- i++;
20669
- continue;
21063
+ if (tableRows.length === 1 && !sepSeen) {
21064
+ const sepCells = row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|");
21065
+ if (sepCells.every((c) => /^\s*:?-+:?\s*$/.test(c))) {
21066
+ sepSeen = true;
21067
+ i++;
21068
+ continue;
21069
+ }
20670
21070
  }
20671
- const cells = row.split("|").slice(1, -1).map((c) => c.trim());
21071
+ const cells = row.split(/(?<!\\)\|/).slice(1, -1).map((c) => c.trim().replace(/\\\|/g, "|"));
20672
21072
  if (cells.length > 0) tableRows.push(cells);
20673
21073
  i++;
20674
21074
  }
@@ -21617,13 +22017,26 @@ function normalize(s) {
21617
22017
  return s.replace(/\s+/g, " ").trim();
21618
22018
  }
21619
22019
  var MAX_LEVENSHTEIN_LEN = 1e4;
22020
+ function approxDistance(a, b) {
22021
+ const bigramCounts = (s) => {
22022
+ const m = /* @__PURE__ */ new Map();
22023
+ for (let i = 0; i < s.length - 1; i++) {
22024
+ const g = s.slice(i, i + 2);
22025
+ m.set(g, (m.get(g) ?? 0) + 1);
22026
+ }
22027
+ return m;
22028
+ };
22029
+ const ca = bigramCounts(a);
22030
+ const cb = bigramCounts(b);
22031
+ let inter = 0;
22032
+ for (const [g, n] of ca) inter += Math.min(n, cb.get(g) ?? 0);
22033
+ const total = Math.max(a.length - 1, 0) + Math.max(b.length - 1, 0);
22034
+ const dice = total > 0 ? 2 * inter / total : 1;
22035
+ return Math.round(Math.max(a.length, b.length) * (1 - dice));
22036
+ }
21620
22037
  function levenshtein(a, b) {
21621
22038
  if (a.length + b.length > MAX_LEVENSHTEIN_LEN) {
21622
- const sampleLen = Math.min(500, a.length, b.length);
21623
- let diffs = 0;
21624
- for (let i = 0; i < sampleLen; i++) if (a[i] !== b[i]) diffs++;
21625
- const sampleRate = sampleLen > 0 ? diffs / sampleLen : 1;
21626
- return Math.abs(a.length - b.length) + Math.round(Math.min(a.length, b.length) * sampleRate);
22039
+ return approxDistance(a, b);
21627
22040
  }
21628
22041
  if (a.length > b.length) [a, b] = [b, a];
21629
22042
  const m = a.length;
@@ -21775,9 +22188,16 @@ function bestSimInRange(arr, from, to, target) {
21775
22188
  return best;
21776
22189
  }
21777
22190
  function escapeGfm(text) {
21778
- return text.replace(/([~*])/g, "\\$1");
22191
+ const NUL = String.fromCharCode(0);
22192
+ const spans = [];
22193
+ const masked = text.replace(/!\[[^\]]*\]\([^)\n]*\)|\]\((?:https?:|mailto:|tel:|#)[^)\n]*\)/gi, (m) => {
22194
+ spans.push(m);
22195
+ return NUL + (spans.length - 1) + NUL;
22196
+ });
22197
+ const escaped = masked.replace(/([~*_`])/g, "\\$1");
22198
+ return escaped.replace(new RegExp(NUL + "(\\d+)" + NUL, "g"), (_, n) => spans[Number(n)]);
21779
22199
  }
21780
- var HWP_SHAPE_ALT_TEXT_RE = /(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|이등변 삼각형|직각 삼각형|선|직선|곡선|화살표|굵은 화살표|이중 화살표|오각형|육각형|팔각형|별|[4-8]점별|십자|십자형|구름|구름형|마름모|도넛|평행사변형|사다리꼴|부채꼴|호|반원|물결|번개|하트|빗금|블록 화살표|수식|표|그림|개체|그리기\s?개체|묶음\s?개체|글상자|수식\s?개체|OLE\s?개체)\s?입니다\.?/g;
22200
+ var HWP_SHAPE_ALT_TEXT_RE = /^(?:모서리가 둥근 |둥근 )?(?:사각형|직사각형|정사각형|원|타원|삼각형|이등변 삼각형|직각 삼각형|선|직선|곡선|화살표|굵은 화살표|이중 화살표|오각형|육각형|팔각형|별|[4-8]점별|십자|십자형|구름|구름형|마름모|도넛|평행사변형|사다리꼴|부채꼴|호|반원|물결|번개|하트|빗금|블록 화살표|수식|표|그림|개체|그리기\s?개체|묶음\s?개체|글상자|수식\s?개체|OLE\s?개체)\s?입니다\.?$/gm;
21781
22201
  function sanitizeText(text) {
21782
22202
  let result = mapPuaText(text).replace(/[\u{F0000}-\u{FFFFD}]/gu, "").replace(HWP_SHAPE_ALT_TEXT_RE, "").replace(/ +/g, " ").trim();
21783
22203
  if (result.length <= 30 && result.includes(" ")) {
@@ -21793,7 +22213,7 @@ function normForMatch(text) {
21793
22213
  return sanitizeText(text).replace(/\s+/g, " ").trim();
21794
22214
  }
21795
22215
  function unescapeGfm(text) {
21796
- return text.replace(/\\([~*])/g, "$1");
22216
+ return text.replace(/\\([~*_`])/g, "$1");
21797
22217
  }
21798
22218
  function summarize(text) {
21799
22219
  const t = text.replace(/\s+/g, " ").trim();
@@ -21860,7 +22280,7 @@ function parseGfmTable(lines) {
21860
22280
  return rows;
21861
22281
  }
21862
22282
  function unescapeGfmCell(text) {
21863
- return text.replace(/<br\s*\/?>/gi, "\n").replace(/\\\|/g, "|").replace(/\\([~*])/g, "$1");
22283
+ return text.replace(/<br\s*\/?>/gi, "\n").replace(/\\\|/g, "|").replace(/\\([~*_`])/g, "$1");
21864
22284
  }
21865
22285
  function replicateCellInnerHtml(cell2) {
21866
22286
  if (cell2.blocks?.length) {
@@ -21948,10 +22368,10 @@ function parseHtmlTable(raw) {
21948
22368
  }
21949
22369
  } else {
21950
22370
  if (!isClose) {
21951
- const cs = parseInt(attrs.match(/colspan\s*=\s*"(\d+)"/i)?.[1] || "1", 10);
21952
- const rs = parseInt(attrs.match(/rowspan\s*=\s*"(\d+)"/i)?.[1] || "1", 10);
22371
+ const cs = parseInt(attrs.match(/colspan\s*=\s*["']?(\d+)/i)?.[1] || "1", 10);
22372
+ const rs = parseInt(attrs.match(/rowspan\s*=\s*["']?(\d+)/i)?.[1] || "1", 10);
21953
22373
  cellStart = m.index + m[0].length;
21954
- cellInfo = { colSpan: isNaN(cs) ? 1 : cs, rowSpan: isNaN(rs) ? 1 : rs };
22374
+ cellInfo = { colSpan: clampSpan(isNaN(cs) ? 1 : cs, MAX_COLS), rowSpan: clampSpan(isNaN(rs) ? 1 : rs, MAX_ROWS) };
21955
22375
  } else if (cellStart >= 0 && cellInfo && currentRow) {
21956
22376
  currentRow.push({ inner: raw.slice(cellStart, m.index), colSpan: cellInfo.colSpan, rowSpan: cellInfo.rowSpan });
21957
22377
  cellStart = -1;
@@ -22595,8 +23015,8 @@ function layoutHtmlRows(rows) {
22595
23015
  let c = 0;
22596
23016
  for (const cell2 of rows[r].cells) {
22597
23017
  while (occupied.has(`${r},${c}`)) c++;
22598
- const colSpan = Math.max(1, cell2.colSpan);
22599
- const rowSpan = Math.max(1, cell2.rowSpan);
23018
+ const colSpan = clampSpan(cell2.colSpan, MAX_COLS);
23019
+ const rowSpan = clampSpan(cell2.rowSpan, MAX_ROWS);
22600
23020
  placed.push({ r, c, colSpan, rowSpan, inner: cell2.inner, isHeader: rows[r].tag === "th" });
22601
23021
  for (let dr = 0; dr < rowSpan; dr++) {
22602
23022
  for (let dc = 0; dc < colSpan; dc++) occupied.add(`${r + dr},${c + dc}`);
@@ -23755,12 +24175,20 @@ function diffBlocks(blocksA, blocksB) {
23755
24175
  function alignBlocks(a, b) {
23756
24176
  const m = a.length, n = b.length;
23757
24177
  if (m * n > 1e7) return fallbackAlign(a, b);
24178
+ const lenOf = (blk) => {
24179
+ const t = blk.text !== void 0 ? blk.text : blk.type === "table" && blk.table ? blk.table.cells.flat().map((c) => c?.text ?? "").join(" ") : "";
24180
+ return t.replace(/\s+/g, " ").trim().length;
24181
+ };
24182
+ const aLen = a.map(lenOf);
24183
+ const bLen = b.map(lenOf);
23758
24184
  const simCache = /* @__PURE__ */ new Map();
23759
24185
  const getSim = (i2, j2) => {
23760
24186
  const key = `${i2},${j2}`;
23761
24187
  let v = simCache.get(key);
23762
24188
  if (v === void 0) {
23763
- v = blockSimilarity(a[i2], b[j2]);
24189
+ const mx = Math.max(aLen[i2], bLen[j2]);
24190
+ const cut = a[i2].type === "table" || b[j2].type === "table" ? 6 / 7 : 1 - SIMILARITY_THRESHOLD;
24191
+ v = mx > 0 && (mx - Math.min(aLen[i2], bLen[j2])) / mx > cut ? 0 : blockSimilarity(a[i2], b[j2]);
23764
24192
  simCache.set(key, v);
23765
24193
  }
23766
24194
  return v;
@@ -23821,8 +24249,8 @@ function blockSimilarity(a, b) {
23821
24249
  }
23822
24250
  function tableSimilarity(a, b) {
23823
24251
  const dimSim = 1 - Math.abs(a.rows * a.cols - b.rows * b.cols) / Math.max(a.rows * a.cols, b.rows * b.cols, 1);
23824
- const textsA = a.cells.flat().map((c) => c.text).join(" ");
23825
- const textsB = b.cells.flat().map((c) => c.text).join(" ");
24252
+ const textsA = a.cells.flat().map((c) => c?.text ?? "").join(" ");
24253
+ const textsB = b.cells.flat().map((c) => c?.text ?? "").join(" ");
23826
24254
  const contentSim = normalizedSimilarity(textsA, textsB);
23827
24255
  return dimSim * 0.3 + contentSim * 0.7;
23828
24256
  }
@@ -23833,8 +24261,8 @@ function diffTableCells(a, b) {
23833
24261
  for (let r = 0; r < maxRows; r++) {
23834
24262
  const row = [];
23835
24263
  for (let c = 0; c < maxCols; c++) {
23836
- const cellA = r < a.rows && c < a.cols ? a.cells[r][c].text : void 0;
23837
- const cellB = r < b.rows && c < b.cols ? b.cells[r][c].text : void 0;
24264
+ const cellA = r < a.rows && c < a.cols ? a.cells[r]?.[c]?.text : void 0;
24265
+ const cellB = r < b.rows && c < b.cols ? b.cells[r]?.[c]?.text : void 0;
23838
24266
  let type;
23839
24267
  if (cellA === void 0) type = "added";
23840
24268
  else if (cellB === void 0) type = "removed";
@@ -25371,6 +25799,9 @@ var Surgeon = class {
25371
25799
  miniFatSectors = [];
25372
25800
  dirSectors = [];
25373
25801
  entries = [];
25802
+ /** replace()가 FREESECT로 해제한 섹터 — finish()에서 재할당 안 된 것만 0으로 지움 (데이터 잔존 방지) */
25803
+ freedSectors = [];
25804
+ freedMiniSectors = [];
25374
25805
  constructor(file) {
25375
25806
  if (file.length < SECTOR || file.readUInt32LE(0) !== 3759263696) {
25376
25807
  throw new OleSurgeonError("OLE \uC2DC\uADF8\uB2C8\uCC98\uAC00 \uC544\uB2D9\uB2C8\uB2E4");
@@ -25494,6 +25925,8 @@ var Surgeon = class {
25494
25925
  if (this.fat[i] !== FREESECT) continue;
25495
25926
  if (SECTOR + (i + 1) * SECTOR > this.buf.length) continue;
25496
25927
  this.fat[i] = ENDOFCHAIN;
25928
+ const off = this.sectorOffset(i);
25929
+ this.buf.fill(0, off, off + SECTOR);
25497
25930
  out.push(i);
25498
25931
  }
25499
25932
  while (out.length < n) {
@@ -25593,9 +26026,15 @@ var Surgeon = class {
25593
26026
  const entry = this.findEntry(path);
25594
26027
  if (entry.size > 0 && entry.start !== ENDOFCHAIN) {
25595
26028
  if (entry.size < MINI_CUTOFF) {
25596
- for (const s of this.miniChain(entry.start)) this.miniFat[s] = FREESECT;
26029
+ for (const s of this.miniChain(entry.start)) {
26030
+ this.miniFat[s] = FREESECT;
26031
+ this.freedMiniSectors.push(s);
26032
+ }
25597
26033
  } else {
25598
- for (const s of this.chain(entry.start)) this.fat[s] = FREESECT;
26034
+ for (const s of this.chain(entry.start)) {
26035
+ this.fat[s] = FREESECT;
26036
+ this.freedSectors.push(s);
26037
+ }
25599
26038
  }
25600
26039
  }
25601
26040
  if (newData.length < MINI_CUTOFF) {
@@ -25624,9 +26063,27 @@ var Surgeon = class {
25624
26063
  this.writeDirEntry(entry);
25625
26064
  }
25626
26065
  finish() {
26066
+ this.wipeFreedSectors();
25627
26067
  this.flushFat();
25628
26068
  return this.buf;
25629
26069
  }
26070
+ /** 해제 후 재할당되지 않고 남은 FREESECT 섹터의 바이트를 0으로 채움 (데이터 remanence 제거) */
26071
+ wipeFreedSectors() {
26072
+ for (const s of this.freedSectors) {
26073
+ if (this.fat[s] !== FREESECT) continue;
26074
+ const off = this.sectorOffset(s);
26075
+ this.buf.fill(0, off, off + SECTOR);
26076
+ }
26077
+ if (this.freedMiniSectors.length > 0) {
26078
+ const root = this.rootEntry();
26079
+ const rootChain = root.start === ENDOFCHAIN || root.size === 0 ? [] : this.chain(root.start);
26080
+ for (const s of this.freedMiniSectors) {
26081
+ if (this.miniFat[s] !== FREESECT) continue;
26082
+ const off = this.miniOffset(s, rootChain);
26083
+ this.buf.fill(0, off, off + MINI_SECTOR);
26084
+ }
26085
+ }
26086
+ }
25630
26087
  };
25631
26088
 
25632
26089
  // src/roundtrip/hwp5-patch.ts
@@ -27057,7 +27514,7 @@ async function parseHwp3(buffer, options) {
27057
27514
  const { markdown, blocks, metadata, outline, warnings } = parseHwp3Document(buffer, options);
27058
27515
  return { success: true, fileType: "hwp3", markdown, blocks, metadata, outline, warnings };
27059
27516
  } catch (err) {
27060
- return { success: false, fileType: "hwp3", error: err instanceof Error ? err.message : "HWP3 \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27517
+ return { success: false, fileType: "hwp3", error: sanitizeError(err), code: classifyError(err) };
27061
27518
  }
27062
27519
  }
27063
27520
  async function parseHwpx(buffer, options) {
@@ -27065,7 +27522,7 @@ async function parseHwpx(buffer, options) {
27065
27522
  const { markdown, blocks, metadata, outline, warnings, images } = await parseHwpxDocument(buffer, options);
27066
27523
  return { success: true, fileType: "hwpx", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
27067
27524
  } catch (err) {
27068
- return { success: false, fileType: "hwpx", error: err instanceof Error ? err.message : "HWPX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27525
+ return { success: false, fileType: "hwpx", error: sanitizeError(err), code: classifyError(err) };
27069
27526
  }
27070
27527
  }
27071
27528
  async function parseHwp(buffer, options) {
@@ -27090,13 +27547,13 @@ async function parseHwp(buffer, options) {
27090
27547
  }
27091
27548
  return { success: true, fileType: "hwp", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
27092
27549
  } catch (err) {
27093
- return { success: false, fileType: "hwp", error: err instanceof Error ? err.message : "HWP \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27550
+ return { success: false, fileType: "hwp", error: sanitizeError(err), code: classifyError(err) };
27094
27551
  }
27095
27552
  }
27096
27553
  async function parsePdf(buffer, options) {
27097
27554
  let parsePdfDocument;
27098
27555
  try {
27099
- const mod = await import("./parser-TCTCSBYZ.js");
27556
+ const mod = await import("./parser-RBZHA6SY.js");
27100
27557
  parsePdfDocument = mod.parsePdfDocument;
27101
27558
  } catch {
27102
27559
  return {
@@ -27111,7 +27568,7 @@ async function parsePdf(buffer, options) {
27111
27568
  return { success: true, fileType: "pdf", markdown, blocks, metadata, outline, warnings, isImageBased, pageQuality, qualitySummary, images };
27112
27569
  } catch (err) {
27113
27570
  const isImageBased = err instanceof Error && "isImageBased" in err ? true : void 0;
27114
- return { success: false, fileType: "pdf", error: err instanceof Error ? err.message : "PDF \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err), isImageBased };
27571
+ return { success: false, fileType: "pdf", error: sanitizeError(err), code: classifyError(err), isImageBased };
27115
27572
  }
27116
27573
  }
27117
27574
  async function parseXlsx(buffer, options) {
@@ -27119,7 +27576,7 @@ async function parseXlsx(buffer, options) {
27119
27576
  const { markdown, blocks, metadata, warnings } = await parseXlsxDocument(buffer, options);
27120
27577
  return { success: true, fileType: "xlsx", markdown, blocks, metadata, warnings };
27121
27578
  } catch (err) {
27122
- return { success: false, fileType: "xlsx", error: err instanceof Error ? err.message : "XLSX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27579
+ return { success: false, fileType: "xlsx", error: sanitizeError(err), code: classifyError(err) };
27123
27580
  }
27124
27581
  }
27125
27582
  async function parseXls(buffer, options) {
@@ -27127,7 +27584,7 @@ async function parseXls(buffer, options) {
27127
27584
  const { markdown, blocks, metadata, warnings } = await parseXlsDocument(buffer, options);
27128
27585
  return { success: true, fileType: "xls", markdown, blocks, metadata, warnings };
27129
27586
  } catch (err) {
27130
- return { success: false, fileType: "xls", error: err instanceof Error ? err.message : "XLS \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27587
+ return { success: false, fileType: "xls", error: sanitizeError(err), code: classifyError(err) };
27131
27588
  }
27132
27589
  }
27133
27590
  async function parseDocx(buffer, options) {
@@ -27135,7 +27592,7 @@ async function parseDocx(buffer, options) {
27135
27592
  const { markdown, blocks, metadata, outline, warnings, images } = await parseDocxDocument(buffer, options);
27136
27593
  return { success: true, fileType: "docx", markdown, blocks, metadata, outline, warnings, images: images?.length ? images : void 0 };
27137
27594
  } catch (err) {
27138
- return { success: false, fileType: "docx", error: err instanceof Error ? err.message : "DOCX \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27595
+ return { success: false, fileType: "docx", error: sanitizeError(err), code: classifyError(err) };
27139
27596
  }
27140
27597
  }
27141
27598
  async function parseHwpml(buffer, options) {
@@ -27143,7 +27600,7 @@ async function parseHwpml(buffer, options) {
27143
27600
  const { markdown, blocks, metadata, outline, warnings } = parseHwpmlDocument(buffer, options);
27144
27601
  return { success: true, fileType: "hwpml", markdown, blocks, metadata, outline, warnings };
27145
27602
  } catch (err) {
27146
- return { success: false, fileType: "hwpml", error: err instanceof Error ? err.message : "HWPML \uD30C\uC2F1 \uC2E4\uD328", code: classifyError(err) };
27603
+ return { success: false, fileType: "hwpml", error: sanitizeError(err), code: classifyError(err) };
27147
27604
  }
27148
27605
  }
27149
27606
  async function fillForm(input, values, outputFormat = "markdown") {
@@ -27229,4 +27686,4 @@ export {
27229
27686
  parseHwpml,
27230
27687
  fillForm
27231
27688
  };
27232
- //# sourceMappingURL=chunk-7S3M4N4E.js.map
27689
+ //# sourceMappingURL=chunk-IGGRLWT5.js.map